cyborg-hunter 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +110 -0
- package/CITATION.cff +29 -0
- package/LICENSE +21 -0
- package/README.md +78 -21
- package/package.json +10 -3
- package/src/cli/analyzers/edge-exit.js +4 -1
- package/src/cli/analyzers/phase-scope.js +83 -0
- package/src/cli/analyzers/summary.js +161 -28
- package/src/cli/analyzers/triage.js +59 -27
- package/src/cli/config.js +26 -1
- package/src/cli/ingest.js +623 -41
- package/src/cli/init.js +1 -1
- package/src/cli/renderers/event-log.js +18 -19
- package/src/cli/renderers/extensions.js +12 -3
- package/src/cli/renderers/html-index.js +163 -24
- package/src/cli/renderers/replay-assets.js +177 -0
- package/src/cli/renderers/replay-viewer.client.js +1022 -0
- package/src/cli/renderers/session-timeline.js +917 -0
- package/src/cli/renderers/summary-csv.js +5 -0
- package/src/cli/renderers/trajectories.js +68 -7
- package/src/cli/renderers/triage-md.js +10 -4
- package/src/cli/renderers/typing-profile.js +7 -1
- package/src/cli/report.js +42 -8
- package/src/core/monitor.js +60 -7
- package/src/core/scoring.js +11 -2
- package/src/core/signals/browser.js +51 -18
- package/src/core/signals/clipboard.js +10 -2
- package/src/core/signals/dom-protection.js +9 -0
- package/src/core/signals/focus.js +16 -2
- package/src/jspsych/extension-cyborg-hunter-replay.js +135 -0
- package/src/jspsych/extension-cyborg-hunter.js +9 -2
- package/src/jspsych/extension-guard-friction.js +32 -12
- package/src/jspsych/extension-guard-honeypot.js +25 -1
- package/src/replay/capture-dom.js +575 -0
- package/src/replay/capture-trace.js +468 -0
- package/src/replay/index.js +104 -0
- package/src/replay/persistence.js +141 -0
- package/src/replay/recorder.js +315 -0
- package/src/replay/serializer.js +119 -0
- package/src/shared/constants.js +12 -6
- package/src/shared/schema.js +5 -0
- package/src/shared/validation.js +55 -0
- package/dist/cyborg-hunter.esm.js +0 -1527
- package/dist/cyborg-hunter.min.js +0 -6
- package/dist/extension-cyborg-hunter.js +0 -1
- package/dist/extension-guard-friction.js +0 -36
- package/dist/extension-guard-honeypot.js +0 -1
- package/src/cli/renderers/tab-timeline.js +0 -149
package/src/cli/ingest.js
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
// 1. JSON Shape 1 — { participantId: 'P1', trials: [{ integrity: {...} }] }
|
|
5
5
|
// jsPsych extension data with one file per participant.
|
|
6
6
|
// 2. JSON Shape 2 — { metadata: {...}, responses: [{ mouseTrack, tabAwayEvents, ... }] }
|
|
7
|
-
// The original
|
|
7
|
+
// The original legacy format. Signal data lives flat on each response
|
|
8
8
|
// (no `integrity` wrapper). Pre-dates the standalone library.
|
|
9
9
|
// 3. CSV — jsPsych default save format. Each row is a trial; nested objects
|
|
10
10
|
// (integrity, integritySession, integrityScore) are JSON-stringified into
|
|
@@ -12,19 +12,70 @@
|
|
|
12
12
|
|
|
13
13
|
import { readFileSync, readdirSync } from 'fs';
|
|
14
14
|
import { join, extname } from 'path';
|
|
15
|
+
import { gunzipSync } from 'zlib';
|
|
15
16
|
import Papa from 'papaparse';
|
|
16
17
|
import { TRIAL_REPORT_FIELDS } from '../shared/schema.js';
|
|
18
|
+
import { sanitizeId } from '../shared/constants.js';
|
|
19
|
+
|
|
20
|
+
// Replay artifacts saved by the replay extension:
|
|
21
|
+
// <sanitizedPid>-replay-<sessionStartEpochMs>.json[.gz]
|
|
22
|
+
// They sit in dataDir (or replayDir) next to the participant files and must
|
|
23
|
+
// never enter the participant-file pass.
|
|
24
|
+
const REPLAY_FILE_RE = /-replay-\d+\.json(\.gz)?$/i;
|
|
25
|
+
|
|
26
|
+
// Content sniff: replay artifacts (ours or #3661's) are identified by
|
|
27
|
+
// structure, not just filename — schema_version plus either our recorder
|
|
28
|
+
// stamp or a #3661-shaped trials array.
|
|
29
|
+
function looksLikeReplayArtifact(text) {
|
|
30
|
+
try {
|
|
31
|
+
const j = JSON.parse(text);
|
|
32
|
+
return j && typeof j === 'object' && 'schema_version' in j &&
|
|
33
|
+
(String(j.metadata?.recorder || '').startsWith('cyborg-hunter-replay') ||
|
|
34
|
+
(Array.isArray(j.trials) && j.trials.length > 0 &&
|
|
35
|
+
j.trials.every(t => t && 'events' in t && 'initial_dom' in t)));
|
|
36
|
+
} catch (e) {
|
|
37
|
+
return true; // unparseable + replay-named → let the replay pass report it
|
|
38
|
+
}
|
|
39
|
+
}
|
|
17
40
|
|
|
18
41
|
export async function ingest(config) {
|
|
19
|
-
const
|
|
42
|
+
const allFiles = findFiles(config.dataDir, config.filePattern);
|
|
20
43
|
const participants = [];
|
|
21
44
|
const warnings = [];
|
|
22
45
|
|
|
46
|
+
// Replay artifacts are excluded from the participant pass by filename —
|
|
47
|
+
// but only after a content check, so a participant export that happens to
|
|
48
|
+
// match the naming pattern is never silently dropped.
|
|
49
|
+
const files = [];
|
|
50
|
+
for (const file of allFiles) {
|
|
51
|
+
if (REPLAY_FILE_RE.test(file)) {
|
|
52
|
+
let text = null;
|
|
53
|
+
try { text = readFileSync(file, 'utf8'); } catch (e) { text = null; }
|
|
54
|
+
let parseable = true;
|
|
55
|
+
if (text !== null) {
|
|
56
|
+
try { JSON.parse(text); } catch (e) { parseable = false; }
|
|
57
|
+
}
|
|
58
|
+
if (text === null || !parseable) {
|
|
59
|
+
// Never let a replay-named file vanish silently: if its pid maps to
|
|
60
|
+
// a discovered participant the attach pass warns again with more
|
|
61
|
+
// context, but an orphan (no matching participant) would otherwise
|
|
62
|
+
// disappear without a trace.
|
|
63
|
+
warnings.push({ file,
|
|
64
|
+
warnings: ['Replay-named file could not be parsed (truncated upload or a misnamed participant export?) — skipped from the participant pass; if a matching participant exists, the replay pass reports it too.'] });
|
|
65
|
+
continue;
|
|
66
|
+
}
|
|
67
|
+
if (looksLikeReplayArtifact(text)) continue;
|
|
68
|
+
warnings.push({ file,
|
|
69
|
+
warnings: ['File matches the replay-artifact naming pattern (<pid>-replay-<epoch>.json) but contains participant data — parsed as a participant file. Consider renaming it to avoid ambiguity.'] });
|
|
70
|
+
}
|
|
71
|
+
files.push(file);
|
|
72
|
+
}
|
|
73
|
+
|
|
23
74
|
for (const file of files) {
|
|
24
75
|
try {
|
|
25
76
|
const text = readFileSync(file, 'utf8');
|
|
26
77
|
// Branch by extension. CSV is jsPsych's default save format; JSON is what
|
|
27
|
-
//
|
|
78
|
+
// server-side-saving experiments use.
|
|
28
79
|
const raw = extname(file).toLowerCase() === '.csv'
|
|
29
80
|
? parseCsvToRaw(text, config)
|
|
30
81
|
: JSON.parse(text);
|
|
@@ -46,9 +97,223 @@ export async function ingest(config) {
|
|
|
46
97
|
}
|
|
47
98
|
}
|
|
48
99
|
|
|
100
|
+
// Duplicate-upload guard. Two files resolving to the same participantId are
|
|
101
|
+
// conflated downstream (triage/HTML/image outputs key by id), so the second
|
|
102
|
+
// upload's evidence can silently overwrite or vanish. We do NOT auto-dedup —
|
|
103
|
+
// the analyst must decide which upload is canonical — but we surface it.
|
|
104
|
+
// (attachReplayArtifacts below adds its own duplicate-id note describing the
|
|
105
|
+
// replay-association consequence specifically.)
|
|
106
|
+
const idCounts = new Map();
|
|
107
|
+
for (const p of participants) idCounts.set(p.participantId, (idCounts.get(p.participantId) || 0) + 1);
|
|
108
|
+
for (const [id, n] of idCounts) {
|
|
109
|
+
if (n > 1) {
|
|
110
|
+
warnings.push({ file: '(multiple)', warnings: [
|
|
111
|
+
`duplicate participantId "${id}" appears in ${n} files — downstream ` +
|
|
112
|
+
`outputs key by id, so entries may be conflated. Keep one upload per participant.`
|
|
113
|
+
] });
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
attachReplayArtifacts(participants, config, warnings);
|
|
118
|
+
|
|
49
119
|
return { participants, warnings };
|
|
50
120
|
}
|
|
51
121
|
|
|
122
|
+
// Finds and attaches each participant's replay artifact (if any) as
|
|
123
|
+
// `participant.replay`:
|
|
124
|
+
// { recording, file, meta } — parsed and attached
|
|
125
|
+
// { error: 'parse_failed', reason } — artifact exists but unreadable
|
|
126
|
+
// null — no artifact (silent unless meta
|
|
127
|
+
// says one went to 'download')
|
|
128
|
+
function attachReplayArtifacts(participants, config, warnings) {
|
|
129
|
+
const dir = config.replayDir || config.dataDir;
|
|
130
|
+
let entries = [];
|
|
131
|
+
try {
|
|
132
|
+
entries = readdirSync(dir);
|
|
133
|
+
} catch (e) {
|
|
134
|
+
if (config.replayDir) {
|
|
135
|
+
warnings.push({ file: dir, warnings: [`replayDir not readable: ${e.message}`] });
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
// Sanitized-name census: filename sanitization is many-to-one, so a
|
|
140
|
+
// no-embedded-id artifact may only attach when exactly one participant
|
|
141
|
+
// maps to its sanitized name (otherwise ownership is ambiguous).
|
|
142
|
+
// The census is keyed on LOWERCASED sanitized ids because the filename
|
|
143
|
+
// match below is case-insensitive (macOS filesystems are) — both
|
|
144
|
+
// mechanisms must share the same equivalence classes or an ownerless
|
|
145
|
+
// artifact could attach to two case-variant participants at once.
|
|
146
|
+
// Null-prototype maps: a participant id that collides with an
|
|
147
|
+
// Object.prototype key ("__proto__", "constructor") must count like any
|
|
148
|
+
// other id — on a literal {}, assigning a primitive to __proto__ is a
|
|
149
|
+
// silent no-op, which skipped the duplicate warning AND the
|
|
150
|
+
// ambiguous-association guard below.
|
|
151
|
+
const sanitize = sanitizeId;
|
|
152
|
+
const saneCounts = Object.create(null);
|
|
153
|
+
const idCounts = Object.create(null);
|
|
154
|
+
for (const p of participants) {
|
|
155
|
+
const s = sanitize(p.participantId).toLowerCase();
|
|
156
|
+
saneCounts[s] = (saneCounts[s] || 0) + 1;
|
|
157
|
+
idCounts[p.participantId] = (idCounts[p.participantId] || 0) + 1;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// Duplicate participant ids (repeat runs, duplicate exports) are outside
|
|
161
|
+
// the pipeline's data model — every renderer keys outputs by pid, so the
|
|
162
|
+
// whole report already treats them as one person. Replay attachment
|
|
163
|
+
// follows the same semantics (both records get the same latest artifact);
|
|
164
|
+
// say so once per duplicated id instead of silently doing it.
|
|
165
|
+
for (const [id, n] of Object.entries(idCounts)) {
|
|
166
|
+
if (n > 1) {
|
|
167
|
+
warnings.push({ file: dir,
|
|
168
|
+
warnings: [`Duplicate participant id "${id}" across ${n} files — the report (including replay attachment) treats these as one person; per-session replay association is not attempted.`] });
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
for (const p of participants) {
|
|
173
|
+
// Same sanitization the browser-side filename builder applies. The
|
|
174
|
+
// match is ANCHORED (^<sane>-replay-<digits>.json$): a bare prefix
|
|
175
|
+
// would let participant "a" swallow "a-replay-replay-<epoch>.json",
|
|
176
|
+
// which belongs to participant "a-replay".
|
|
177
|
+
const sane = sanitize(p.participantId);
|
|
178
|
+
const escaped = sane.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
179
|
+
const exactRe = new RegExp(`^${escaped}-replay-\\d+\\.json(\\.gz)?$`, 'i');
|
|
180
|
+
const mine = entries.filter(f => exactRe.test(f));
|
|
181
|
+
// The meta pointer rides on every trial row via addProperties.
|
|
182
|
+
const meta = (p.trials && p.trials[0] && p.trials[0].integrityReplayMeta) || null;
|
|
183
|
+
// Replay finalize failures ride the same way — surface them where the
|
|
184
|
+
// analyst looks (they mean the artifact was probably never saved).
|
|
185
|
+
const finErr = p.trials && p.trials[0] && p.trials[0].replayFinalizeError;
|
|
186
|
+
if (finErr) {
|
|
187
|
+
warnings.push({ file: dir,
|
|
188
|
+
warnings: [`Replay finalize failed for ${p.participantId}: ${finErr} — the artifact was likely never saved.`] });
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
if (mine.length === 0) {
|
|
192
|
+
p.replay = null;
|
|
193
|
+
if (meta && meta.saved_to === 'download') {
|
|
194
|
+
warnings.push({
|
|
195
|
+
file: dir,
|
|
196
|
+
warnings: [`Replay artifact for ${p.participantId} was downloaded to the participant's machine (autoSave mode "download") and is not recoverable from here — check the autosave configuration for future runs.`]
|
|
197
|
+
});
|
|
198
|
+
}
|
|
199
|
+
continue;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
const parsed = [];
|
|
203
|
+
for (const f of mine) {
|
|
204
|
+
try {
|
|
205
|
+
const buf = readFileSync(join(dir, f));
|
|
206
|
+
const json = f.toLowerCase().endsWith('.gz')
|
|
207
|
+
? gunzipSync(buf).toString('utf8')
|
|
208
|
+
: buf.toString('utf8');
|
|
209
|
+
// Same structural sniff as the participant pass: a misnamed
|
|
210
|
+
// participant export was rescued as participant data there and
|
|
211
|
+
// must not double as its own "replay" here.
|
|
212
|
+
if (!looksLikeReplayArtifact(json)) continue;
|
|
213
|
+
parsed.push({ file: f, recording: JSON.parse(json) });
|
|
214
|
+
} catch (e) {
|
|
215
|
+
parsed.push({ file: f, reason: e.message });
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
if (parsed.length === 0) {
|
|
219
|
+
p.replay = null;
|
|
220
|
+
continue;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
// Every unreadable artifact warns individually — a corrupt NEWEST
|
|
224
|
+
// session must never be silently masked by an older readable one.
|
|
225
|
+
for (const bad of parsed.filter(x => !x.recording)) {
|
|
226
|
+
warnings.push({ file: join(dir, bad.file),
|
|
227
|
+
warnings: [`Replay artifact ${bad.file} unreadable: ${bad.reason} — if this is the newest session, its replay is lost.`] });
|
|
228
|
+
}
|
|
229
|
+
const readable = parsed.filter(x => x.recording);
|
|
230
|
+
if (readable.length === 0) {
|
|
231
|
+
p.replay = { error: 'parse_failed', reason: parsed[0].reason, file: parsed[0].file };
|
|
232
|
+
continue;
|
|
233
|
+
}
|
|
234
|
+
if (mine.length > 1) {
|
|
235
|
+
warnings.push({ file: dir,
|
|
236
|
+
warnings: [`Multiple replay artifacts for ${p.participantId} (page reload?) — using the latest by start_time.`] });
|
|
237
|
+
}
|
|
238
|
+
// Filename sanitization is many-to-one ('a/b' and 'a_b' both map to
|
|
239
|
+
// 'a_b'), so ownership is verified against the UNsanitized
|
|
240
|
+
// participant_id embedded in the recording. Artifacts without one
|
|
241
|
+
// (e.g. plain #3661 recordings) attach with a soft warning.
|
|
242
|
+
const owned = [];
|
|
243
|
+
for (const cand of readable) {
|
|
244
|
+
const embedded = cand.recording.metadata?.participant_id;
|
|
245
|
+
if (embedded == null) {
|
|
246
|
+
// Ownerless artifacts skip id verification entirely, so the filename
|
|
247
|
+
// must match EXACT-case (our recorder writes sanitize(pid) verbatim).
|
|
248
|
+
// Case-tolerant matching stays for discovery, where the embedded-id
|
|
249
|
+
// check catches cross-case impostors.
|
|
250
|
+
if (!cand.file.startsWith(sane + '-replay-')) {
|
|
251
|
+
warnings.push({ file: join(dir, cand.file),
|
|
252
|
+
warnings: [`Replay artifact has no embedded participant_id and its filename case does not match "${sane}" exactly — not attached.`] });
|
|
253
|
+
} else if (saneCounts[sane.toLowerCase()] > 1) {
|
|
254
|
+
warnings.push({ file: join(dir, cand.file),
|
|
255
|
+
warnings: [`Replay artifact has no embedded participant_id and its filename is ambiguous (${saneCounts[sane.toLowerCase()]} participants sanitize to "${sane}") — not attached to anyone.`] });
|
|
256
|
+
} else {
|
|
257
|
+
warnings.push({ file: join(dir, cand.file),
|
|
258
|
+
warnings: [`Replay artifact has no embedded participant_id — cannot verify ownership; attaching to ${p.participantId} by unique filename match.`] });
|
|
259
|
+
owned.push(cand);
|
|
260
|
+
}
|
|
261
|
+
} else if (String(embedded) === String(p.participantId)) {
|
|
262
|
+
owned.push(cand);
|
|
263
|
+
} else {
|
|
264
|
+
warnings.push({ file: join(dir, cand.file),
|
|
265
|
+
warnings: [`Replay artifact participant_id mismatch: file matches ${p.participantId} by name but was recorded for ${embedded} (sanitization collision?) — not attached.`] });
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
if (owned.length === 0) {
|
|
269
|
+
p.replay = null;
|
|
270
|
+
continue;
|
|
271
|
+
}
|
|
272
|
+
// Duplicate records for this id + multiple owned artifacts: the
|
|
273
|
+
// per-session mapping is genuinely ambiguous. Attach nothing rather
|
|
274
|
+
// than knowingly mis-associate a session's replay.
|
|
275
|
+
if (idCounts[p.participantId] > 1 && owned.length > 1) {
|
|
276
|
+
warnings.push({ file: dir,
|
|
277
|
+
warnings: [`Cannot associate ${owned.length} replay artifacts with ${idCounts[p.participantId]} duplicate records of "${p.participantId}" — none attached. Separate the sessions into distinct data dirs (or ids) to view their replays.`] });
|
|
278
|
+
p.replay = null;
|
|
279
|
+
continue;
|
|
280
|
+
}
|
|
281
|
+
// Latest-session pick tolerates non-ISO start_time in third-party
|
|
282
|
+
// artifacts: ISO string → numeric epoch → the filename's own epoch.
|
|
283
|
+
const sessionEpoch = (cand) => {
|
|
284
|
+
const v = cand.recording.metadata?.start_time;
|
|
285
|
+
const n = typeof v === 'number' ? v : Date.parse(v);
|
|
286
|
+
if (Number.isFinite(n)) return n;
|
|
287
|
+
const m = cand.file.match(/-replay-(\d+)\.json/i);
|
|
288
|
+
return m ? Number(m[1]) : 0;
|
|
289
|
+
};
|
|
290
|
+
owned.sort((a, b) => sessionEpoch(a) - sessionEpoch(b));
|
|
291
|
+
const chosen = owned[owned.length - 1];
|
|
292
|
+
if (chosen.recording.schema_version !== 1) {
|
|
293
|
+
warnings.push({ file: join(dir, chosen.file),
|
|
294
|
+
warnings: [`Replay schema_version ${chosen.recording.schema_version} (this CLI targets 1) — attaching anyway; the viewer may degrade.`] });
|
|
295
|
+
}
|
|
296
|
+
p.replay = { recording: chosen.recording, file: chosen.file, meta };
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
// Resolves a possibly-dotted field path against an object (0.6.1).
|
|
301
|
+
// A flat key wins over a dotted walk, so data that literally contains a
|
|
302
|
+
// "metadata.sessionId" column stays addressable; otherwise the path is
|
|
303
|
+
// walked one segment at a time. Returns undefined when any segment is
|
|
304
|
+
// missing or a non-object is hit mid-path.
|
|
305
|
+
export function getByPath(obj, path) {
|
|
306
|
+
if (obj == null || typeof path !== 'string' || path.length === 0) return undefined;
|
|
307
|
+
if (Object.prototype.hasOwnProperty.call(obj, path)) return obj[path];
|
|
308
|
+
if (!path.includes('.')) return undefined;
|
|
309
|
+
let cur = obj;
|
|
310
|
+
for (const seg of path.split('.')) {
|
|
311
|
+
if (cur == null || typeof cur !== 'object') return undefined;
|
|
312
|
+
cur = cur[seg];
|
|
313
|
+
}
|
|
314
|
+
return cur;
|
|
315
|
+
}
|
|
316
|
+
|
|
52
317
|
// Extracts integrity trial data from a single participant's raw JSON.
|
|
53
318
|
// Returns { participantId, trials, warnings, metadata }.
|
|
54
319
|
export function extractIntegrityData(raw, config) {
|
|
@@ -56,39 +321,94 @@ export function extractIntegrityData(raw, config) {
|
|
|
56
321
|
const pidField = config.participantIdField || 'participantId';
|
|
57
322
|
const intField = config.integrityField || 'integrity';
|
|
58
323
|
|
|
59
|
-
// Determine participant ID — check top level, then metadata sub-object
|
|
60
|
-
|
|
324
|
+
// Determine participant ID — check top level, then metadata sub-object.
|
|
325
|
+
// Since 0.6.1 the field supports dot-paths ("metadata.sessionId"); plain
|
|
326
|
+
// names keep the historical top-level → metadata fallback. `||` (not `??`)
|
|
327
|
+
// preserves the pre-0.6.1 treatment of empty-string IDs as missing.
|
|
328
|
+
const participantId = getByPath(raw, pidField) || getByPath(raw.metadata, pidField) || 'unknown';
|
|
329
|
+
// An id field that resolves to nothing yields 'unknown'. Silently, that both
|
|
330
|
+
// loses the real id AND collides every such file under one 'unknown' bucket
|
|
331
|
+
// downstream (see the duplicate-id check in ingest()). Warn so a mistyped
|
|
332
|
+
// participantIdField is visible instead of producing an all-'unknown' cohort.
|
|
333
|
+
if (participantId === 'unknown') {
|
|
334
|
+
warnings.push(
|
|
335
|
+
`participantId unresolved (field "${pidField}" not found at top level or in ` +
|
|
336
|
+
`metadata) — defaulted to "unknown". Check participantIdField.`
|
|
337
|
+
);
|
|
338
|
+
}
|
|
61
339
|
|
|
62
340
|
let trials = [];
|
|
63
341
|
|
|
64
342
|
// Shape 1: { trials: [{ integrity: {...} }, ...] } — jsPsych extension data.
|
|
65
343
|
// Each trial has an `integrity` sub-object added by CyborgHunter.endTrial().
|
|
344
|
+
// We merge the trial's own fields (ruleId, timestamp, rt, rulePosition, etc.)
|
|
345
|
+
// with the integrity sub-object's fields (tabAwayEvents, copyEvents, etc.).
|
|
346
|
+
// On a key collision the integrity sub-object wins, since it's the more
|
|
347
|
+
// authoritative source for those fields. This preserves both the renderer-
|
|
348
|
+
// needed experiment metadata AND the cyborg-hunter signal data on the same
|
|
349
|
+
// trial object.
|
|
66
350
|
if (Array.isArray(raw.trials)) {
|
|
67
351
|
trials = raw.trials
|
|
68
|
-
.filter(t => t[intField])
|
|
69
|
-
.map(t => t[intField]);
|
|
352
|
+
.filter(t => t && t[intField])
|
|
353
|
+
.map(t => ({ ...t, ...t[intField] }));
|
|
70
354
|
// jsPsych extension data carries trialStart_perfNow per trial (set by the
|
|
71
355
|
// wrapper's on_load), so normalization takes the exact-subtraction fast
|
|
72
356
|
// path. Without it, renderers see only session-absolute `start` values
|
|
73
357
|
// and plot tab-away markers far off the per-trial axis.
|
|
74
358
|
normalizeTabAwayTimestamps(trials, raw);
|
|
75
359
|
}
|
|
76
|
-
|
|
77
|
-
//
|
|
78
|
-
//
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
360
|
+
|
|
361
|
+
// Phase-trial extension: per-phase integrity reports for
|
|
362
|
+
// gallery / post-gallery-query / end-requery phases live on raw.phaseTrials,
|
|
363
|
+
// mirroring the per-classification-trial shape so the renderer can place
|
|
364
|
+
// gallery mouse trajectories and tab-away events on the same axes.
|
|
365
|
+
//
|
|
366
|
+
// After merging, sort by (rulePosition, phase-rank, trialNumber) so the
|
|
367
|
+
// trajectories grid shows trials in the chronological order each rule was
|
|
368
|
+
// actually experienced: gallery → post-gallery query → classification t1..t6.
|
|
369
|
+
// End-requery trials carry rulePosition=null and sort to the very end of
|
|
370
|
+
// the array, matching their session-end timing.
|
|
371
|
+
if (Array.isArray(raw.phaseTrials) && raw.phaseTrials.length > 0) {
|
|
372
|
+
const phaseTrials = raw.phaseTrials
|
|
373
|
+
.filter(t => t && t[intField])
|
|
374
|
+
.map(t => ({ ...t, ...t[intField] }));
|
|
375
|
+
normalizeTabAwayTimestamps(phaseTrials, raw);
|
|
376
|
+
trials = trials.concat(phaseTrials);
|
|
377
|
+
trials.sort(ruleChronologicalCompare);
|
|
88
378
|
}
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
379
|
+
|
|
380
|
+
// Shapes 2 and 3 are ALTERNATIVE top-level layouts, tried only when the
|
|
381
|
+
// Shape-1 trials/phaseTrials path produced nothing. Previously the Shape-2
|
|
382
|
+
// branch was chained as `else if` off the phaseTrials `if`, so a payload
|
|
383
|
+
// carrying both `trials` (real integrity) and `responses` — or even a stray
|
|
384
|
+
// empty `responses: []` — had its already-extracted integrity trials clobbered
|
|
385
|
+
// (or the participant silently dropped). Gate on trials.length so Shape-1 wins.
|
|
386
|
+
if (trials.length === 0) {
|
|
387
|
+
// Shape 2: { responses: [{ mouseTrack, tabAwayEvents, ... }] } (legacy).
|
|
388
|
+
// Signal data lives directly on the response — no integrity wrapper. We apply
|
|
389
|
+
// field name mapping (mouseTrack → mouseEvents).
|
|
390
|
+
if (Array.isArray(raw.responses)) {
|
|
391
|
+
trials = raw.responses.map((r, i) => mapLegacyFields({
|
|
392
|
+
...r,
|
|
393
|
+
_sourceIndex: i
|
|
394
|
+
}));
|
|
395
|
+
// Normalize tab-away timestamps from session-relative performance.now()
|
|
396
|
+
// to trial-relative milliseconds so renderers can plot them on the
|
|
397
|
+
// same x-axis as mouseEvents[].t (which is already trial-relative).
|
|
398
|
+
normalizeTabAwayTimestamps(trials, raw);
|
|
399
|
+
}
|
|
400
|
+
// Shape 3: Top-level array of trials. Same merge contract as Shape 1:
|
|
401
|
+
// outer trial fields survive, integrity wins on collision, and tab-away
|
|
402
|
+
// timestamps get normalized. The spread builds OUR copy, so the
|
|
403
|
+
// array-field coercion below mutates that copy — never a caller-owned or
|
|
404
|
+
// frozen object (a frozen one would throw instead of coercing). The
|
|
405
|
+
// outer-merge also keeps replay pointers (integrityReplayMeta /
|
|
406
|
+
// replayFinalizeError ride the outer trial rows) visible to
|
|
407
|
+
// attachReplayArtifacts.
|
|
408
|
+
else if (Array.isArray(raw)) {
|
|
409
|
+
trials = raw.filter(t => t && t[intField]).map(t => ({ ...t, ...t[intField] }));
|
|
410
|
+
normalizeTabAwayTimestamps(trials, raw);
|
|
411
|
+
}
|
|
92
412
|
}
|
|
93
413
|
|
|
94
414
|
if (trials.length === 0) {
|
|
@@ -96,53 +416,312 @@ export function extractIntegrityData(raw, config) {
|
|
|
96
416
|
}
|
|
97
417
|
|
|
98
418
|
// Validate each trial has the minimum required fields from the schema.
|
|
99
|
-
// Missing fields get a warning but don't prevent analysis.
|
|
419
|
+
// Missing fields get a warning but don't prevent analysis. Array-typed signal
|
|
420
|
+
// fields that arrive as a non-array (e.g. a hand-edited `pasteEvents: {}`) are
|
|
421
|
+
// coerced to [] — otherwise a downstream `for (const e of trial.pasteEvents)`
|
|
422
|
+
// throws "not iterable" and, because report.js has no per-renderer boundary,
|
|
423
|
+
// one malformed file aborts the ENTIRE report. Coercing here keeps the run
|
|
424
|
+
// alive and localizes the damage to a warning on that trial.
|
|
100
425
|
for (const trial of trials) {
|
|
101
426
|
const missing = [];
|
|
102
427
|
for (const [field, spec] of Object.entries(TRIAL_REPORT_FIELDS)) {
|
|
103
428
|
if (spec.required && trial[field] === undefined) {
|
|
104
429
|
missing.push(field);
|
|
105
430
|
}
|
|
431
|
+
if (spec.type === 'array' && trial[field] != null && !Array.isArray(trial[field])) {
|
|
432
|
+
warnings.push(
|
|
433
|
+
`Trial ${trial.trialId ?? '?'}: field "${field}" is not an array ` +
|
|
434
|
+
`(got ${typeof trial[field]}) — coerced to [] to keep analysis running.`
|
|
435
|
+
);
|
|
436
|
+
trial[field] = [];
|
|
437
|
+
}
|
|
106
438
|
}
|
|
107
439
|
if (missing.length > 0) {
|
|
108
440
|
warnings.push(`Trial ${trial.trialId || '?'}: missing fields: ${missing.join(', ')}`);
|
|
109
441
|
}
|
|
110
442
|
}
|
|
111
443
|
|
|
112
|
-
const { session, score } = findSessionData(raw);
|
|
444
|
+
const { session, score } = findSessionData(raw, config);
|
|
113
445
|
if (session === null && trials.length > 0) {
|
|
114
446
|
warnings.push('No session-level integrity data — some signals unavailable (did the experiment call getSessionReport()?)');
|
|
115
447
|
}
|
|
116
448
|
|
|
117
|
-
|
|
449
|
+
// The jsPsych adapter drops a `cyborgHunterFinalizeError` marker (via
|
|
450
|
+
// addProperties) when finalize() throws, so analysts can tell a missing-session
|
|
451
|
+
// run apart from a genuine finalize() failure. Surface it as a warning instead
|
|
452
|
+
// of leaving it dead in the data behind a generic "no session data" message.
|
|
453
|
+
const finalizeError = raw.cyborgHunterFinalizeError
|
|
454
|
+
?? raw.metadata?.cyborgHunterFinalizeError
|
|
455
|
+
?? (Array.isArray(raw.trials)
|
|
456
|
+
? raw.trials.find(t => t?.cyborgHunterFinalizeError)?.cyborgHunterFinalizeError
|
|
457
|
+
: undefined);
|
|
458
|
+
if (finalizeError) {
|
|
459
|
+
warnings.push(`finalize() failed for this participant (cyborgHunterFinalizeError): ${finalizeError} — session data may be incomplete`);
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
return {
|
|
463
|
+
participantId,
|
|
464
|
+
trials,
|
|
465
|
+
warnings,
|
|
466
|
+
metadata: raw.metadata || {},
|
|
467
|
+
session,
|
|
468
|
+
score,
|
|
469
|
+
// Surface a few top-level payload fields that some renderers need but
|
|
470
|
+
// that aren't part of the session-level integrity object. Keeping the
|
|
471
|
+
// list explicit (rather than exposing `raw` wholesale) avoids future
|
|
472
|
+
// renderers silently coupling to payload internals.
|
|
473
|
+
galleryStudyMs: Array.isArray(raw.galleryStudyMs) ? raw.galleryStudyMs : null,
|
|
474
|
+
postGalleryGuesses: Array.isArray(raw.postGalleryGuesses) ? raw.postGalleryGuesses : null,
|
|
475
|
+
// App-written top-level guardFriction wins; otherwise synthesize the guard
|
|
476
|
+
// lane from the honeypot's session violation log (which the shipped guard
|
|
477
|
+
// extensions actually emit) so the timeline renders for library-only data.
|
|
478
|
+
guardFriction: raw.guardFriction ?? (() => {
|
|
479
|
+
const v = findGuardViolations(raw);
|
|
480
|
+
return v ? { violations: v } : null;
|
|
481
|
+
})(),
|
|
482
|
+
// Guard-honeypot self-disclosure (visible bait). null when the honeypot
|
|
483
|
+
// extension was not used.
|
|
484
|
+
honeypot: findHoneypotDisclosure(raw),
|
|
485
|
+
};
|
|
118
486
|
}
|
|
119
487
|
|
|
120
|
-
//
|
|
121
|
-
//
|
|
488
|
+
// True when obj plausibly IS a getSessionReport() output — i.e. it carries at
|
|
489
|
+
// least one of the well-known top-level session-report keys (see
|
|
490
|
+
// src/core/monitor.js sessionData / getSessionReport()). Guards the
|
|
491
|
+
// analyst-supplied sessionIntegrityPath below: `typeof === 'object'` alone
|
|
492
|
+
// accepts any object the path happens to resolve to, including a near-miss
|
|
493
|
+
// wrapper one level up the tree (e.g. `metadata` instead of
|
|
494
|
+
// `metadata.integritySession`), which would otherwise silently zero out every
|
|
495
|
+
// downstream signal instead of falling through.
|
|
496
|
+
function looksLikeSessionData(obj) {
|
|
497
|
+
if (!obj || typeof obj !== 'object') return false;
|
|
498
|
+
return ['tabAwaySums', 'hardScore', 'softScore', 'anyHardTriggered', 'trialsCompleted']
|
|
499
|
+
.some(key => Object.prototype.hasOwnProperty.call(obj, key));
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
// Locates session-level integrity data. An analyst-supplied dotted path
|
|
503
|
+
// (config.sessionIntegrityPath, 0.6.1) is checked first; then the built-in
|
|
504
|
+
// conventions, in priority order:
|
|
505
|
+
// 1. raw.metadata.integritySession / integrityScore — the metadata convention.
|
|
122
506
|
// 2. Last trial's integritySession / integrityScore — jsPsych addDataToLastTrial (Option A).
|
|
123
507
|
// 3. Any trial's integritySession — fallback (Option B).
|
|
508
|
+
// 4. raw.cyborgHunter — native top-level location used by raw-DOM
|
|
509
|
+
// adopters before they adopt the
|
|
510
|
+
// metadata.integritySession mirror. Added 2026-05-26.
|
|
124
511
|
// Returns { session, score }, both null if not found.
|
|
125
|
-
function findSessionData(raw) {
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
512
|
+
function findSessionData(raw, config) {
|
|
513
|
+
let session = null;
|
|
514
|
+
let score = null;
|
|
515
|
+
|
|
516
|
+
// 0. Analyst-specified location, e.g. "payload.cyborgHunter" for pipelines
|
|
517
|
+
// that nest the getSessionReport() output somewhere non-standard. Falls
|
|
518
|
+
// through to the built-in conventions when the path resolves to nothing
|
|
519
|
+
// OR to something that doesn't look like a session report (looksLikeSessionData,
|
|
520
|
+
// 0.6.1 — a malformed/near-miss path used to be accepted on typeof alone),
|
|
521
|
+
// so a partially-migrated cohort still ingests. The score is synthesized
|
|
522
|
+
// from the session object below (getSessionReport() embeds it).
|
|
523
|
+
const customSession = config?.sessionIntegrityPath
|
|
524
|
+
? getByPath(raw, config.sessionIntegrityPath) : null;
|
|
525
|
+
if (looksLikeSessionData(customSession)) {
|
|
526
|
+
session = customSession;
|
|
527
|
+
}
|
|
528
|
+
// 1. metadata convention
|
|
529
|
+
else if (raw.metadata?.integritySession) {
|
|
530
|
+
session = raw.metadata.integritySession;
|
|
531
|
+
score = raw.metadata.integrityScore || null;
|
|
129
532
|
}
|
|
130
533
|
// 2. jsPsych addDataToLastTrial convention (Option A)
|
|
131
|
-
if (Array.isArray(raw.trials) && raw.trials.length > 0
|
|
534
|
+
else if (Array.isArray(raw.trials) && raw.trials.length > 0 &&
|
|
535
|
+
raw.trials[raw.trials.length - 1]?.integritySession) {
|
|
132
536
|
const last = raw.trials[raw.trials.length - 1];
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
}
|
|
537
|
+
session = last.integritySession;
|
|
538
|
+
score = last.integrityScore || null;
|
|
136
539
|
}
|
|
137
540
|
// 3. Any-trial fallback
|
|
138
|
-
if (Array.isArray(raw.trials)) {
|
|
541
|
+
else if (Array.isArray(raw.trials) && raw.trials.find(x => x?.integritySession)) {
|
|
139
542
|
const t = raw.trials.find(x => x?.integritySession);
|
|
140
|
-
|
|
543
|
+
session = t.integritySession;
|
|
544
|
+
score = t.integrityScore || null;
|
|
141
545
|
}
|
|
142
|
-
|
|
546
|
+
// 4. Native top-level location. An early raw-DOM adopter app
|
|
547
|
+
// writes the getSessionReport() output directly to `raw.cyborgHunter`.
|
|
548
|
+
// Sessions saved before the metadata.integritySession mirror landed
|
|
549
|
+
// (2026-05-24) have ONLY this top-level field. Without
|
|
550
|
+
// this fallback, the renderer can't find windowPositions and falls back
|
|
551
|
+
// to stale top-level metadata.windowWidth — visible as a misaligned
|
|
552
|
+
// "browser window" dashed rectangle in the trajectory PNGs.
|
|
553
|
+
else if (raw.cyborgHunter && typeof raw.cyborgHunter === 'object') {
|
|
554
|
+
session = raw.cyborgHunter;
|
|
555
|
+
score = null;
|
|
556
|
+
}
|
|
557
|
+
|
|
558
|
+
// When no separate integrityScore blob was saved, synthesize the authoritative
|
|
559
|
+
// score from the session object itself. getSessionReport() embeds the scoring
|
|
560
|
+
// summary (hardScore/softScore/anyHardTriggered/softScoreThreshold/
|
|
561
|
+
// trialsCompleted) directly in the session report, so integritySession-only
|
|
562
|
+
// and raw.cyborgHunter payloads already carry it. Without this, summary.js
|
|
563
|
+
// falls back to a per-trial `trialHits > 0` rule that OVERSTATES hard flags
|
|
564
|
+
// (any single hit flags hard, ignoring the count threshold).
|
|
565
|
+
if (!score && session) score = scoreFromSession(session);
|
|
566
|
+
|
|
567
|
+
return { session, score };
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
// Builds a { hardScore, softScore, anyHardTriggered, softScoreThreshold,
|
|
571
|
+
// trialsCompleted } score object from a session report that embeds those
|
|
572
|
+
// fields. Returns null if the session carries no scoring fields at all.
|
|
573
|
+
function scoreFromSession(session) {
|
|
574
|
+
if (!session || typeof session !== 'object') return null;
|
|
575
|
+
const hasScore = session.softScore !== undefined
|
|
576
|
+
|| session.anyHardTriggered !== undefined
|
|
577
|
+
|| session.hardScore !== undefined;
|
|
578
|
+
if (!hasScore) return null;
|
|
579
|
+
// Derive anyHardTriggered from hardScore when the boolean is absent (some
|
|
580
|
+
// session payloads carry the per-signal hardScore map but not the rolled-up
|
|
581
|
+
// flag). Without this, summary.js would fall back to its per-trial trialHits>0
|
|
582
|
+
// rule, which overstates hard flags.
|
|
583
|
+
let anyHardTriggered = session.anyHardTriggered;
|
|
584
|
+
if (anyHardTriggered === undefined && session.hardScore && typeof session.hardScore === 'object') {
|
|
585
|
+
anyHardTriggered = Object.values(session.hardScore).some(s => s && s.triggered);
|
|
586
|
+
}
|
|
587
|
+
return {
|
|
588
|
+
hardScore: session.hardScore,
|
|
589
|
+
softScore: session.softScore,
|
|
590
|
+
anyHardTriggered,
|
|
591
|
+
softScoreThreshold: session.softScoreThreshold,
|
|
592
|
+
trialsCompleted: session.trialsCompleted,
|
|
593
|
+
};
|
|
594
|
+
}
|
|
595
|
+
|
|
596
|
+
// Normalizes guard-honeypot violation evidence into the { violations: [...] }
|
|
597
|
+
// shape the session-timeline renderer expects under `guardFriction`. The
|
|
598
|
+
// honeypot extension writes the friction violation log (it subscribes to
|
|
599
|
+
// guard-friction.onViolation) as a STRINGIFIED `guard_assistance_violations_session`
|
|
600
|
+
// field via addProperties, so on a saved payload it lands on the trial rows
|
|
601
|
+
// (or metadata). Each entry is { reason, start, end, duration, in_progress };
|
|
602
|
+
// the renderer keys off `t` (perfNow ms), so map start → t. Apps that hand-write
|
|
603
|
+
// a top-level `raw.guardFriction` object still take precedence over this.
|
|
604
|
+
function findGuardViolations(raw) {
|
|
605
|
+
const parseArr = (v) => {
|
|
606
|
+
if (Array.isArray(v)) return v;
|
|
607
|
+
if (typeof v === 'string') { try { return JSON.parse(v); } catch { return null; } }
|
|
608
|
+
return null;
|
|
609
|
+
};
|
|
610
|
+
// True when a candidate value parses to at least one REAL violation (an entry
|
|
611
|
+
// with a numeric start). Used to scan trials for real data rather than the
|
|
612
|
+
// first trial that merely has a truthy field.
|
|
613
|
+
const hasReal = (v) => {
|
|
614
|
+
const a = parseArr(v);
|
|
615
|
+
return Array.isArray(a) && a.some(x => x && typeof x.start === 'number');
|
|
616
|
+
};
|
|
617
|
+
const candidates = [
|
|
618
|
+
raw.metadata?.guard_assistance_violations_session,
|
|
619
|
+
Array.isArray(raw.trials) && raw.trials.length > 0
|
|
620
|
+
? raw.trials[raw.trials.length - 1]?.guard_assistance_violations_session : null,
|
|
621
|
+
// Scan ALL trials for one carrying REAL violations — not just the first trial
|
|
622
|
+
// with a truthy field. A truthy-but-empty "[]" on an earlier trial used to
|
|
623
|
+
// make .find() lock on, then the outer loop fell through to candidate 4 and
|
|
624
|
+
// a later trial's real violations were never reached. Not live for the
|
|
625
|
+
// shipped honeypot (it stamps this field identically on every trial), but a
|
|
626
|
+
// latent bug for any non-uniform / merged producer. (Sol R2 candidate-3.)
|
|
627
|
+
Array.isArray(raw.trials)
|
|
628
|
+
? raw.trials.map(t => t?.guard_assistance_violations_session).find(hasReal)
|
|
629
|
+
: null,
|
|
630
|
+
raw.guard_assistance_violations_session,
|
|
631
|
+
];
|
|
632
|
+
for (const c of candidates) {
|
|
633
|
+
const arr = parseArr(c);
|
|
634
|
+
if (Array.isArray(arr)) {
|
|
635
|
+
const violations = arr
|
|
636
|
+
.filter(v => v && typeof v.start === 'number')
|
|
637
|
+
.map(v => ({ t: v.start, reason: v.reason || 'unknown', phase: 'unknown', duration_ms: v.duration }));
|
|
638
|
+
// Fall through to the next source on an empty/violation-less candidate
|
|
639
|
+
// instead of locking onto it — otherwise an empty placeholder (e.g. a
|
|
640
|
+
// metadata mirror set to []) would shadow real violations on the trial rows.
|
|
641
|
+
if (violations.length > 0) return violations;
|
|
642
|
+
}
|
|
643
|
+
}
|
|
644
|
+
return null;
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
// Surfaces the guard-honeypot self-disclosure fields (the visible bait: an "I used
|
|
648
|
+
// AI" checkbox and a free-text box). The honeypot writes ai_use_session /
|
|
649
|
+
// ai_report_session via addProperties and per-trial ai_use / ai_report via
|
|
650
|
+
// on_finish. The runtime emits them but no analyzer consumed them — this lifts
|
|
651
|
+
// them onto the participant so summary.csv / triage can surface them. Returns
|
|
652
|
+
// null when the honeypot was not used (fields absent everywhere).
|
|
653
|
+
function findHoneypotDisclosure(raw) {
|
|
654
|
+
// Aggregate across every source rather than returning the first that carries a
|
|
655
|
+
// honeypot key. addProperties stamps ai_use_session=false / ai_report_session=''
|
|
656
|
+
// onto ALL trials, so the final trial is usually blank-but-present; returning it
|
|
657
|
+
// first would mask a positive disclosure recorded on an earlier trial. A
|
|
658
|
+
// disclosure anywhere (ticked checkbox OR non-empty report) makes the whole
|
|
659
|
+
// participant positive.
|
|
660
|
+
const sources = [];
|
|
661
|
+
if (raw.metadata && typeof raw.metadata === 'object') sources.push(raw.metadata);
|
|
662
|
+
if (Array.isArray(raw.trials)) sources.push(...raw.trials);
|
|
663
|
+
sources.push(raw);
|
|
664
|
+
|
|
665
|
+
// Accept the checkbox as a real boolean OR its CSV string form ("true"/"false"
|
|
666
|
+
// survive Papa's dynamicTyping in some pipelines). Crucially, presence requires
|
|
667
|
+
// an actual checkbox value or non-empty report text — NOT mere key existence:
|
|
668
|
+
// shared-header CSVs give honeypot-less participants blank ('' / null) cells,
|
|
669
|
+
// and counting those as "present" would manufacture a `no` instead of the
|
|
670
|
+
// documented empty (honeypot-absent) state.
|
|
671
|
+
const isBool = (v) => v === true || v === false || v === 'true' || v === 'false';
|
|
672
|
+
const isTrue = (v) => v === true || v === 'true';
|
|
673
|
+
const reportText = (v) => (typeof v === 'string' ? v : '');
|
|
674
|
+
|
|
675
|
+
let present = false;
|
|
676
|
+
let ticked = false;
|
|
677
|
+
let sessionReport = ''; // authoritative *_session report text (prefer longest)
|
|
678
|
+
let trialReport = ''; // longest per-trial report snapshot
|
|
679
|
+
for (const obj of sources) {
|
|
680
|
+
if (!obj || typeof obj !== 'object') continue;
|
|
681
|
+
const repS = reportText(obj.ai_report_session);
|
|
682
|
+
const repT = reportText(obj.ai_report);
|
|
683
|
+
const hasDisclosureField = isBool(obj.ai_use_session) || isBool(obj.ai_use)
|
|
684
|
+
|| repS.length > 0 || repT.length > 0;
|
|
685
|
+
if (!hasDisclosureField) continue;
|
|
686
|
+
present = true;
|
|
687
|
+
if (isTrue(obj.ai_use_session) || isTrue(obj.ai_use)) ticked = true;
|
|
688
|
+
if (repS.length > sessionReport.length) sessionReport = repS;
|
|
689
|
+
if (repT.length > trialReport.length) trialReport = repT;
|
|
690
|
+
}
|
|
691
|
+
if (!present) return null;
|
|
692
|
+
const aiReport = sessionReport || trialReport;
|
|
693
|
+
// aiUse reflects the explicit "I used AI" checkbox ONLY. The free-text box is
|
|
694
|
+
// surfaced separately (aiReport / honeypot_ai_report) for the reviewer to read
|
|
695
|
+
// and judge — auto-classifying any non-empty text as a positive would
|
|
696
|
+
// false-flag entries like "none" / "didn't use AI" and contradicts the
|
|
697
|
+
// documented "YES = ticked" semantics. The tool produces evidence, not verdicts.
|
|
698
|
+
return { aiUse: ticked, aiReport };
|
|
699
|
+
}
|
|
700
|
+
|
|
701
|
+
// Chronological-by-rule trial ordering: (rulePosition, phase-rank,
|
|
702
|
+
// trialNumber). Shows trials in the order each rule was actually
|
|
703
|
+
// experienced — gallery → post-gallery query → classification t1..t6 — with
|
|
704
|
+
// end-requery trials (rulePosition=null → Infinity) sorting to the very end,
|
|
705
|
+
// matching their session-end timing. Used by the phaseTrials merge above and
|
|
706
|
+
// by the trajectories renderer's default displayOrder. Stable-sort friendly:
|
|
707
|
+
// trials without any of these fields compare equal and keep their input order.
|
|
708
|
+
export function ruleChronologicalCompare(a, b) {
|
|
709
|
+
const phaseRank = (ph) => {
|
|
710
|
+
if (ph === 'gallery') return 0;
|
|
711
|
+
if (ph === 'post_gallery_query') return 1;
|
|
712
|
+
if (ph === 'end_requery') return 99;
|
|
713
|
+
return 2; // classification or unspecified
|
|
714
|
+
};
|
|
715
|
+
const aPos = a.rulePosition ?? Infinity;
|
|
716
|
+
const bPos = b.rulePosition ?? Infinity;
|
|
717
|
+
if (aPos !== bPos) return aPos - bPos;
|
|
718
|
+
const aP = phaseRank(a.phase);
|
|
719
|
+
const bP = phaseRank(b.phase);
|
|
720
|
+
if (aP !== bP) return aP - bP;
|
|
721
|
+
return (a.trialNumber ?? 0) - (b.trialNumber ?? 0);
|
|
143
722
|
}
|
|
144
723
|
|
|
145
|
-
// Maps
|
|
724
|
+
// Maps Shape-2 legacy field names to CyborgHunter schema names.
|
|
146
725
|
// The CLI can then use a single set of field names downstream.
|
|
147
726
|
const LEGACY_FIELD_MAP = {
|
|
148
727
|
mouseTrack: 'mouseEvents',
|
|
@@ -158,7 +737,7 @@ const LEGACY_FIELD_MAP = {
|
|
|
158
737
|
// FAST PATH — every trial has a per-trial performance.now() anchor (either
|
|
159
738
|
// trialStart_perfNow set by the jsPsych wrapper, or startTime set by the
|
|
160
739
|
// standalone monitor). Subtract directly; result is exact.
|
|
161
|
-
// ESTIMATOR PATH —
|
|
740
|
+
// ESTIMATOR PATH — Shape-2 / pre-monitor data has neither anchor.
|
|
162
741
|
// We infer the session-start performance.now() value from the
|
|
163
742
|
// relationship between trial-end wall-clocks (`timestamp`),
|
|
164
743
|
// `responseTime_ms`, and the first tab-away's session-relative `start`,
|
|
@@ -335,7 +914,10 @@ function parseCsvToRaw(text, config) {
|
|
|
335
914
|
|
|
336
915
|
// Hoist participant ID from the first row to the top level so Shape-1 ingest
|
|
337
916
|
// finds it via raw[pidField]. Falls back to 'unknown' if the column isn't there.
|
|
338
|
-
|
|
917
|
+
// getByPath supports dotted paths into JSON-parsed cells (e.g. a "metadata"
|
|
918
|
+
// column that held a stringified object); Shape-1's flat-key-first lookup
|
|
919
|
+
// then finds the hoisted value under the same (possibly dotted) key name.
|
|
920
|
+
const participantId = getByPath(rows[0], pidField) ?? 'unknown';
|
|
339
921
|
|
|
340
922
|
return {
|
|
341
923
|
[pidField]: participantId,
|