cyborg-hunter 0.7.0 → 0.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,552 @@
1
+ // src/cli/extract-core.js
2
+ // Pure core of the participant-file extraction step — no Node APIs, so a
3
+ // browser demo can bundle it directly (0.7.2 extraction from cli/ingest.js,
4
+ // which re-exports these below for existing callers/tests). ingest.js keeps
5
+ // everything that touches fs/zlib/papaparse (file discovery, CSV parsing,
6
+ // replay-artifact attachment); this module keeps the raw-JSON-object →
7
+ // { participantId, trials, ... } transform, which never touches the
8
+ // filesystem — extractIntegrityData(raw, config) has always taken an
9
+ // already-parsed `raw` object, not a file path.
10
+ //
11
+ // Moved verbatim from ingest.js: extractIntegrityData, findSessionData,
12
+ // looksLikeSessionData, scoreFromSession, findGuardViolations,
13
+ // findHoneypotDisclosure, normalizeTabAwayTimestamps, mapLegacyFields (+ its
14
+ // LEGACY_FIELD_MAP), ruleChronologicalCompare.
15
+
16
+ import { TRIAL_REPORT_FIELDS } from '../shared/schema.js';
17
+ import { getByPath } from '../shared/paths.js';
18
+
19
+ // Extracts integrity trial data from a single participant's raw JSON.
20
+ // Returns { participantId, trials, warnings, metadata }.
21
+ export function extractIntegrityData(raw, config) {
22
+ const warnings = [];
23
+ const pidField = config.participantIdField || 'participantId';
24
+ const intField = config.integrityField || 'integrity';
25
+
26
+ // Determine participant ID — check top level, then metadata sub-object.
27
+ // Since 0.6.1 the field supports dot-paths ("metadata.sessionId"); plain
28
+ // names keep the historical top-level → metadata fallback. `||` (not `??`)
29
+ // preserves the pre-0.6.1 treatment of empty-string IDs as missing.
30
+ const participantId = getByPath(raw, pidField) || getByPath(raw.metadata, pidField) || 'unknown';
31
+ // An id field that resolves to nothing yields 'unknown'. Silently, that both
32
+ // loses the real id AND collides every such file under one 'unknown' bucket
33
+ // downstream (see the duplicate-id check in ingest()). Warn so a mistyped
34
+ // participantIdField is visible instead of producing an all-'unknown' cohort.
35
+ if (participantId === 'unknown') {
36
+ warnings.push(
37
+ `participantId unresolved (field "${pidField}" not found at top level or in ` +
38
+ `metadata) — defaulted to "unknown". Check participantIdField.`
39
+ );
40
+ }
41
+
42
+ let trials = [];
43
+
44
+ // Shape 1: { trials: [{ integrity: {...} }, ...] } — jsPsych extension data.
45
+ // Each trial has an `integrity` sub-object added by CyborgHunter.endTrial().
46
+ // We merge the trial's own fields (ruleId, timestamp, rt, rulePosition, etc.)
47
+ // with the integrity sub-object's fields (tabAwayEvents, copyEvents, etc.).
48
+ // On a key collision the integrity sub-object wins, since it's the more
49
+ // authoritative source for those fields. This preserves both the renderer-
50
+ // needed experiment metadata AND the cyborg-hunter signal data on the same
51
+ // trial object.
52
+ if (Array.isArray(raw.trials)) {
53
+ trials = raw.trials
54
+ .filter(t => t && t[intField])
55
+ .map(t => ({ ...t, ...t[intField] }));
56
+ // jsPsych extension data carries trialStart_perfNow per trial (set by the
57
+ // wrapper's on_load), so normalization takes the exact-subtraction fast
58
+ // path. Without it, renderers see only session-absolute `start` values
59
+ // and plot tab-away markers far off the per-trial axis.
60
+ normalizeTabAwayTimestamps(trials, raw);
61
+ }
62
+
63
+ // Phase-trial extension: per-phase integrity reports for
64
+ // gallery / post-gallery-query / end-requery phases live on raw.phaseTrials,
65
+ // mirroring the per-classification-trial shape so the renderer can place
66
+ // gallery mouse trajectories and tab-away events on the same axes.
67
+ //
68
+ // After merging, sort by (rulePosition, phase-rank, trialNumber) so the
69
+ // trajectories grid shows trials in the chronological order each rule was
70
+ // actually experienced: gallery → post-gallery query → classification t1..t6.
71
+ // End-requery trials carry rulePosition=null and sort to the very end of
72
+ // the array, matching their session-end timing.
73
+ if (Array.isArray(raw.phaseTrials) && raw.phaseTrials.length > 0) {
74
+ const phaseTrials = raw.phaseTrials
75
+ .filter(t => t && t[intField])
76
+ .map(t => ({ ...t, ...t[intField] }));
77
+ normalizeTabAwayTimestamps(phaseTrials, raw);
78
+ trials = trials.concat(phaseTrials);
79
+ trials.sort(ruleChronologicalCompare);
80
+ }
81
+
82
+ // Shapes 2 and 3 are ALTERNATIVE top-level layouts, tried only when the
83
+ // Shape-1 trials/phaseTrials path produced nothing. Previously the Shape-2
84
+ // branch was chained as `else if` off the phaseTrials `if`, so a payload
85
+ // carrying both `trials` (real integrity) and `responses` — or even a stray
86
+ // empty `responses: []` — had its already-extracted integrity trials clobbered
87
+ // (or the participant silently dropped). Gate on trials.length so Shape-1 wins.
88
+ if (trials.length === 0) {
89
+ // Shape 2: { responses: [{ mouseTrack, tabAwayEvents, ... }] } (legacy).
90
+ // Signal data lives directly on the response — no integrity wrapper. We apply
91
+ // field name mapping (mouseTrack → mouseEvents).
92
+ if (Array.isArray(raw.responses)) {
93
+ trials = raw.responses.map((r, i) => mapLegacyFields({
94
+ ...r,
95
+ _sourceIndex: i
96
+ }));
97
+ // Normalize tab-away timestamps from session-relative performance.now()
98
+ // to trial-relative milliseconds so renderers can plot them on the
99
+ // same x-axis as mouseEvents[].t (which is already trial-relative).
100
+ normalizeTabAwayTimestamps(trials, raw);
101
+ }
102
+ // Shape 3: Top-level array of trials. Same merge contract as Shape 1:
103
+ // outer trial fields survive, integrity wins on collision, and tab-away
104
+ // timestamps get normalized. The spread builds OUR copy, so the
105
+ // array-field coercion below mutates that copy — never a caller-owned or
106
+ // frozen object (a frozen one would throw instead of coercing). The
107
+ // outer-merge also keeps replay pointers (integrityReplayMeta /
108
+ // replayFinalizeError ride the outer trial rows) visible to
109
+ // attachReplayArtifacts.
110
+ else if (Array.isArray(raw)) {
111
+ trials = raw.filter(t => t && t[intField]).map(t => ({ ...t, ...t[intField] }));
112
+ normalizeTabAwayTimestamps(trials, raw);
113
+ }
114
+ }
115
+
116
+ // Normalize the raw mouseTrack → mouseEvents field name and derive
117
+ // mouseDataAvailable uniformly, regardless of which shape produced
118
+ // `trials`. Historically this only ran inside the Shape-2 branch above,
119
+ // because only Shape-2's legacy `responses[].mouseTrack` field ever used
120
+ // that name. Since monitor.js started persisting the post-hoc raw mouse
121
+ // trace under collectForPostHoc.rawMouseTrack (as report.mouseTrack —
122
+ // src/core/monitor.js endTrial()), Shape-1/3 trials can ALSO carry a
123
+ // `mouseTrack` field via the integrity sub-object. Applying the mapping
124
+ // once here, after every shape is resolved, covers all of them with one
125
+ // mechanism instead of duplicating it per branch. Idempotent for Shape-2
126
+ // trials that already passed through mapLegacyFields above.
127
+ trials = trials.map(mapLegacyFields);
128
+
129
+ if (trials.length === 0) {
130
+ warnings.push(`No integrity data found (looked for "${intField}" field)`);
131
+ }
132
+
133
+ // Validate each trial has the minimum required fields from the schema.
134
+ // Missing fields get a warning but don't prevent analysis. Array-typed signal
135
+ // fields that arrive as a non-array (e.g. a hand-edited `pasteEvents: {}`) are
136
+ // coerced to [] — otherwise a downstream `for (const e of trial.pasteEvents)`
137
+ // throws "not iterable" and, because report.js has no per-renderer boundary,
138
+ // one malformed file aborts the ENTIRE report. Coercing here keeps the run
139
+ // alive and localizes the damage to a warning on that trial.
140
+ for (const trial of trials) {
141
+ const missing = [];
142
+ for (const [field, spec] of Object.entries(TRIAL_REPORT_FIELDS)) {
143
+ if (spec.required && trial[field] === undefined) {
144
+ missing.push(field);
145
+ }
146
+ if (spec.type === 'array' && trial[field] != null && !Array.isArray(trial[field])) {
147
+ warnings.push(
148
+ `Trial ${trial.trialId ?? '?'}: field "${field}" is not an array ` +
149
+ `(got ${typeof trial[field]}) — coerced to [] to keep analysis running.`
150
+ );
151
+ trial[field] = [];
152
+ }
153
+ }
154
+ if (missing.length > 0) {
155
+ warnings.push(`Trial ${trial.trialId || '?'}: missing fields: ${missing.join(', ')}`);
156
+ }
157
+ }
158
+
159
+ const { session, score } = findSessionData(raw, config);
160
+ if (session === null && trials.length > 0) {
161
+ warnings.push('No session-level integrity data — some signals unavailable (did the experiment call getSessionReport()?)');
162
+ }
163
+
164
+ // The jsPsych adapter drops a `cyborgHunterFinalizeError` marker (via
165
+ // addProperties) when finalize() throws, so analysts can tell a missing-session
166
+ // run apart from a genuine finalize() failure. Surface it as a warning instead
167
+ // of leaving it dead in the data behind a generic "no session data" message.
168
+ const finalizeError = raw.cyborgHunterFinalizeError
169
+ ?? raw.metadata?.cyborgHunterFinalizeError
170
+ ?? (Array.isArray(raw.trials)
171
+ ? raw.trials.find(t => t?.cyborgHunterFinalizeError)?.cyborgHunterFinalizeError
172
+ : undefined);
173
+ if (finalizeError) {
174
+ warnings.push(`finalize() failed for this participant (cyborgHunterFinalizeError): ${finalizeError} — session data may be incomplete`);
175
+ }
176
+
177
+ return {
178
+ participantId,
179
+ trials,
180
+ warnings,
181
+ metadata: raw.metadata || {},
182
+ session,
183
+ score,
184
+ // Surface a few top-level payload fields that some renderers need but
185
+ // that aren't part of the session-level integrity object. Keeping the
186
+ // list explicit (rather than exposing `raw` wholesale) avoids future
187
+ // renderers silently coupling to payload internals.
188
+ galleryStudyMs: Array.isArray(raw.galleryStudyMs) ? raw.galleryStudyMs : null,
189
+ postGalleryGuesses: Array.isArray(raw.postGalleryGuesses) ? raw.postGalleryGuesses : null,
190
+ // App-written top-level guardFriction wins; otherwise synthesize the guard
191
+ // lane from the honeypot's session violation log (which the shipped guard
192
+ // extensions actually emit) so the timeline renders for library-only data.
193
+ guardFriction: raw.guardFriction ?? (() => {
194
+ const v = findGuardViolations(raw);
195
+ return v ? { violations: v } : null;
196
+ })(),
197
+ // Guard-honeypot self-disclosure (visible bait). null when the honeypot
198
+ // extension was not used.
199
+ honeypot: findHoneypotDisclosure(raw),
200
+ };
201
+ }
202
+
203
+ // True when obj plausibly IS a getSessionReport() output — i.e. it carries at
204
+ // least one of the well-known top-level session-report keys (see
205
+ // src/core/monitor.js sessionData / getSessionReport()). Guards the
206
+ // analyst-supplied sessionIntegrityPath below: `typeof === 'object'` alone
207
+ // accepts any object the path happens to resolve to, including a near-miss
208
+ // wrapper one level up the tree (e.g. `metadata` instead of
209
+ // `metadata.integritySession`), which would otherwise silently zero out every
210
+ // downstream signal instead of falling through.
211
+ function looksLikeSessionData(obj) {
212
+ if (!obj || typeof obj !== 'object') return false;
213
+ return ['tabAwaySums', 'hardScore', 'softScore', 'anyHardTriggered', 'trialsCompleted']
214
+ .some(key => Object.prototype.hasOwnProperty.call(obj, key));
215
+ }
216
+
217
+ // Locates session-level integrity data. An analyst-supplied dotted path
218
+ // (config.sessionIntegrityPath, 0.6.1) is checked first; then the built-in
219
+ // conventions, in priority order:
220
+ // 1. raw.metadata.integritySession / integrityScore — the metadata convention.
221
+ // 2. Last trial's integritySession / integrityScore — jsPsych addDataToLastTrial (Option A).
222
+ // 3. Any trial's integritySession — fallback (Option B).
223
+ // 4. raw.cyborgHunter — native top-level location used by raw-DOM
224
+ // adopters before they adopt the
225
+ // metadata.integritySession mirror. Added 2026-05-26.
226
+ // Returns { session, score }, both null if not found.
227
+ function findSessionData(raw, config) {
228
+ let session = null;
229
+ let score = null;
230
+
231
+ // 0. Analyst-specified location, e.g. "payload.cyborgHunter" for pipelines
232
+ // that nest the getSessionReport() output somewhere non-standard. Falls
233
+ // through to the built-in conventions when the path resolves to nothing
234
+ // OR to something that doesn't look like a session report (looksLikeSessionData,
235
+ // 0.6.1 — a malformed/near-miss path used to be accepted on typeof alone),
236
+ // so a partially-migrated cohort still ingests. The score is synthesized
237
+ // from the session object below (getSessionReport() embeds it).
238
+ const customSession = config?.sessionIntegrityPath
239
+ ? getByPath(raw, config.sessionIntegrityPath) : null;
240
+ if (looksLikeSessionData(customSession)) {
241
+ session = customSession;
242
+ }
243
+ // 1. metadata convention
244
+ else if (raw.metadata?.integritySession) {
245
+ session = raw.metadata.integritySession;
246
+ score = raw.metadata.integrityScore || null;
247
+ }
248
+ // 2. jsPsych addDataToLastTrial convention (Option A)
249
+ else if (Array.isArray(raw.trials) && raw.trials.length > 0 &&
250
+ raw.trials[raw.trials.length - 1]?.integritySession) {
251
+ const last = raw.trials[raw.trials.length - 1];
252
+ session = last.integritySession;
253
+ score = last.integrityScore || null;
254
+ }
255
+ // 3. Any-trial fallback
256
+ else if (Array.isArray(raw.trials) && raw.trials.find(x => x?.integritySession)) {
257
+ const t = raw.trials.find(x => x?.integritySession);
258
+ session = t.integritySession;
259
+ score = t.integrityScore || null;
260
+ }
261
+ // 4. Native top-level location. An early raw-DOM adopter app
262
+ // writes the getSessionReport() output directly to `raw.cyborgHunter`.
263
+ // Sessions saved before the metadata.integritySession mirror landed
264
+ // (2026-05-24) have ONLY this top-level field. Without
265
+ // this fallback, the renderer can't find windowPositions and falls back
266
+ // to stale top-level metadata.windowWidth — visible as a misaligned
267
+ // "browser window" dashed rectangle in the trajectory PNGs.
268
+ else if (raw.cyborgHunter && typeof raw.cyborgHunter === 'object') {
269
+ session = raw.cyborgHunter;
270
+ score = null;
271
+ }
272
+
273
+ // When no separate integrityScore blob was saved, synthesize the authoritative
274
+ // score from the session object itself. getSessionReport() embeds the scoring
275
+ // summary (hardScore/softScore/anyHardTriggered/softScoreThreshold/
276
+ // trialsCompleted) directly in the session report, so integritySession-only
277
+ // and raw.cyborgHunter payloads already carry it. Without this, summary.js
278
+ // falls back to a per-trial `trialHits > 0` rule that OVERSTATES hard flags
279
+ // (any single hit flags hard, ignoring the count threshold).
280
+ if (!score && session) score = scoreFromSession(session);
281
+
282
+ return { session, score };
283
+ }
284
+
285
+ // Builds a { hardScore, softScore, anyHardTriggered, softScoreThreshold,
286
+ // trialsCompleted } score object from a session report that embeds those
287
+ // fields. Returns null if the session carries no scoring fields at all.
288
+ function scoreFromSession(session) {
289
+ if (!session || typeof session !== 'object') return null;
290
+ const hasScore = session.softScore !== undefined
291
+ || session.anyHardTriggered !== undefined
292
+ || session.hardScore !== undefined;
293
+ if (!hasScore) return null;
294
+ // Derive anyHardTriggered from hardScore when the boolean is absent (some
295
+ // session payloads carry the per-signal hardScore map but not the rolled-up
296
+ // flag). Without this, summary.js would fall back to its per-trial trialHits>0
297
+ // rule, which overstates hard flags.
298
+ let anyHardTriggered = session.anyHardTriggered;
299
+ if (anyHardTriggered === undefined && session.hardScore && typeof session.hardScore === 'object') {
300
+ anyHardTriggered = Object.values(session.hardScore).some(s => s && s.triggered);
301
+ }
302
+ return {
303
+ hardScore: session.hardScore,
304
+ softScore: session.softScore,
305
+ anyHardTriggered,
306
+ softScoreThreshold: session.softScoreThreshold,
307
+ trialsCompleted: session.trialsCompleted,
308
+ };
309
+ }
310
+
311
+ // Normalizes guard-honeypot violation evidence into the { violations: [...] }
312
+ // shape the session-timeline renderer expects under `guardFriction`. The
313
+ // honeypot extension writes the friction violation log (it subscribes to
314
+ // guard-friction.onViolation) as a STRINGIFIED `guard_assistance_violations_session`
315
+ // field via addProperties, so on a saved payload it lands on the trial rows
316
+ // (or metadata). Each entry is { reason, start, end, duration, in_progress };
317
+ // the renderer keys off `t` (perfNow ms), so map start → t. Apps that hand-write
318
+ // a top-level `raw.guardFriction` object still take precedence over this.
319
+ function findGuardViolations(raw) {
320
+ const parseArr = (v) => {
321
+ if (Array.isArray(v)) return v;
322
+ if (typeof v === 'string') { try { return JSON.parse(v); } catch { return null; } }
323
+ return null;
324
+ };
325
+ // True when a candidate value parses to at least one REAL violation (an entry
326
+ // with a numeric start). Used to scan trials for real data rather than the
327
+ // first trial that merely has a truthy field.
328
+ const hasReal = (v) => {
329
+ const a = parseArr(v);
330
+ return Array.isArray(a) && a.some(x => x && typeof x.start === 'number');
331
+ };
332
+ const candidates = [
333
+ raw.metadata?.guard_assistance_violations_session,
334
+ Array.isArray(raw.trials) && raw.trials.length > 0
335
+ ? raw.trials[raw.trials.length - 1]?.guard_assistance_violations_session : null,
336
+ // Scan ALL trials for one carrying REAL violations — not just the first trial
337
+ // with a truthy field. A truthy-but-empty "[]" on an earlier trial used to
338
+ // make .find() lock on, then the outer loop fell through to candidate 4 and
339
+ // a later trial's real violations were never reached. Not live for the
340
+ // shipped honeypot (it stamps this field identically on every trial), but a
341
+ // latent bug for any non-uniform / merged producer. (Sol R2 candidate-3.)
342
+ Array.isArray(raw.trials)
343
+ ? raw.trials.map(t => t?.guard_assistance_violations_session).find(hasReal)
344
+ : null,
345
+ raw.guard_assistance_violations_session,
346
+ ];
347
+ for (const c of candidates) {
348
+ const arr = parseArr(c);
349
+ if (Array.isArray(arr)) {
350
+ const violations = arr
351
+ .filter(v => v && typeof v.start === 'number')
352
+ .map(v => ({ t: v.start, reason: v.reason || 'unknown', phase: 'unknown', duration_ms: v.duration }));
353
+ // Fall through to the next source on an empty/violation-less candidate
354
+ // instead of locking onto it — otherwise an empty placeholder (e.g. a
355
+ // metadata mirror set to []) would shadow real violations on the trial rows.
356
+ if (violations.length > 0) return violations;
357
+ }
358
+ }
359
+ return null;
360
+ }
361
+
362
+ // Surfaces the guard-honeypot self-disclosure fields (the visible bait: an "I used
363
+ // AI" checkbox and a free-text box). The honeypot writes ai_use_session /
364
+ // ai_report_session via addProperties and per-trial ai_use / ai_report via
365
+ // on_finish. The runtime emits them but no analyzer consumed them — this lifts
366
+ // them onto the participant so summary.csv / triage can surface them. Returns
367
+ // null when the honeypot was not used (fields absent everywhere).
368
+ function findHoneypotDisclosure(raw) {
369
+ // Aggregate across every source rather than returning the first that carries a
370
+ // honeypot key. addProperties stamps ai_use_session=false / ai_report_session=''
371
+ // onto ALL trials, so the final trial is usually blank-but-present; returning it
372
+ // first would mask a positive disclosure recorded on an earlier trial. A
373
+ // disclosure anywhere (ticked checkbox OR non-empty report) makes the whole
374
+ // participant positive.
375
+ const sources = [];
376
+ if (raw.metadata && typeof raw.metadata === 'object') sources.push(raw.metadata);
377
+ if (Array.isArray(raw.trials)) sources.push(...raw.trials);
378
+ sources.push(raw);
379
+
380
+ // Accept the checkbox as a real boolean OR its CSV string form ("true"/"false"
381
+ // survive Papa's dynamicTyping in some pipelines). Crucially, presence requires
382
+ // an actual checkbox value or non-empty report text — NOT mere key existence:
383
+ // shared-header CSVs give honeypot-less participants blank ('' / null) cells,
384
+ // and counting those as "present" would manufacture a `no` instead of the
385
+ // documented empty (honeypot-absent) state.
386
+ const isBool = (v) => v === true || v === false || v === 'true' || v === 'false';
387
+ const isTrue = (v) => v === true || v === 'true';
388
+ const reportText = (v) => (typeof v === 'string' ? v : '');
389
+
390
+ let present = false;
391
+ let ticked = false;
392
+ let sessionReport = ''; // authoritative *_session report text (prefer longest)
393
+ let trialReport = ''; // longest per-trial report snapshot
394
+ for (const obj of sources) {
395
+ if (!obj || typeof obj !== 'object') continue;
396
+ const repS = reportText(obj.ai_report_session);
397
+ const repT = reportText(obj.ai_report);
398
+ const hasDisclosureField = isBool(obj.ai_use_session) || isBool(obj.ai_use)
399
+ || repS.length > 0 || repT.length > 0;
400
+ if (!hasDisclosureField) continue;
401
+ present = true;
402
+ if (isTrue(obj.ai_use_session) || isTrue(obj.ai_use)) ticked = true;
403
+ if (repS.length > sessionReport.length) sessionReport = repS;
404
+ if (repT.length > trialReport.length) trialReport = repT;
405
+ }
406
+ if (!present) return null;
407
+ const aiReport = sessionReport || trialReport;
408
+ // aiUse reflects the explicit "I used AI" checkbox ONLY. The free-text box is
409
+ // surfaced separately (aiReport / honeypot_ai_report) for the reviewer to read
410
+ // and judge — auto-classifying any non-empty text as a positive would
411
+ // false-flag entries like "none" / "didn't use AI" and contradicts the
412
+ // documented "YES = ticked" semantics. The tool produces evidence, not verdicts.
413
+ return { aiUse: ticked, aiReport };
414
+ }
415
+
416
+ // Chronological-by-rule trial ordering: (rulePosition, phase-rank,
417
+ // trialNumber). Shows trials in the order each rule was actually
418
+ // experienced — gallery → post-gallery query → classification t1..t6 — with
419
+ // end-requery trials (rulePosition=null → Infinity) sorting to the very end,
420
+ // matching their session-end timing. Used by the phaseTrials merge above and
421
+ // by the trajectories renderer's default displayOrder. Stable-sort friendly:
422
+ // trials without any of these fields compare equal and keep their input order.
423
+ export function ruleChronologicalCompare(a, b) {
424
+ const phaseRank = (ph) => {
425
+ if (ph === 'gallery') return 0;
426
+ if (ph === 'post_gallery_query') return 1;
427
+ if (ph === 'end_requery') return 99;
428
+ return 2; // classification or unspecified
429
+ };
430
+ const aPos = a.rulePosition ?? Infinity;
431
+ const bPos = b.rulePosition ?? Infinity;
432
+ if (aPos !== bPos) return aPos - bPos;
433
+ const aP = phaseRank(a.phase);
434
+ const bP = phaseRank(b.phase);
435
+ if (aP !== bP) return aP - bP;
436
+ return (a.trialNumber ?? 0) - (b.trialNumber ?? 0);
437
+ }
438
+
439
+ // Maps legacy/alternate field names to CyborgHunter schema names. Originally
440
+ // Shape-2-only (responses[].mouseTrack); now applied to every shape (see the
441
+ // call site above) since monitor.js's collectForPostHoc.rawMouseTrack can
442
+ // also produce a `mouseTrack` field on Shape-1/3 trials. The CLI can then use
443
+ // a single set of field names downstream regardless of source.
444
+ const LEGACY_FIELD_MAP = {
445
+ mouseTrack: 'mouseEvents',
446
+ };
447
+
448
+ // Normalizes tabAwayEvents[*].start from session-relative performance.now()
449
+ // (ms since browser navigation) to trial-relative ms (ms since this trial
450
+ // began), stored as a new `startRel_ms` field. Renderers plot per-trial
451
+ // against trial-relative time; without normalization, tab-away markers
452
+ // land tens of thousands of ms off the right edge — invisible.
453
+ //
454
+ // Two paths:
455
+ // FAST PATH — every trial has a per-trial performance.now() anchor (either
456
+ // trialStart_perfNow set by the jsPsych wrapper, or startTime set by the
457
+ // standalone monitor). Subtract directly; result is exact.
458
+ // ESTIMATOR PATH — Shape-2 / pre-monitor data has neither anchor.
459
+ // We infer the session-start performance.now() value from the
460
+ // relationship between trial-end wall-clocks (`timestamp`),
461
+ // `responseTime_ms`, and the first tab-away's session-relative `start`,
462
+ // then subtract the inferred offset to get a trial-relative value.
463
+ // Less precise than the fast path; used only when nothing better exists.
464
+ function normalizeTabAwayTimestamps(trials, raw) {
465
+ // Fast path — every trial has a usable anchor.
466
+ const everyTrialHasAnchor = trials.length > 0 && trials.every(t =>
467
+ typeof (t.trialStart_perfNow ?? t.startTime) === 'number'
468
+ );
469
+ if (everyTrialHasAnchor) {
470
+ for (const trial of trials) {
471
+ const tabs = trial.tabAwayEvents;
472
+ if (!Array.isArray(tabs) || tabs.length === 0) continue;
473
+ const trialStart = trial.trialStart_perfNow ?? trial.startTime;
474
+ trial.tabAwayEvents = tabs.map(ta => ({
475
+ ...ta,
476
+ startRel_ms: typeof ta.start === 'number' ? ta.start - trialStart : null
477
+ }));
478
+ }
479
+ return;
480
+ }
481
+
482
+ // Estimator path. Need a session-start wall-clock anchor; if there isn't
483
+ // one, we have no way to relate trial-end timestamps to a 0 reference.
484
+ const sessStartIso = raw.metadata?.startTime;
485
+ if (!sessStartIso) return;
486
+ const sessStartMs = Date.parse(sessStartIso);
487
+ if (!Number.isFinite(sessStartMs)) return;
488
+
489
+ // For each trial, compute trial-start in ms-since-session-start (wall-clock).
490
+ // trial-end is response.timestamp; trial-start = trial-end − duration.
491
+ const trialStartRels = trials.map(t => {
492
+ if (!t.timestamp || t.responseTime_ms == null) return null;
493
+ const endMs = Date.parse(t.timestamp);
494
+ if (!Number.isFinite(endMs)) return null;
495
+ return (endMs - sessStartMs) - t.responseTime_ms;
496
+ });
497
+
498
+ // For each trial that has ≥1 tab-away, estimate the session offset:
499
+ // sessionOffset = (performance.now value at session start, in ms)
500
+ // The first tab-away of the trial must fall inside the trial's response
501
+ // window, so:
502
+ // firstTabStart ∈ [sessionOffset + trialStartRel,
503
+ // sessionOffset + trialStartRel + trialDuration]
504
+ // Solving for sessionOffset gives an interval; we use the midpoint as
505
+ // each trial's candidate, then take the median across trials as the
506
+ // robust estimate. This is approximate but converges quickly with even
507
+ // a few tab-away-bearing trials.
508
+ const candidates = [];
509
+ for (let i = 0; i < trials.length; i++) {
510
+ const tabs = trials[i].tabAwayEvents;
511
+ if (!Array.isArray(tabs) || tabs.length === 0) continue;
512
+ const trialStartRel = trialStartRels[i];
513
+ const trialDurationMs = trials[i].responseTime_ms;
514
+ if (trialStartRel == null || trialDurationMs == null) continue;
515
+ const firstTabStart = tabs[0].start;
516
+ if (typeof firstTabStart !== 'number') continue;
517
+ candidates.push(firstTabStart - trialStartRel - trialDurationMs / 2);
518
+ }
519
+
520
+ if (candidates.length === 0) return;
521
+
522
+ candidates.sort((a, b) => a - b);
523
+ const sessionOffset = candidates[Math.floor(candidates.length / 2)];
524
+
525
+ // Apply normalization. Original `start` is preserved alongside `startRel_ms`
526
+ // so consumers can still see the absolute time if they want it.
527
+ for (let i = 0; i < trials.length; i++) {
528
+ const tabs = trials[i].tabAwayEvents;
529
+ if (!Array.isArray(tabs) || tabs.length === 0) continue;
530
+ const trialStartRel = trialStartRels[i];
531
+ if (trialStartRel == null) continue;
532
+ trials[i].tabAwayEvents = tabs.map(ta => ({
533
+ ...ta,
534
+ startRel_ms: typeof ta.start === 'number'
535
+ ? ta.start - sessionOffset - trialStartRel
536
+ : null
537
+ }));
538
+ }
539
+ }
540
+
541
+ function mapLegacyFields(trial) {
542
+ const mapped = { ...trial };
543
+ for (const [oldName, newName] of Object.entries(LEGACY_FIELD_MAP)) {
544
+ if (mapped[oldName] !== undefined && mapped[newName] === undefined) {
545
+ mapped[newName] = mapped[oldName];
546
+ delete mapped[oldName];
547
+ }
548
+ }
549
+ // Flag distinguishes "no tracking hardware" from "tracked, zero events"
550
+ mapped.mouseDataAvailable = Array.isArray(mapped.mouseEvents) && mapped.mouseEvents.length > 0;
551
+ return mapped;
552
+ }