cyborg-hunter 0.5.0 → 0.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/CHANGELOG.md +156 -0
  2. package/CITATION.cff +29 -0
  3. package/LICENSE +21 -0
  4. package/README.md +80 -21
  5. package/bin/cyborg-hunter.js +6 -2
  6. package/dist/cyborg-hunter-replay.js +3 -0
  7. package/dist/cyborg-hunter.esm.js +114 -22
  8. package/dist/cyborg-hunter.min.js +3 -3
  9. package/dist/extension-cyborg-hunter.js +1 -1
  10. package/dist/extension-guard-friction.js +5 -5
  11. package/dist/extension-guard-honeypot.js +1 -1
  12. package/package.json +15 -3
  13. package/src/cli/analyzers/edge-exit.js +4 -1
  14. package/src/cli/analyzers/phase-scope.js +83 -0
  15. package/src/cli/analyzers/summary.js +161 -28
  16. package/src/cli/analyzers/triage.js +59 -27
  17. package/src/cli/config.js +26 -1
  18. package/src/cli/extract-core.js +552 -0
  19. package/src/cli/ingest.js +250 -193
  20. package/src/cli/init.js +1 -1
  21. package/src/cli/preview-entry.js +36 -0
  22. package/src/cli/renderers/event-log.js +18 -19
  23. package/src/cli/renderers/extensions.js +12 -3
  24. package/src/cli/renderers/html-index-core.js +1273 -0
  25. package/src/cli/renderers/html-index.js +13 -1047
  26. package/src/cli/renderers/replay-assets.js +67 -0
  27. package/src/cli/renderers/replay-viewer.client.js +1185 -0
  28. package/src/cli/renderers/session-timeline-core.js +907 -0
  29. package/src/cli/renderers/session-timeline.js +33 -0
  30. package/src/cli/renderers/summary-csv.js +5 -0
  31. package/src/cli/renderers/trajectories-core.js +717 -0
  32. package/src/cli/renderers/trajectories.js +31 -635
  33. package/src/cli/renderers/triage-md.js +10 -4
  34. package/src/cli/renderers/typing-profile-core.js +211 -0
  35. package/src/cli/renderers/typing-profile.js +16 -186
  36. package/src/cli/report.js +42 -8
  37. package/src/core/monitor.js +77 -7
  38. package/src/core/scoring.js +11 -2
  39. package/src/core/signals/browser.js +51 -18
  40. package/src/core/signals/clipboard.js +10 -2
  41. package/src/core/signals/dom-protection.js +9 -0
  42. package/src/core/signals/focus.js +16 -2
  43. package/src/jspsych/extension-cyborg-hunter-replay.js +135 -0
  44. package/src/jspsych/extension-cyborg-hunter.js +9 -2
  45. package/src/jspsych/extension-guard-friction.js +43 -14
  46. package/src/jspsych/extension-guard-honeypot.js +25 -1
  47. package/src/replay/capture-dom.js +575 -0
  48. package/src/replay/capture-trace.js +468 -0
  49. package/src/replay/index.js +104 -0
  50. package/src/replay/persistence.js +141 -0
  51. package/src/replay/recorder.js +315 -0
  52. package/src/replay/serializer.js +119 -0
  53. package/src/replay/viewer-model.js +125 -0
  54. package/src/shared/constants.js +12 -6
  55. package/src/shared/paths.js +20 -0
  56. package/src/shared/schema.js +5 -0
  57. package/src/shared/validation.js +55 -0
  58. package/src/cli/renderers/tab-timeline.js +0 -149
package/src/cli/ingest.js CHANGED
@@ -4,7 +4,7 @@
4
4
  // 1. JSON Shape 1 — { participantId: 'P1', trials: [{ integrity: {...} }] }
5
5
  // jsPsych extension data with one file per participant.
6
6
  // 2. JSON Shape 2 — { metadata: {...}, responses: [{ mouseTrack, tabAwayEvents, ... }] }
7
- // The original rule-gallery format. Signal data lives flat on each response
7
+ // The original legacy format. Signal data lives flat on each response
8
8
  // (no `integrity` wrapper). Pre-dates the standalone library.
9
9
  // 3. CSV — jsPsych default save format. Each row is a trial; nested objects
10
10
  // (integrity, integritySession, integrityScore) are JSON-stringified into
@@ -12,19 +12,71 @@
12
12
 
13
13
  import { readFileSync, readdirSync } from 'fs';
14
14
  import { join, extname } from 'path';
15
+ import { gunzipSync } from 'zlib';
15
16
  import Papa from 'papaparse';
16
- import { TRIAL_REPORT_FIELDS } from '../shared/schema.js';
17
+ import { sanitizeId } from '../shared/constants.js';
18
+ import { getByPath } from '../shared/paths.js';
19
+ import { extractIntegrityData, ruleChronologicalCompare } from './extract-core.js';
20
+
21
+ // Replay artifacts saved by the replay extension:
22
+ // <sanitizedPid>-replay-<sessionStartEpochMs>.json[.gz]
23
+ // They sit in dataDir (or replayDir) next to the participant files and must
24
+ // never enter the participant-file pass.
25
+ const REPLAY_FILE_RE = /-replay-\d+\.json(\.gz)?$/i;
26
+
27
+ // Content sniff: replay artifacts (ours or #3661's) are identified by
28
+ // structure, not just filename — schema_version plus either our recorder
29
+ // stamp or a #3661-shaped trials array.
30
+ function looksLikeReplayArtifact(text) {
31
+ try {
32
+ const j = JSON.parse(text);
33
+ return j && typeof j === 'object' && 'schema_version' in j &&
34
+ (String(j.metadata?.recorder || '').startsWith('cyborg-hunter-replay') ||
35
+ (Array.isArray(j.trials) && j.trials.length > 0 &&
36
+ j.trials.every(t => t && 'events' in t && 'initial_dom' in t)));
37
+ } catch (e) {
38
+ return true; // unparseable + replay-named → let the replay pass report it
39
+ }
40
+ }
17
41
 
18
42
  export async function ingest(config) {
19
- const files = findFiles(config.dataDir, config.filePattern);
43
+ const allFiles = findFiles(config.dataDir, config.filePattern);
20
44
  const participants = [];
21
45
  const warnings = [];
22
46
 
47
+ // Replay artifacts are excluded from the participant pass by filename —
48
+ // but only after a content check, so a participant export that happens to
49
+ // match the naming pattern is never silently dropped.
50
+ const files = [];
51
+ for (const file of allFiles) {
52
+ if (REPLAY_FILE_RE.test(file)) {
53
+ let text = null;
54
+ try { text = readFileSync(file, 'utf8'); } catch (e) { text = null; }
55
+ let parseable = true;
56
+ if (text !== null) {
57
+ try { JSON.parse(text); } catch (e) { parseable = false; }
58
+ }
59
+ if (text === null || !parseable) {
60
+ // Never let a replay-named file vanish silently: if its pid maps to
61
+ // a discovered participant the attach pass warns again with more
62
+ // context, but an orphan (no matching participant) would otherwise
63
+ // disappear without a trace.
64
+ warnings.push({ file,
65
+ warnings: ['Replay-named file could not be parsed (truncated upload or a misnamed participant export?) — skipped from the participant pass; if a matching participant exists, the replay pass reports it too.'] });
66
+ continue;
67
+ }
68
+ if (looksLikeReplayArtifact(text)) continue;
69
+ warnings.push({ file,
70
+ warnings: ['File matches the replay-artifact naming pattern (<pid>-replay-<epoch>.json) but contains participant data — parsed as a participant file. Consider renaming it to avoid ambiguity.'] });
71
+ }
72
+ files.push(file);
73
+ }
74
+
23
75
  for (const file of files) {
24
76
  try {
25
77
  const text = readFileSync(file, 'utf8');
26
78
  // Branch by extension. CSV is jsPsych's default save format; JSON is what
27
- // rule-gallery and other server-side-saving experiments use.
79
+ // server-side-saving experiments use.
28
80
  const raw = extname(file).toLowerCase() === '.csv'
29
81
  ? parseCsvToRaw(text, config)
30
82
  : JSON.parse(text);
@@ -46,214 +98,216 @@ export async function ingest(config) {
46
98
  }
47
99
  }
48
100
 
49
- return { participants, warnings };
50
- }
51
-
52
- // Extracts integrity trial data from a single participant's raw JSON.
53
- // Returns { participantId, trials, warnings, metadata }.
54
- export function extractIntegrityData(raw, config) {
55
- const warnings = [];
56
- const pidField = config.participantIdField || 'participantId';
57
- const intField = config.integrityField || 'integrity';
58
-
59
- // Determine participant ID — check top level, then metadata sub-object
60
- const participantId = raw[pidField] || raw.metadata?.[pidField] || 'unknown';
61
-
62
- let trials = [];
63
-
64
- // Shape 1: { trials: [{ integrity: {...} }, ...] } — jsPsych extension data.
65
- // Each trial has an `integrity` sub-object added by CyborgHunter.endTrial().
66
- if (Array.isArray(raw.trials)) {
67
- trials = raw.trials
68
- .filter(t => t[intField])
69
- .map(t => t[intField]);
70
- // jsPsych extension data carries trialStart_perfNow per trial (set by the
71
- // wrapper's on_load), so normalization takes the exact-subtraction fast
72
- // path. Without it, renderers see only session-absolute `start` values
73
- // and plot tab-away markers far off the per-trial axis.
74
- normalizeTabAwayTimestamps(trials, raw);
75
- }
76
- // Shape 2: { responses: [{ mouseTrack, tabAwayEvents, ... }] } (rule-gallery legacy format)
77
- // Signal data lives directly on the response — no integrity wrapper.
78
- // We apply field name mapping (mouseTrack → mouseEvents).
79
- else if (Array.isArray(raw.responses)) {
80
- trials = raw.responses.map((r, i) => mapLegacyFields({
81
- ...r,
82
- _sourceIndex: i
83
- }));
84
- // Normalize tab-away timestamps from session-relative performance.now()
85
- // to trial-relative milliseconds so renderers can plot them on the
86
- // same x-axis as mouseEvents[].t (which is already trial-relative).
87
- normalizeTabAwayTimestamps(trials, raw);
88
- }
89
- // Shape 3: Top-level array of trials
90
- else if (Array.isArray(raw)) {
91
- trials = raw.filter(t => t[intField]).map(t => t[intField]);
92
- }
93
-
94
- if (trials.length === 0) {
95
- warnings.push(`No integrity data found (looked for "${intField}" field)`);
96
- }
97
-
98
- // Validate each trial has the minimum required fields from the schema.
99
- // Missing fields get a warning but don't prevent analysis.
100
- for (const trial of trials) {
101
- const missing = [];
102
- for (const [field, spec] of Object.entries(TRIAL_REPORT_FIELDS)) {
103
- if (spec.required && trial[field] === undefined) {
104
- missing.push(field);
105
- }
106
- }
107
- if (missing.length > 0) {
108
- warnings.push(`Trial ${trial.trialId || '?'}: missing fields: ${missing.join(', ')}`);
101
+ // Duplicate-upload guard. Two files resolving to the same participantId are
102
+ // conflated downstream (triage/HTML/image outputs key by id), so the second
103
+ // upload's evidence can silently overwrite or vanish. We do NOT auto-dedup —
104
+ // the analyst must decide which upload is canonical — but we surface it.
105
+ // (attachReplayArtifacts below adds its own duplicate-id note describing the
106
+ // replay-association consequence specifically.)
107
+ const idCounts = new Map();
108
+ for (const p of participants) idCounts.set(p.participantId, (idCounts.get(p.participantId) || 0) + 1);
109
+ for (const [id, n] of idCounts) {
110
+ if (n > 1) {
111
+ warnings.push({ file: '(multiple)', warnings: [
112
+ `duplicate participantId "${id}" appears in ${n} files — downstream ` +
113
+ `outputs key by id, so entries may be conflated. Keep one upload per participant.`
114
+ ] });
109
115
  }
110
116
  }
111
117
 
112
- const { session, score } = findSessionData(raw);
113
- if (session === null && trials.length > 0) {
114
- warnings.push('No session-level integrity data — some signals unavailable (did the experiment call getSessionReport()?)');
115
- }
118
+ attachReplayArtifacts(participants, config, warnings);
116
119
 
117
- return { participantId, trials, warnings, metadata: raw.metadata || {}, session, score };
120
+ return { participants, warnings };
118
121
  }
119
122
 
120
- // Locates session-level integrity data in one of three locations, in priority order:
121
- // 1. raw.metadata.integritySession / integrityScore — the card-games convention.
122
- // 2. Last trial's integritySession / integrityScore — jsPsych addDataToLastTrial (Option A).
123
- // 3. Any trial's integritySession — fallback (Option B).
124
- // Returns { session, score }, both null if not found.
125
- function findSessionData(raw) {
126
- // 1. card-games metadata convention
127
- if (raw.metadata?.integritySession) {
128
- return { session: raw.metadata.integritySession, score: raw.metadata.integrityScore || null };
129
- }
130
- // 2. jsPsych addDataToLastTrial convention (Option A)
131
- if (Array.isArray(raw.trials) && raw.trials.length > 0) {
132
- const last = raw.trials[raw.trials.length - 1];
133
- if (last?.integritySession) {
134
- return { session: last.integritySession, score: last.integrityScore || null };
123
+ // Finds and attaches each participant's replay artifact (if any) as
124
+ // `participant.replay`:
125
+ // { recording, file, meta } — parsed and attached
126
+ // { error: 'parse_failed', reason } — artifact exists but unreadable
127
+ // null — no artifact (silent unless meta
128
+ // says one went to 'download')
129
+ function attachReplayArtifacts(participants, config, warnings) {
130
+ const dir = config.replayDir || config.dataDir;
131
+ let entries = [];
132
+ try {
133
+ entries = readdirSync(dir);
134
+ } catch (e) {
135
+ if (config.replayDir) {
136
+ warnings.push({ file: dir, warnings: [`replayDir not readable: ${e.message}`] });
135
137
  }
136
138
  }
137
- // 3. Any-trial fallback
138
- if (Array.isArray(raw.trials)) {
139
- const t = raw.trials.find(x => x?.integritySession);
140
- if (t) return { session: t.integritySession, score: t.integrityScore || null };
141
- }
142
- return { session: null, score: null };
143
- }
144
-
145
- // Maps rule-gallery legacy field names to CyborgHunter schema names.
146
- // The CLI can then use a single set of field names downstream.
147
- const LEGACY_FIELD_MAP = {
148
- mouseTrack: 'mouseEvents',
149
- };
150
139
 
151
- // Normalizes tabAwayEvents[*].start from session-relative performance.now()
152
- // (ms since browser navigation) to trial-relative ms (ms since this trial
153
- // began), stored as a new `startRel_ms` field. Renderers plot per-trial
154
- // against trial-relative time; without normalization, tab-away markers
155
- // land tens of thousands of ms off the right edge — invisible.
156
- //
157
- // Two paths:
158
- // FAST PATH — every trial has a per-trial performance.now() anchor (either
159
- // trialStart_perfNow set by the jsPsych wrapper, or startTime set by the
160
- // standalone monitor). Subtract directly; result is exact.
161
- // ESTIMATOR PATH — rule-gallery / pre-monitor data has neither anchor.
162
- // We infer the session-start performance.now() value from the
163
- // relationship between trial-end wall-clocks (`timestamp`),
164
- // `responseTime_ms`, and the first tab-away's session-relative `start`,
165
- // then subtract the inferred offset to get a trial-relative value.
166
- // Less precise than the fast path; used only when nothing better exists.
167
- function normalizeTabAwayTimestamps(trials, raw) {
168
- // Fast path — every trial has a usable anchor.
169
- const everyTrialHasAnchor = trials.length > 0 && trials.every(t =>
170
- typeof (t.trialStart_perfNow ?? t.startTime) === 'number'
171
- );
172
- if (everyTrialHasAnchor) {
173
- for (const trial of trials) {
174
- const tabs = trial.tabAwayEvents;
175
- if (!Array.isArray(tabs) || tabs.length === 0) continue;
176
- const trialStart = trial.trialStart_perfNow ?? trial.startTime;
177
- trial.tabAwayEvents = tabs.map(ta => ({
178
- ...ta,
179
- startRel_ms: typeof ta.start === 'number' ? ta.start - trialStart : null
180
- }));
181
- }
182
- return;
140
+ // Sanitized-name census: filename sanitization is many-to-one, so a
141
+ // no-embedded-id artifact may only attach when exactly one participant
142
+ // maps to its sanitized name (otherwise ownership is ambiguous).
143
+ // The census is keyed on LOWERCASED sanitized ids because the filename
144
+ // match below is case-insensitive (macOS filesystems are) — both
145
+ // mechanisms must share the same equivalence classes or an ownerless
146
+ // artifact could attach to two case-variant participants at once.
147
+ // Null-prototype maps: a participant id that collides with an
148
+ // Object.prototype key ("__proto__", "constructor") must count like any
149
+ // other id — on a literal {}, assigning a primitive to __proto__ is a
150
+ // silent no-op, which skipped the duplicate warning AND the
151
+ // ambiguous-association guard below.
152
+ const sanitize = sanitizeId;
153
+ const saneCounts = Object.create(null);
154
+ const idCounts = Object.create(null);
155
+ for (const p of participants) {
156
+ const s = sanitize(p.participantId).toLowerCase();
157
+ saneCounts[s] = (saneCounts[s] || 0) + 1;
158
+ idCounts[p.participantId] = (idCounts[p.participantId] || 0) + 1;
183
159
  }
184
160
 
185
- // Estimator path. Need a session-start wall-clock anchor; if there isn't
186
- // one, we have no way to relate trial-end timestamps to a 0 reference.
187
- const sessStartIso = raw.metadata?.startTime;
188
- if (!sessStartIso) return;
189
- const sessStartMs = Date.parse(sessStartIso);
190
- if (!Number.isFinite(sessStartMs)) return;
191
-
192
- // For each trial, compute trial-start in ms-since-session-start (wall-clock).
193
- // trial-end is response.timestamp; trial-start = trial-end − duration.
194
- const trialStartRels = trials.map(t => {
195
- if (!t.timestamp || t.responseTime_ms == null) return null;
196
- const endMs = Date.parse(t.timestamp);
197
- if (!Number.isFinite(endMs)) return null;
198
- return (endMs - sessStartMs) - t.responseTime_ms;
199
- });
200
-
201
- // For each trial that has ≥1 tab-away, estimate the session offset:
202
- // sessionOffset = (performance.now value at session start, in ms)
203
- // The first tab-away of the trial must fall inside the trial's response
204
- // window, so:
205
- // firstTabStart ∈ [sessionOffset + trialStartRel,
206
- // sessionOffset + trialStartRel + trialDuration]
207
- // Solving for sessionOffset gives an interval; we use the midpoint as
208
- // each trial's candidate, then take the median across trials as the
209
- // robust estimate. This is approximate but converges quickly with even
210
- // a few tab-away-bearing trials.
211
- const candidates = [];
212
- for (let i = 0; i < trials.length; i++) {
213
- const tabs = trials[i].tabAwayEvents;
214
- if (!Array.isArray(tabs) || tabs.length === 0) continue;
215
- const trialStartRel = trialStartRels[i];
216
- const trialDurationMs = trials[i].responseTime_ms;
217
- if (trialStartRel == null || trialDurationMs == null) continue;
218
- const firstTabStart = tabs[0].start;
219
- if (typeof firstTabStart !== 'number') continue;
220
- candidates.push(firstTabStart - trialStartRel - trialDurationMs / 2);
161
+ // Duplicate participant ids (repeat runs, duplicate exports) are outside
162
+ // the pipeline's data model — every renderer keys outputs by pid, so the
163
+ // whole report already treats them as one person. Replay attachment
164
+ // follows the same semantics (both records get the same latest artifact);
165
+ // say so once per duplicated id instead of silently doing it.
166
+ for (const [id, n] of Object.entries(idCounts)) {
167
+ if (n > 1) {
168
+ warnings.push({ file: dir,
169
+ warnings: [`Duplicate participant id "${id}" across ${n} files — the report (including replay attachment) treats these as one person; per-session replay association is not attempted.`] });
170
+ }
221
171
  }
222
172
 
223
- if (candidates.length === 0) return;
173
+ for (const p of participants) {
174
+ // Same sanitization the browser-side filename builder applies. The
175
+ // match is ANCHORED (^<sane>-replay-<digits>.json$): a bare prefix
176
+ // would let participant "a" swallow "a-replay-replay-<epoch>.json",
177
+ // which belongs to participant "a-replay".
178
+ const sane = sanitize(p.participantId);
179
+ const escaped = sane.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
180
+ const exactRe = new RegExp(`^${escaped}-replay-\\d+\\.json(\\.gz)?$`, 'i');
181
+ const mine = entries.filter(f => exactRe.test(f));
182
+ // The meta pointer rides on every trial row via addProperties.
183
+ const meta = (p.trials && p.trials[0] && p.trials[0].integrityReplayMeta) || null;
184
+ // Replay finalize failures ride the same way — surface them where the
185
+ // analyst looks (they mean the artifact was probably never saved).
186
+ const finErr = p.trials && p.trials[0] && p.trials[0].replayFinalizeError;
187
+ if (finErr) {
188
+ warnings.push({ file: dir,
189
+ warnings: [`Replay finalize failed for ${p.participantId}: ${finErr} — the artifact was likely never saved.`] });
190
+ }
224
191
 
225
- candidates.sort((a, b) => a - b);
226
- const sessionOffset = candidates[Math.floor(candidates.length / 2)];
192
+ if (mine.length === 0) {
193
+ p.replay = null;
194
+ if (meta && meta.saved_to === 'download') {
195
+ warnings.push({
196
+ file: dir,
197
+ warnings: [`Replay artifact for ${p.participantId} was downloaded to the participant's machine (autoSave mode "download") and is not recoverable from here — check the autosave configuration for future runs.`]
198
+ });
199
+ }
200
+ continue;
201
+ }
227
202
 
228
- // Apply normalization. Original `start` is preserved alongside `startRel_ms`
229
- // so consumers can still see the absolute time if they want it.
230
- for (let i = 0; i < trials.length; i++) {
231
- const tabs = trials[i].tabAwayEvents;
232
- if (!Array.isArray(tabs) || tabs.length === 0) continue;
233
- const trialStartRel = trialStartRels[i];
234
- if (trialStartRel == null) continue;
235
- trials[i].tabAwayEvents = tabs.map(ta => ({
236
- ...ta,
237
- startRel_ms: typeof ta.start === 'number'
238
- ? ta.start - sessionOffset - trialStartRel
239
- : null
240
- }));
241
- }
242
- }
203
+ const parsed = [];
204
+ for (const f of mine) {
205
+ try {
206
+ const buf = readFileSync(join(dir, f));
207
+ const json = f.toLowerCase().endsWith('.gz')
208
+ ? gunzipSync(buf).toString('utf8')
209
+ : buf.toString('utf8');
210
+ // Same structural sniff as the participant pass: a misnamed
211
+ // participant export was rescued as participant data there and
212
+ // must not double as its own "replay" here.
213
+ if (!looksLikeReplayArtifact(json)) continue;
214
+ parsed.push({ file: f, recording: JSON.parse(json) });
215
+ } catch (e) {
216
+ parsed.push({ file: f, reason: e.message });
217
+ }
218
+ }
219
+ if (parsed.length === 0) {
220
+ p.replay = null;
221
+ continue;
222
+ }
243
223
 
244
- function mapLegacyFields(trial) {
245
- const mapped = { ...trial };
246
- for (const [oldName, newName] of Object.entries(LEGACY_FIELD_MAP)) {
247
- if (mapped[oldName] !== undefined && mapped[newName] === undefined) {
248
- mapped[newName] = mapped[oldName];
249
- delete mapped[oldName];
224
+ // Every unreadable artifact warns individually — a corrupt NEWEST
225
+ // session must never be silently masked by an older readable one.
226
+ for (const bad of parsed.filter(x => !x.recording)) {
227
+ warnings.push({ file: join(dir, bad.file),
228
+ warnings: [`Replay artifact ${bad.file} unreadable: ${bad.reason} — if this is the newest session, its replay is lost.`] });
229
+ }
230
+ const readable = parsed.filter(x => x.recording);
231
+ if (readable.length === 0) {
232
+ p.replay = { error: 'parse_failed', reason: parsed[0].reason, file: parsed[0].file };
233
+ continue;
234
+ }
235
+ if (mine.length > 1) {
236
+ warnings.push({ file: dir,
237
+ warnings: [`Multiple replay artifacts for ${p.participantId} (page reload?) — using the latest by start_time.`] });
250
238
  }
239
+ // Filename sanitization is many-to-one ('a/b' and 'a_b' both map to
240
+ // 'a_b'), so ownership is verified against the UNsanitized
241
+ // participant_id embedded in the recording. Artifacts without one
242
+ // (e.g. plain #3661 recordings) attach with a soft warning.
243
+ const owned = [];
244
+ for (const cand of readable) {
245
+ const embedded = cand.recording.metadata?.participant_id;
246
+ if (embedded == null) {
247
+ // Ownerless artifacts skip id verification entirely, so the filename
248
+ // must match EXACT-case (our recorder writes sanitize(pid) verbatim).
249
+ // Case-tolerant matching stays for discovery, where the embedded-id
250
+ // check catches cross-case impostors.
251
+ if (!cand.file.startsWith(sane + '-replay-')) {
252
+ warnings.push({ file: join(dir, cand.file),
253
+ warnings: [`Replay artifact has no embedded participant_id and its filename case does not match "${sane}" exactly — not attached.`] });
254
+ } else if (saneCounts[sane.toLowerCase()] > 1) {
255
+ warnings.push({ file: join(dir, cand.file),
256
+ warnings: [`Replay artifact has no embedded participant_id and its filename is ambiguous (${saneCounts[sane.toLowerCase()]} participants sanitize to "${sane}") — not attached to anyone.`] });
257
+ } else {
258
+ warnings.push({ file: join(dir, cand.file),
259
+ warnings: [`Replay artifact has no embedded participant_id — cannot verify ownership; attaching to ${p.participantId} by unique filename match.`] });
260
+ owned.push(cand);
261
+ }
262
+ } else if (String(embedded) === String(p.participantId)) {
263
+ owned.push(cand);
264
+ } else {
265
+ warnings.push({ file: join(dir, cand.file),
266
+ warnings: [`Replay artifact participant_id mismatch: file matches ${p.participantId} by name but was recorded for ${embedded} (sanitization collision?) — not attached.`] });
267
+ }
268
+ }
269
+ if (owned.length === 0) {
270
+ p.replay = null;
271
+ continue;
272
+ }
273
+ // Duplicate records for this id + multiple owned artifacts: the
274
+ // per-session mapping is genuinely ambiguous. Attach nothing rather
275
+ // than knowingly mis-associate a session's replay.
276
+ if (idCounts[p.participantId] > 1 && owned.length > 1) {
277
+ warnings.push({ file: dir,
278
+ warnings: [`Cannot associate ${owned.length} replay artifacts with ${idCounts[p.participantId]} duplicate records of "${p.participantId}" — none attached. Separate the sessions into distinct data dirs (or ids) to view their replays.`] });
279
+ p.replay = null;
280
+ continue;
281
+ }
282
+ // Latest-session pick tolerates non-ISO start_time in third-party
283
+ // artifacts: ISO string → numeric epoch → the filename's own epoch.
284
+ const sessionEpoch = (cand) => {
285
+ const v = cand.recording.metadata?.start_time;
286
+ const n = typeof v === 'number' ? v : Date.parse(v);
287
+ if (Number.isFinite(n)) return n;
288
+ const m = cand.file.match(/-replay-(\d+)\.json/i);
289
+ return m ? Number(m[1]) : 0;
290
+ };
291
+ owned.sort((a, b) => sessionEpoch(a) - sessionEpoch(b));
292
+ const chosen = owned[owned.length - 1];
293
+ if (chosen.recording.schema_version !== 1) {
294
+ warnings.push({ file: join(dir, chosen.file),
295
+ warnings: [`Replay schema_version ${chosen.recording.schema_version} (this CLI targets 1) — attaching anyway; the viewer may degrade.`] });
296
+ }
297
+ p.replay = { recording: chosen.recording, file: chosen.file, meta };
251
298
  }
252
- // Flag distinguishes "no tracking hardware" from "tracked, zero events"
253
- mapped.mouseDataAvailable = Array.isArray(mapped.mouseEvents) && mapped.mouseEvents.length > 0;
254
- return mapped;
255
299
  }
256
300
 
301
+ // Load-bearing re-exports:
302
+ // - getByPath: html-index.js (and external adopters) import it from ingest.js.
303
+ // - extractIntegrityData, ruleChronologicalCompare: moved to extract-core.js
304
+ // (0.7.2 extraction — pure/no Node APIs so a browser demo can bundle it);
305
+ // this file re-exports both for existing callers — trajectories.js imports
306
+ // ruleChronologicalCompare from here, and tests/cli/*.test.js import
307
+ // extractIntegrityData from here.
308
+ export { getByPath };
309
+ export { extractIntegrityData, ruleChronologicalCompare };
310
+
257
311
  // Finds files matching a glob pattern in the given directory.
258
312
  // Supports:
259
313
  // - `*.json` (default, broadened to also match `*.csv`)
@@ -335,7 +389,10 @@ function parseCsvToRaw(text, config) {
335
389
 
336
390
  // Hoist participant ID from the first row to the top level so Shape-1 ingest
337
391
  // finds it via raw[pidField]. Falls back to 'unknown' if the column isn't there.
338
- const participantId = rows[0]?.[pidField] ?? 'unknown';
392
+ // getByPath supports dotted paths into JSON-parsed cells (e.g. a "metadata"
393
+ // column that held a stringified object); Shape-1's flat-key-first lookup
394
+ // then finds the hoisted value under the same (possibly dotted) key name.
395
+ const participantId = getByPath(rows[0], pidField) ?? 'unknown';
339
396
 
340
397
  return {
341
398
  [pidField]: participantId,
package/src/cli/init.js CHANGED
@@ -16,7 +16,7 @@ export async function runInit() {
16
16
 
17
17
  // Minimal config — just the fields every project needs to set.
18
18
  // filePattern defaults to JSON-or-CSV because jsPsych's `.csv()` save is
19
- // the most common format we see in the wild (rule-gallery uses JSON; most
19
+ // the most common format we see in the wild (Shape-2 producers use JSON; most
20
20
  // jsPsych setups use CSV).
21
21
  const config = {
22
22
  dataDir: './data',
@@ -0,0 +1,36 @@
1
+ // src/cli/preview-entry.js
2
+ // Browser-preview surface (0.7.2): the pure subset of the CLI pipeline the
3
+ // in-browser results screen (demo/results.js) needs to run the same
4
+ // ingest → analyze → render steps the real CLI runs, without any Node APIs.
5
+ // Bundled by tools/build-preview-core.mjs into demo/preview-core.js (esbuild,
6
+ // platform: browser) — every re-export below must resolve to a module with
7
+ // no fs/path/zlib/papaparse imports, or the bundle step fails with an
8
+ // unresolved-import error (the empirical gate the 0.7.2 scope relies on).
9
+ //
10
+ // extractIntegrityData is NOT re-exported from src/cli/ingest.js here even
11
+ // though that's its public home — ingest.js imports Node's `fs`/`zlib` at
12
+ // module scope for the file-discovery/CSV/replay-artifact machinery that
13
+ // surrounds it, so bundling ingest.js itself pulls those in. extract-core.js
14
+ // is the pure extraction the ingest.js module re-exports from (0.7.2), and
15
+ // is what this entry point bundles instead.
16
+
17
+ export { renderIndexHtml } from './renderers/html-index-core.js';
18
+ export { computeSummary } from './analyzers/summary.js';
19
+ export { detectEdgeExits } from './analyzers/edge-exit.js';
20
+ export { rankTriage } from './analyzers/triage.js';
21
+ export { extractIntegrityData } from './extract-core.js';
22
+ // buildViewerModel: pure wire->viewer conversion (src/replay/viewer-model.js,
23
+ // see that file's docblock) — the demo's results build needs it to construct
24
+ // the visitor's replay viewer-model in-browser (demo/results.js, C2) without
25
+ // re-implementing the time-conversion logic. No Node APIs, so it bundles
26
+ // cleanly here.
27
+ export { buildViewerModel } from '../replay/viewer-model.js';
28
+
29
+ // The three pure plot cores (0.7.2-style extraction — see each file's own
30
+ // docblock): drawSessionTimeline, drawTrajectoryGrid, drawTypingProfile.
31
+ // Each takes an injected createCanvas factory instead of importing the
32
+ // `canvas` package directly, so they bundle cleanly here too. Consumed by
33
+ // demo/plot-adapter.js to render the visitor's own plots in-browser.
34
+ export { drawSessionTimeline } from './renderers/session-timeline-core.js';
35
+ export { drawTrajectoryGrid } from './renderers/trajectories-core.js';
36
+ export { drawTypingProfile } from './renderers/typing-profile-core.js';
@@ -12,41 +12,40 @@ export async function renderEventLog(participants, config) {
12
12
  const rows = [];
13
13
 
14
14
  for (const p of participants) {
15
+ // Collect this participant's events with their session-absolute timestamp,
16
+ // then sort chronologically before serializing. All event timestamps are on
17
+ // the same performance.now() clock (copy/paste/drop/synthetic use `e.t`,
18
+ // tab-aways use `e.start`), so a single numeric sort interleaves them
19
+ // correctly. Without this the rows came out in loop/category order
20
+ // (all pastes, then all copies, …), contradicting the "chronological" docs.
21
+ const pEvents = [];
15
22
  for (const trial of p.trials) {
16
23
  const trialId = trial.trialId || trial.ruleId || '?';
17
24
 
18
- // Paste events
19
25
  for (const e of (trial.pasteEvents || [])) {
20
- rows.push(formatRow(p.participantId, trialId, 'paste', e.t, '', e.text));
26
+ pEvents.push({ ts: e.t, row: formatRow(p.participantId, trialId, 'paste', e.t, '', e.text) });
21
27
  }
22
-
23
- // Copy events
24
28
  for (const e of (trial.copyEvents || [])) {
25
- rows.push(formatRow(p.participantId, trialId, 'copy', e.t, '', ''));
29
+ pEvents.push({ ts: e.t, row: formatRow(p.participantId, trialId, 'copy', e.t, '', '') });
26
30
  }
27
-
28
- // Drop events
29
31
  for (const e of (trial.dropEvents || [])) {
30
- rows.push(formatRow(p.participantId, trialId, 'drop', e.t, '', e.text));
32
+ pEvents.push({ ts: e.t, row: formatRow(p.participantId, trialId, 'drop', e.t, '', e.text) });
31
33
  }
32
-
33
- // Synthetic insertions (text appeared without keystrokes)
34
34
  for (const e of (trial.syntheticInsertions || [])) {
35
- rows.push(formatRow(p.participantId, trialId, 'synthetic', e.t, '', e.text));
35
+ pEvents.push({ ts: e.t, row: formatRow(p.participantId, trialId, 'synthetic', e.t, '', e.text) });
36
36
  }
37
-
38
- // Tab-away events. timestamp uses the session-absolute `start` (same
39
- // performance.now() clock the copy/paste `t` field uses), so all rows
40
- // in this CSV share one chronological scale. The `text` column carries
41
- // the tab-away type (windowBlur / visibilityChange / etc.) so analysts
42
- // can filter by trigger.
37
+ // Tab-away timestamp uses the session-absolute `start` (the `text` column
38
+ // carries the trigger type — windowBlur / visibilityChange / etc. — so
39
+ // analysts can filter by trigger).
43
40
  for (const e of (trial.tabAwayEvents || [])) {
44
- rows.push(formatRow(p.participantId, trialId, 'tabAway', e.start, e.duration_ms, e.type || ''));
41
+ pEvents.push({ ts: e.start, row: formatRow(p.participantId, trialId, 'tabAway', e.start, e.duration_ms, e.type || '') });
45
42
  }
46
43
  }
44
+ // Stable chronological sort; events with no usable timestamp sort last.
45
+ pEvents.sort((a, b) => (a.ts ?? Infinity) - (b.ts ?? Infinity));
46
+ for (const e of pEvents) rows.push(e.row);
47
47
  }
48
48
 
49
- // Sort chronologically within each participant
50
49
  const csv = [header, ...rows].join('\n') + '\n';
51
50
  const outPath = join(config.outputDir, 'event-log.csv');
52
51
  writeFileSync(outPath, csv);
@@ -5,6 +5,7 @@
5
5
 
6
6
  import { writeFileSync } from 'fs';
7
7
  import { join } from 'path';
8
+ import { countSidebarOpenings } from '../analyzers/summary.js';
8
9
 
9
10
  export async function renderExtensions(participants, config) {
10
11
  const header = 'participantId,detectionType,name,details';
@@ -20,9 +21,17 @@ export async function renderExtensions(participants, config) {
20
21
  rows.push(`${pid},extension,${escapeCSV(name)},`);
21
22
  }
22
23
 
23
- // Sidebar detection
24
- const hasSidebar = p.trials.some(t => (t.sidebarGapPx || 0) > 0);
25
- if (hasSidebar) {
24
+ // Sidebar detection. Prefer the current library's session-level
25
+ // sidebarEvents (the monitor records sidebars session-scoped, not per-trial);
26
+ // fall back to the legacy per-trial sidebarGapPx only when no session events
27
+ // exist (pre-session / Shape-2 legacy data). countSidebarOpenings() counts
28
+ // distinct openings (collapsing the paired open/close records and the
29
+ // innerWidth_delta + layout_compression double-detection), matching the
30
+ // summary/triage count.
31
+ const sidebarOpens = countSidebarOpenings(p.session?.sidebarEvents);
32
+ if (sidebarOpens > 0) {
33
+ rows.push(`${pid},sidebar,browser_sidebar,${sidebarOpens} open event${sidebarOpens === 1 ? '' : 's'}`);
34
+ } else if (p.trials.some(t => (t.sidebarGapPx || 0) > 0)) {
26
35
  const maxGap = Math.max(...p.trials.map(t => t.sidebarGapPx || 0));
27
36
  rows.push(`${pid},sidebar,browser_sidebar,${maxGap}px gap`);
28
37
  }