cyborg-hunter 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -32,33 +32,45 @@ export function rankTriage(summaries, edgeExits, config) {
32
32
  return triageList;
33
33
  }
34
34
 
35
- // Combines all signal categories into a single numeric score.
36
- // Hard signals add +100 to ensure they always rank above soft-only participants.
37
- // Session-level signals (sidebar events, AI extensions, keyboard shortcuts, layout
38
- // shifts, zoom changes) are surfaced from session metadata when available.
39
- function computeTriageScore(summary, edgeExitCount) {
40
- // Prefer session-authoritative soft score; fall back to per-trial sum.
41
- let score = summary.authoritativeSoftScore ?? summary.totalSoftScore;
42
- if (summary.hardTriggered) score += 100;
43
- score += edgeExitCount * 3;
44
-
45
- // Session-level: +5 per AI extension detected (canonical name: aiExtensionsFound)
35
+ // Decomposes the composite triage score into [label, contribution] pairs.
36
+ // Used by both computeTriageScore (sums them) and the HTML renderer (displays them
37
+ // with bars). Keeping the formula here means the score breakdown in the report can
38
+ // never silently disagree with the ranked score it explains.
39
+ //
40
+ // IMPORTANT: this function is **strictly behavior-preserving** vs the previous
41
+ // inline computeTriageScore. The defensive `|| 0` fallbacks below match the
42
+ // original exactly — no extra ones added, no original ones removed. Adding
43
+ // new defenses changes how malformed data (e.g. a non-numeric softScore object)
44
+ // is rendered into csv/md/png outputs, even when the well-formed-data score
45
+ // would be the same. We promised existing users no surprises in this refactor.
46
+ export function decomposeScore(summary, edgeExitCount, hardTriggered) {
46
47
  const aiExtensions = summary.aiExtensionsFound || summary.extensionsDetected || [];
47
- score += aiExtensions.length * 5;
48
-
49
- // Session-level: +3 per sidebar event, capped at 15 (≈ 5 events max contribution)
50
48
  const sidebarCount = summary.sidebarEventCount ?? (summary.sidebarDetected ? 1 : 0);
51
- score += Math.min(sidebarCount * 3, 15);
52
-
53
- // Session-level: +2 per DevTools-ish keyboard shortcut
54
- score += (summary.keyboardShortcutCount || 0) * 2;
55
-
56
- // Session-level: +1 per layout shift / zoom change (workspace perturbations)
57
- score += (summary.layoutShiftCount || 0);
58
- score += (summary.zoomChangeCount || 0);
49
+ const base = summary.authoritativeSoftScore ?? summary.totalSoftScore;
50
+ return [
51
+ ['soft', base],
52
+ ['hard', hardTriggered ? 100 : 0],
53
+ ['edge', edgeExitCount * 3],
54
+ ['ext', aiExtensions.length * 5],
55
+ ['sidebar', Math.min(sidebarCount * 3, 15)],
56
+ ['kb', (summary.keyboardShortcutCount || 0) * 2],
57
+ ['layout', (summary.layoutShiftCount || 0)],
58
+ ['zoom', (summary.zoomChangeCount || 0)],
59
+ ['synth', summary.totalSyntheticInsertions * 4],
60
+ ['foreign', summary.totalForeignInputEvents * 3],
61
+ ];
62
+ }
59
63
 
60
- score += summary.totalSyntheticInsertions * 4;
61
- score += summary.totalForeignInputEvents * 3;
64
+ // computeTriageScore stays an internal helper. It accumulates the decomposition
65
+ // in the original iteration order, starting from the soft base directly (NOT
66
+ // from a defensive 0) — same `let score = base; score += hardTerm; score +=
67
+ // edgeTerm; …` behaviour as the pre-refactor inline version. For malformed data
68
+ // where `base` is a non-number (e.g. an object), every subsequent `+=` becomes
69
+ // a string concat — exactly as it did before this refactor.
70
+ function computeTriageScore(summary, edgeExitCount) {
71
+ const terms = decomposeScore(summary, edgeExitCount, summary.hardTriggered);
72
+ let score = terms[0][1];
73
+ for (let i = 1; i < terms.length; i++) score += terms[i][1];
62
74
  return score;
63
75
  }
64
76
 
package/src/cli/config.js CHANGED
@@ -23,7 +23,7 @@ export function loadConfig(cliArgs) {
23
23
  fileConfig = JSON.parse(readFileSync(configPath, 'utf8'));
24
24
  console.log(`Loaded config from ${configPath}`);
25
25
  } catch (e) {
26
- console.warn(`Warning: Failed to parse ${configPath}: ${e.message}`);
26
+ console.warn(`[cyborg-hunter] failed to parse ${configPath}: ${e.message}`);
27
27
  }
28
28
  }
29
29
 
@@ -43,7 +43,7 @@ export function loadConfig(cliArgs) {
43
43
 
44
44
  // Validate config file keys — warns about typos
45
45
  const warnings = validateConfig(fileConfig);
46
- warnings.forEach(w => console.warn(`Warning: ${w}`));
46
+ warnings.forEach(w => console.warn(`[cyborg-hunter] ${w}`));
47
47
 
48
48
  return config;
49
49
  }
package/src/cli/ingest.js CHANGED
@@ -1,17 +1,14 @@
1
- // src/cli/ingest.js
2
- // Loads JSON or CSV data files, extracts integrity data, validates schema.
1
+ // Loads JSON or CSV participant files, extracts integrity data, validates schema.
3
2
  //
4
- // Handles three input pathways:
5
- // 1. JSON, Shape 1 — one file per participant with a trials array containing
6
- // an integrity sub-object: { participantId: 'P1', trials: [{ integrity: {...} }] }
7
- // 2. JSON, Shape 2 — rule-gallery format: { metadata: { subjectId: 'P1' },
8
- // responses: [{ mouseTrack: [...], tabAwayEvents: [...] }] }
3
+ // Three input shapes:
4
+ // 1. JSON Shape 1 — { participantId: 'P1', trials: [{ integrity: {...} }] }
5
+ // jsPsych extension data with one file per participant.
6
+ // 2. JSON Shape 2 — { metadata: {...}, responses: [{ mouseTrack, tabAwayEvents, ... }] }
7
+ // The original rule-gallery format. Signal data lives flat on each response
8
+ // (no `integrity` wrapper). Pre-dates the standalone library.
9
9
  // 3. CSV — jsPsych default save format. Each row is a trial; nested objects
10
10
  // (integrity, integritySession, integrityScore) are JSON-stringified into
11
- // single cells by Papa Parse's unparse. We unwrap them back into objects,
12
- // then route through Shape 1.
13
- //
14
- // Legacy field names (mouseTrack) are mapped to CyborgHunter schema names (mouseEvents).
11
+ // single cells. We unwrap them and route through Shape 1.
15
12
 
16
13
  import { readFileSync, readdirSync } from 'fs';
17
14
  import { join, extname } from 'path';
@@ -64,16 +61,16 @@ export function extractIntegrityData(raw, config) {
64
61
 
65
62
  let trials = [];
66
63
 
67
- // Shape 1: { trials: [{ integrity: {...} }, ...] }
68
- // Each trial has an `integrity` sub-object added by CyborgHunter's endTrial()
64
+ // Shape 1: { trials: [{ integrity: {...} }, ...] } — jsPsych extension data.
65
+ // Each trial has an `integrity` sub-object added by CyborgHunter.endTrial().
69
66
  if (Array.isArray(raw.trials)) {
70
67
  trials = raw.trials
71
68
  .filter(t => t[intField])
72
69
  .map(t => t[intField]);
73
- // P3: jsPsych extension data carries trialStart_perfNow on every trial
74
- // (set by the wrapper's on_load), so the fast path subtracts directly
75
- // and gives exact trial-relative startRel_ms. Without this, renderers
76
- // see only session-absolute `start` and plot tab-away markers off-canvas.
70
+ // jsPsych extension data carries trialStart_perfNow per trial (set by the
71
+ // wrapper's on_load), so normalization takes the exact-subtraction fast
72
+ // path. Without it, renderers see only session-absolute `start` values
73
+ // and plot tab-away markers far off the per-trial axis.
77
74
  normalizeTabAwayTimestamps(trials, raw);
78
75
  }
79
76
  // Shape 2: { responses: [{ mouseTrack, tabAwayEvents, ... }] } (rule-gallery legacy format)
@@ -152,30 +149,27 @@ const LEGACY_FIELD_MAP = {
152
149
  };
153
150
 
154
151
  // Normalizes tabAwayEvents[*].start from session-relative performance.now()
155
- // (milliseconds since browser navigation) to trial-relative milliseconds
156
- // (milliseconds since the trial began), stored as `startRel_ms`.
157
- //
158
- // The rule-gallery data schema stores mouseTrack entries with a per-trial
159
- // relative `t` field (zero-based per trial), but tabAwayEvents are stored
160
- // with session-relative `start` values. Without normalization, renderers
161
- // using the same x-scale plot tab-aways hundreds of thousands of ms off-canvas.
152
+ // (ms since browser navigation) to trial-relative ms (ms since this trial
153
+ // began), stored as a new `startRel_ms` field. Renderers plot per-trial
154
+ // against trial-relative time; without normalization, tab-away markers
155
+ // land tens of thousands of ms off the right edge — invisible.
162
156
  //
163
- // Algorithm: trial boundaries exist in wall-clock (response.timestamp is the
164
- // trial END wall-clock, responseTime_ms is trial duration). We infer a
165
- // participant-level `sessionOffset_ms` = (performance.now() value at session
166
- // start) by using the relationship:
167
- // tab.start == sessionOffset + trialStart_wallclockRel + positionInTrial
168
- // where positionInTrial ∈ [0, responseTime_ms]. We estimate sessionOffset as
169
- // the median of (firstTab.start - trialStartRel) across trials, clipped to
170
- // the valid range implied by each (trial, firstTab) pair.
157
+ // Two paths:
158
+ // FAST PATH — every trial has a per-trial performance.now() anchor (either
159
+ // trialStart_perfNow set by the jsPsych wrapper, or startTime set by the
160
+ // standalone monitor). Subtract directly; result is exact.
161
+ // ESTIMATOR PATH — rule-gallery / pre-monitor data has neither anchor.
162
+ // We infer the session-start performance.now() value from the
163
+ // relationship between trial-end wall-clocks (`timestamp`),
164
+ // `responseTime_ms`, and the first tab-away's session-relative `start`,
165
+ // then subtract the inferred offset to get a trial-relative value.
166
+ // Less precise than the fast path; used only when nothing better exists.
171
167
  function normalizeTabAwayTimestamps(trials, raw) {
172
- // P3 fast path: if every trial carries a monitor-supplied trialStart_perfNow
173
- // (same clock as tabAwayEvents[*].start), compute startRel_ms by direct
174
- // subtraction. This skips the heuristic estimator entirely and is exact.
175
- const everyTrialHasPerfNow = trials.length > 0 && trials.every(t =>
168
+ // Fast path — every trial has a usable anchor.
169
+ const everyTrialHasAnchor = trials.length > 0 && trials.every(t =>
176
170
  typeof (t.trialStart_perfNow ?? t.startTime) === 'number'
177
171
  );
178
- if (everyTrialHasPerfNow) {
172
+ if (everyTrialHasAnchor) {
179
173
  for (const trial of trials) {
180
174
  const tabs = trial.tabAwayEvents;
181
175
  if (!Array.isArray(tabs) || tabs.length === 0) continue;
@@ -188,14 +182,15 @@ function normalizeTabAwayTimestamps(trials, raw) {
188
182
  return;
189
183
  }
190
184
 
191
- // Legacy estimator path — used when the detector didn't record a
192
- // per-trial performance.now() anchor (e.g., rule-gallery pre-P3 data).
185
+ // Estimator path. Need a session-start wall-clock anchor; if there isn't
186
+ // one, we have no way to relate trial-end timestamps to a 0 reference.
193
187
  const sessStartIso = raw.metadata?.startTime;
194
188
  if (!sessStartIso) return;
195
189
  const sessStartMs = Date.parse(sessStartIso);
196
190
  if (!Number.isFinite(sessStartMs)) return;
197
191
 
198
- // Precompute wall-clock trial start (ms since session start) for each trial.
192
+ // For each trial, compute trial-start in ms-since-session-start (wall-clock).
193
+ // trial-end is response.timestamp; trial-start = trial-end − duration.
199
194
  const trialStartRels = trials.map(t => {
200
195
  if (!t.timestamp || t.responseTime_ms == null) return null;
201
196
  const endMs = Date.parse(t.timestamp);
@@ -203,21 +198,26 @@ function normalizeTabAwayTimestamps(trials, raw) {
203
198
  return (endMs - sessStartMs) - t.responseTime_ms;
204
199
  });
205
200
 
206
- // Collect offset candidates from each trial that has ≥1 tab-away.
207
- // For trial i with first tab at T and trialStartRel S, responseTime R:
208
- // offset ∈ [T - (S + R), T - S] (tab must lie within [S, S+R])
209
- // Median of midpoints is a robust estimator.
201
+ // For each trial that has ≥1 tab-away, estimate the session offset:
202
+ // sessionOffset = (performance.now value at session start, in ms)
203
+ // The first tab-away of the trial must fall inside the trial's response
204
+ // window, so:
205
+ // firstTabStart ∈ [sessionOffset + trialStartRel,
206
+ // sessionOffset + trialStartRel + trialDuration]
207
+ // Solving for sessionOffset gives an interval; we use the midpoint as
208
+ // each trial's candidate, then take the median across trials as the
209
+ // robust estimate. This is approximate but converges quickly with even
210
+ // a few tab-away-bearing trials.
210
211
  const candidates = [];
211
212
  for (let i = 0; i < trials.length; i++) {
212
213
  const tabs = trials[i].tabAwayEvents;
213
214
  if (!Array.isArray(tabs) || tabs.length === 0) continue;
214
- const S = trialStartRels[i];
215
- const R = trials[i].responseTime_ms;
216
- if (S == null || R == null) continue;
217
- const T = tabs[0].start;
218
- if (typeof T !== 'number') continue;
219
- // Midpoint of the valid offset interval for this trial's first tab.
220
- candidates.push((T - S - R / 2));
215
+ const trialStartRel = trialStartRels[i];
216
+ const trialDurationMs = trials[i].responseTime_ms;
217
+ if (trialStartRel == null || trialDurationMs == null) continue;
218
+ const firstTabStart = tabs[0].start;
219
+ if (typeof firstTabStart !== 'number') continue;
220
+ candidates.push(firstTabStart - trialStartRel - trialDurationMs / 2);
221
221
  }
222
222
 
223
223
  if (candidates.length === 0) return;
@@ -225,16 +225,17 @@ function normalizeTabAwayTimestamps(trials, raw) {
225
225
  candidates.sort((a, b) => a - b);
226
226
  const sessionOffset = candidates[Math.floor(candidates.length / 2)];
227
227
 
228
- // Apply normalization — add startRel_ms, keep original start.
228
+ // Apply normalization. Original `start` is preserved alongside `startRel_ms`
229
+ // so consumers can still see the absolute time if they want it.
229
230
  for (let i = 0; i < trials.length; i++) {
230
231
  const tabs = trials[i].tabAwayEvents;
231
232
  if (!Array.isArray(tabs) || tabs.length === 0) continue;
232
- const S = trialStartRels[i];
233
- if (S == null) continue;
233
+ const trialStartRel = trialStartRels[i];
234
+ if (trialStartRel == null) continue;
234
235
  trials[i].tabAwayEvents = tabs.map(ta => ({
235
236
  ...ta,
236
- startRel_ms: (typeof ta.start === 'number')
237
- ? ta.start - sessionOffset - S
237
+ startRel_ms: typeof ta.start === 'number'
238
+ ? ta.start - sessionOffset - trialStartRel
238
239
  : null
239
240
  }));
240
241
  }