cyborg-hunter 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/CHANGELOG.md +110 -0
  2. package/CITATION.cff +29 -0
  3. package/LICENSE +21 -0
  4. package/README.md +78 -21
  5. package/package.json +10 -3
  6. package/src/cli/analyzers/edge-exit.js +4 -1
  7. package/src/cli/analyzers/phase-scope.js +83 -0
  8. package/src/cli/analyzers/summary.js +161 -28
  9. package/src/cli/analyzers/triage.js +59 -27
  10. package/src/cli/config.js +26 -1
  11. package/src/cli/ingest.js +623 -41
  12. package/src/cli/init.js +1 -1
  13. package/src/cli/renderers/event-log.js +18 -19
  14. package/src/cli/renderers/extensions.js +12 -3
  15. package/src/cli/renderers/html-index.js +163 -24
  16. package/src/cli/renderers/replay-assets.js +177 -0
  17. package/src/cli/renderers/replay-viewer.client.js +1022 -0
  18. package/src/cli/renderers/session-timeline.js +917 -0
  19. package/src/cli/renderers/summary-csv.js +5 -0
  20. package/src/cli/renderers/trajectories.js +68 -7
  21. package/src/cli/renderers/triage-md.js +10 -4
  22. package/src/cli/renderers/typing-profile.js +7 -1
  23. package/src/cli/report.js +42 -8
  24. package/src/core/monitor.js +60 -7
  25. package/src/core/scoring.js +11 -2
  26. package/src/core/signals/browser.js +51 -18
  27. package/src/core/signals/clipboard.js +10 -2
  28. package/src/core/signals/dom-protection.js +9 -0
  29. package/src/core/signals/focus.js +16 -2
  30. package/src/jspsych/extension-cyborg-hunter-replay.js +135 -0
  31. package/src/jspsych/extension-cyborg-hunter.js +9 -2
  32. package/src/jspsych/extension-guard-friction.js +32 -12
  33. package/src/jspsych/extension-guard-honeypot.js +25 -1
  34. package/src/replay/capture-dom.js +575 -0
  35. package/src/replay/capture-trace.js +468 -0
  36. package/src/replay/index.js +104 -0
  37. package/src/replay/persistence.js +141 -0
  38. package/src/replay/recorder.js +315 -0
  39. package/src/replay/serializer.js +119 -0
  40. package/src/shared/constants.js +12 -6
  41. package/src/shared/schema.js +5 -0
  42. package/src/shared/validation.js +55 -0
  43. package/dist/cyborg-hunter.esm.js +0 -1527
  44. package/dist/cyborg-hunter.min.js +0 -6
  45. package/dist/extension-cyborg-hunter.js +0 -1
  46. package/dist/extension-guard-friction.js +0 -36
  47. package/dist/extension-guard-honeypot.js +0 -1
  48. package/src/cli/renderers/tab-timeline.js +0 -149
@@ -10,6 +10,13 @@ export function computeSummary(participants, config) {
10
10
  export function computeParticipantSummary(participant, config) {
11
11
  const trials = participant.trials;
12
12
  const n = trials.length;
13
+ // Set by applyPhaseScope (config.phaseScope). Session-level aggregates
14
+ // (tabAwaySums, the authoritative soft score, anyHardTriggered) cover the
15
+ // WHOLE session, so a phase-scoped summary must ignore them and aggregate
16
+ // from the scoped trials instead. Ambient environment signals (sidebar,
17
+ // keyboard shortcuts, viewport shifts, zoom) stay session-wide by design —
18
+ // see phase-scope.js.
19
+ const phaseScoped = participant.phaseScoped === true;
13
20
 
14
21
  // Three-way tab-away duration bins for display. The 3s boundary matches
15
22
  // config.thresholds.tabAwayDurationMs — the scoring engine's soft-score cutoff.
@@ -17,8 +24,16 @@ export function computeParticipantSummary(participant, config) {
17
24
  // flicker (<3s): OS/browser noise, notification glances — no scoring penalty
18
25
  // medium (3-10s): deliberate brief checks — count toward soft score
19
26
  // long (≥10s): substantive absences — count toward soft score
20
- const scoringCutoff_ms = config?.thresholds?.tabAwayDurationMs ?? 3000;
27
+ // Prefer the thresholds the LIBRARY actually screened this participant with
28
+ // (persisted in session.config.thresholds by finalize) so the report bins their
29
+ // data the same way the runtime did — e.g. a strict participant's 5s tab-away
30
+ // cutoff. Fall back to an analyst-side CLI threshold, then the default.
31
+ const savedThresholds = participant.session?.config?.thresholds || {};
32
+ const scoringCutoff_ms =
33
+ savedThresholds.tabAwayDurationMs ?? config?.thresholds?.tabAwayDurationMs ?? 3000;
21
34
  const longCutoff_ms = 10000;
35
+ const typingCutoff_cps =
36
+ savedThresholds.typingSpeedCps ?? config.typingSpeedThreshold_cps ?? 10;
22
37
 
23
38
  return {
24
39
  participantId: participant.participantId,
@@ -29,25 +44,26 @@ export function computeParticipantSummary(participant, config) {
29
44
  totalCopyEvents: sum(trials, t => (t.copyEvents || []).length),
30
45
  totalDropEvents: sum(trials, t => (t.dropEvents || []).length),
31
46
 
32
- // Tab-away — leaving the experiment page (visibility change or blur)
33
- totalTabAways: sum(trials, t => (t.tabAwayEvents || []).length),
34
- tabAwayFlickerCount: sum(trials, t =>
35
- (t.tabAwayEvents || []).filter(e => (e.duration_ms || 0) < scoringCutoff_ms).length),
36
- tabAwayMediumCount: sum(trials, t =>
37
- (t.tabAwayEvents || []).filter(e => {
38
- const d = e.duration_ms || 0;
39
- return d >= scoringCutoff_ms && d < longCutoff_ms;
40
- }).length),
41
- tabAwayLongCount: sum(trials, t =>
42
- (t.tabAwayEvents || []).filter(e => (e.duration_ms || 0) >= longCutoff_ms).length),
43
- totalTabAwayDuration_ms: sum(trials, t =>
44
- (t.tabAwayEvents || []).reduce((s, e) => s + (e.duration_ms || 0), 0)),
45
- trialsWithTabAway: trials.filter(t => (t.tabAwayEvents || []).length > 0).length,
47
+ // Tab-away — leaving the experiment page (visibility change or blur).
48
+ // Prefer session-level durations (cyborgHunter.tabAwaySums) which capture
49
+ // events anywhere in the session, including between trials and during
50
+ // gallery / consent / comprehension phases. Per-trial trials[*].tabAwayEvents
51
+ // only covers intervals while a trial is active, so it misses anything
52
+ // that happens during a study/learning phase — which is exactly where many
53
+ // participants go off-task.
54
+ ...computeTabAwayCounts(phaseScoped ? null : participant.session?.tabAwaySums, trials, scoringCutoff_ms, longCutoff_ms),
46
55
 
47
56
  // Typing speed — suspiciously fast typing may indicate paste or autocomplete
48
57
  meanTypingSpeed: avg(trials, t => t.charsPerSec || 0),
49
58
  trialsWithFastTyping: trials.filter(t =>
50
- (t.charsPerSec || 0) > (config.typingSpeedThreshold_cps || 10)).length,
59
+ (t.charsPerSec || 0) > typingCutoff_cps).length,
60
+
61
+ // The runtime preset that screened this participant (when persisted).
62
+ preset: participant.session?.config?.preset ?? null,
63
+ // The resolved per-participant tab-away cutoff (ms) used for the bins above,
64
+ // so the triage reason can describe the bins with this participant's actual
65
+ // threshold (e.g. 5s for strict) instead of a hardcoded 3s.
66
+ tabAwayCutoffMs: scoringCutoff_ms,
51
67
 
52
68
  // Mouse — low event count or very efficient paths can indicate bot activity
53
69
  meanMouseEvents: avg(trials, t => (t.mouseEvents || []).length),
@@ -69,31 +85,148 @@ export function computeParticipantSummary(participant, config) {
69
85
  // Scoring — accumulated soft score and whether any hard signal fired
70
86
  totalSoftScore: sum(trials, t => t.trialSoftScore || 0),
71
87
 
72
- // Session-level fields (preferred over per-trial aggregation when available)
73
- sidebarEventCount: participant.session?.sidebarEvents?.length ?? trials.filter(t => (t.sidebarGapPx || 0) > 0).length,
88
+ // Session-level fields (preferred over per-trial aggregation when available).
89
+ // Count sidebar OPENINGS, not raw log entries. The runtime records a separate
90
+ // "opened" + "closed" entry per incident, and the innerWidth_delta and
91
+ // layout_compression checks can BOTH fire for one physical sidebar (same poll
92
+ // tick → same timestamp). countSidebarOpenings() collapses opens within one
93
+ // poll window into a single incident, so sidebarEventCount matches the number
94
+ // of times a sidebar actually appeared. The triage score weights "3 × sidebar"
95
+ // off this, so over-counting here would over-score.
96
+ sidebarEventCount: participant.session?.sidebarEvents
97
+ ? countSidebarOpenings(participant.session.sidebarEvents)
98
+ : trials.filter(t => (t.sidebarGapPx || 0) > 0).length,
74
99
  aiExtensionsFound: participant.session?.aiExtensionsFound || participant.trials[0]?.extensionsDetected || [],
75
100
  keyboardShortcutCount: participant.session?.keyboardShortcuts?.length || 0,
76
- layoutShiftCount: participant.session?.layoutShifts?.length || 0,
101
+ // Canonical key since 0.6.1 is session.viewportWidthShifts (the signal
102
+ // measures viewport-width changes, not Web-Vitals CLS); layoutShifts is
103
+ // the deprecated alias older payloads carry. The summary field and the
104
+ // layout_shift_count CSV column keep their names for pipeline stability.
105
+ layoutShiftCount: (participant.session?.viewportWidthShifts ?? participant.session?.layoutShifts)?.length || 0,
77
106
  zoomChangeCount: participant.session?.zoomChanges?.length || 0,
78
107
  extensionInjectionCount: participant.session?.extensionInjections?.length || 0,
79
108
  devToolsEventCount: participant.session?.devToolsEvents?.length || 0,
80
- authoritativeSoftScore: participant.score?.softScore ?? null,
81
-
82
- // Authoritative hard-flag: prefer session's anyHardTriggered, else fall back to per-trial trialHits
83
- // v3.1.0+ per-trial data: trialSignals.hard.*.trialHits counts this-trial occurrences
84
- // (no .triggered boolean at per-trial level — that lives on session integrityScore).
85
- // A participant is hard-flagged if any trial had any hard-signal hit.
86
- hardTriggered: participant.score?.anyHardTriggered
87
- ?? trials.some(t =>
88
- t.trialSignals?.hard && Object.values(t.trialSignals.hard).some(s => (s.trialHits || 0) > 0)),
109
+ // Whole-session score — unusable for a phase-scoped verdict, so scoping
110
+ // nulls it and the scoped per-trial totalSoftScore drives soft flagging.
111
+ authoritativeSoftScore: phaseScoped ? null : (participant.score?.softScore ?? null),
112
+ // The soft-score threshold the LIBRARY actually screened this participant
113
+ // against (their preset's value, saved in the session report). Carried so the
114
+ // CLI can flag using each participant's real threshold rather than a generic
115
+ // analyst-side default.
116
+ softScoreThreshold: participant.score?.softScoreThreshold ?? null,
117
+
118
+ // Guard-honeypot self-disclosure (visible bait). null/'' when the honeypot
119
+ // was not used or the participant left it untouched.
120
+ honeypotAiUse: participant.honeypot?.aiUse ?? null,
121
+ honeypotAiReport: participant.honeypot?.aiReport ?? '',
122
+
123
+ // Authoritative hard-flag: prefer the session's anyHardTriggered. Fallback
124
+ // (no session score available) must use the CUMULATIVE sessionTotal vs the
125
+ // signal's countThreshold — NOT per-trial trialHits. A single paste
126
+ // (trialHits=1) when the threshold is 2 is NOT a hard trigger; using
127
+ // trialHits>0 fabricated hard flags and inflated the hard cohort whenever
128
+ // finalize() was missed. trialSignals.hard.*.sessionTotal is the running
129
+ // count at that trial, so any trial crossing its threshold = hard. When that
130
+ // data is absent, the signal can't promote the participant to hard.
131
+ // Phase-scoped verdicts recount raw hard-signal events within the scoped
132
+ // trials (scopedHardTriggered) — both the session's anyHardTriggered and
133
+ // trialSignals' sessionTotal accumulate across the whole session, so
134
+ // either would leak out-of-scope events into the verdict.
135
+ hardTriggered: phaseScoped
136
+ ? scopedHardTriggered(trials)
137
+ : participant.score?.anyHardTriggered
138
+ ?? trials.some(t => {
139
+ const hard = t.trialSignals?.hard;
140
+ if (!hard) return false;
141
+ return Object.values(hard).some(s =>
142
+ s && typeof s.sessionTotal === 'number' && typeof s.countThreshold === 'number'
143
+ && s.sessionTotal >= s.countThreshold);
144
+ }),
89
145
 
90
146
  // Pass through metadata for downstream use
91
147
  metadata: participant.metadata || {}
92
148
  };
93
149
  }
94
150
 
151
+ // Re-derives the hard-flag verdict from raw event counts within a phase-scoped
152
+ // trial set. The countThreshold is taken from the trialSignals snapshots (any
153
+ // trial's — the runtime records the same threshold on every trial); signals
154
+ // with no recorded threshold can't promote the participant to hard.
155
+ function scopedHardTriggered(trials) {
156
+ const EVENT_FIELDS = { paste: 'pasteEvents', copy: 'copyEvents', drop: 'dropEvents' };
157
+ const counts = { paste: 0, copy: 0, drop: 0 };
158
+ const thresholds = {};
159
+ for (const t of trials) {
160
+ for (const [sig, field] of Object.entries(EVENT_FIELDS)) {
161
+ counts[sig] += (t[field] || []).length;
162
+ const th = t.trialSignals?.hard?.[sig]?.countThreshold;
163
+ if (typeof th === 'number') thresholds[sig] = th;
164
+ }
165
+ }
166
+ return Object.keys(counts).some(sig =>
167
+ typeof thresholds[sig] === 'number' && counts[sig] >= thresholds[sig]);
168
+ }
169
+
170
+ // Counts distinct sidebar OPENINGS from the session sidebarEvents log. The
171
+ // runtime can emit two "opened" entries for one physical sidebar (the
172
+ // innerWidth_delta and layout_compression checks fire in the same poll tick), so
173
+ // counting raw "opened" entries over-counts. Walk the open/close events as a
174
+ // state machine: an "opened" starts a new incident ONLY when transitioning from
175
+ // the closed state, so coincident double-detection opens (no intervening close)
176
+ // collapse to one, while a genuine open→close→open reopen counts as two — even a
177
+ // fast one. Events are ordered by timestamp when present (double-detection opens
178
+ // share a t and sort adjacent), else by array (push) order.
179
+ export function countSidebarOpenings(sidebarEvents) {
180
+ const events = (sidebarEvents || []).filter(e => e && (e.type === 'opened' || e.type === 'closed'));
181
+ const ordered = events.every(e => typeof e.t === 'number')
182
+ ? [...events].sort((a, b) => a.t - b.t)
183
+ : events;
184
+ let open = false;
185
+ let count = 0;
186
+ for (const e of ordered) {
187
+ if (e.type === 'opened') {
188
+ if (!open) { count++; open = true; }
189
+ } else {
190
+ open = false;
191
+ }
192
+ }
193
+ return count;
194
+ }
195
+
95
196
  // Helper: sum an array by applying fn to each element
96
197
  function sum(arr, fn) { return arr.reduce((s, x) => s + fn(x), 0); }
97
198
 
98
199
  // Helper: average an array by applying fn to each element
99
200
  function avg(arr, fn) { return arr.length === 0 ? 0 : sum(arr, fn) / arr.length; }
201
+
202
+ // Helper: derive all tab-away count fields from a session-level durations
203
+ // array (preferred — covers the whole session including gaps between trials
204
+ // and during gallery / consent / comprehension phases) or fall back to
205
+ // per-trial trials[*].tabAwayEvents (legacy — only covers active trials).
206
+ function computeTabAwayCounts(sessionDurs, trials, scoringCutoff_ms, longCutoff_ms) {
207
+ let durations;
208
+ if (Array.isArray(sessionDurs)) {
209
+ durations = sessionDurs.map(d => Number(d) || 0);
210
+ } else {
211
+ durations = [];
212
+ for (const t of trials) {
213
+ for (const e of (t.tabAwayEvents || [])) {
214
+ durations.push(Number(e.duration_ms) || 0);
215
+ }
216
+ }
217
+ }
218
+ // "Meaningful" tab-aways (medium + long) are those that count toward the soft
219
+ // score. The runtime soft-scoring rule is STRICT ( duration > cutoff — see
220
+ // scoring.js and constants.js "longer than this counts" ), so the bins use the
221
+ // same `>` boundary: a tab-away exactly at the cutoff is a flicker, not scored.
222
+ // This keeps the report's tab-away count identical to what the library screened.
223
+ return {
224
+ totalTabAways: durations.length,
225
+ tabAwayFlickerCount: durations.filter(d => d <= scoringCutoff_ms).length,
226
+ tabAwayMediumCount: durations.filter(d => d > scoringCutoff_ms && d < longCutoff_ms).length,
227
+ tabAwayLongCount: durations.filter(d => d >= longCutoff_ms).length,
228
+ totalTabAwayDuration_ms: durations.reduce((s, d) => s + d, 0),
229
+ // `trialsWithTabAway` is per-trial-defined; keep it tied to trials[*]
230
+ trialsWithTabAway: trials.filter(t => (t.tabAwayEvents || []).length > 0).length,
231
+ };
232
+ }
@@ -3,10 +3,14 @@
3
3
  // This is the "start here" document for manual review — the researcher
4
4
  // reads the triage list top-to-bottom, inspecting the most suspicious first.
5
5
  //
6
- // The triage score is a single number combining:
7
- // - Accumulated soft score from the scoring engine
8
- // - Hard signal flags (these dominate with +100)
9
- // - Edge-exit patterns, extension detection, synthetic insertions
6
+ // Ranking has two parts (see rankTriage + decomposeScore below):
7
+ // - tier: hard-triggered → soft-flagged → clean (primary sort key)
8
+ // - score: 5×paste + 5×copy + 3×sidebar-open + 1×tab-away (within a tier),
9
+ // where a "tab-away" is one longer than the participant's tab-away
10
+ // threshold (3s by default, 5s for the strict preset)
11
+ // Other signals (AI extensions, keyboard shortcuts, layout/zoom, edge-exits,
12
+ // synthetic insertions, foreign inputs, honeypot disclosure) are surfaced in the
13
+ // reason and detail panes but do NOT contribute to the score.
10
14
 
11
15
  export function rankTriage(summaries, edgeExits, config) {
12
16
  const triageList = summaries.map((s, i) => {
@@ -21,14 +25,31 @@ export function rankTriage(summaries, edgeExits, config) {
21
25
  score,
22
26
  reason,
23
27
  hardTriggered: s.hardTriggered,
24
- softFlagged: (s.authoritativeSoftScore ?? s.totalSoftScore) >= (config.scoring?.softScoreThreshold || 6),
28
+ // Threshold precedence: an explicit analyst-side CLI override
29
+ // (config.scoring.softScoreThreshold) wins for deliberate re-screening;
30
+ // otherwise use the participant's OWN saved threshold (what the library
31
+ // screened them against — e.g. 4 for the strict preset); else default 6.
32
+ // Using a hardcoded 6 here would mislabel strict-preset participants clean.
33
+ softFlagged: (s.authoritativeSoftScore ?? s.totalSoftScore) >=
34
+ (config.scoring?.softScoreThreshold ?? s.softScoreThreshold ?? 6),
25
35
  summary: s,
26
36
  edgeExitCount: totalEdgeExits
27
37
  };
28
38
  });
29
39
 
30
- // Sort descending — most suspicious participants appear first
31
- triageList.sort((a, b) => b.score - a.score);
40
+ // Sort tier-first (hard, then soft, then clean), and by score descending
41
+ // within a tier. A hard-triggered participant (paste/drop over its count
42
+ // threshold) is categorically more actionable than a high soft-signal count,
43
+ // so they must lead the "start here" triage.md even though the four-term
44
+ // score (paste/copy/sidebar/tab-away) no longer includes a hard-trigger term.
45
+ // This matches the HTML index's "Tier" sort (hard:0, soft:1, clean:2; score
46
+ // desc within tier).
47
+ const tierRank = (t) => (t.hardTriggered ? 0 : t.softFlagged ? 1 : 2);
48
+ triageList.sort((a, b) => {
49
+ const dt = tierRank(a) - tierRank(b);
50
+ if (dt !== 0) return dt;
51
+ return b.score - a.score;
52
+ });
32
53
  return triageList;
33
54
  }
34
55
 
@@ -37,27 +58,30 @@ export function rankTriage(summaries, edgeExits, config) {
37
58
  // with bars). Keeping the formula here means the score breakdown in the report can
38
59
  // never silently disagree with the ranked score it explains.
39
60
  //
40
- // IMPORTANT: this function is **strictly behavior-preserving** vs the previous
41
- // inline computeTriageScore. The defensive `|| 0` fallbacks below match the
42
- // original exactly — no extra ones added, no original ones removed. Adding
43
- // new defenses changes how malformed data (e.g. a non-numeric softScore object)
44
- // is rendered into csv/md/png outputs, even when the well-formed-data score
45
- // would be the same. We promised existing users no surprises in this refactor.
61
+ // Scoring policy (fixed 2026-06-01): only four signals contribute to the
62
+ // score —
63
+ // 5 × paste events
64
+ // 5 × copy events
65
+ // 3 × sidebar events (open cycles)
66
+ // 1 × tab-aways longer than the participant's tab-away threshold (3s default,
67
+ // 5s strict) — excludes at-or-below-cutoff flickers, which are mostly noise
68
+ // from brief URL-bar focus / window edge clicks
69
+ // Hard-trigger, AI extensions, layout shifts, zoom changes, keyboard shortcuts,
70
+ // edge exits, synthetic insertions, and foreign inputs no longer affect the
71
+ // *score*. They are NOT hidden from review: every event list still renders in
72
+ // the per-participant detail panes, and hard-triggered participants are still
73
+ // surfaced via the hard/soft/clean tier (which is independent of this score).
74
+ // edgeExitCount and hardTriggered remain in the signature for the renderer's
75
+ // call site but are unused under this policy.
46
76
  export function decomposeScore(summary, edgeExitCount, hardTriggered) {
47
- const aiExtensions = summary.aiExtensionsFound || summary.extensionsDetected || [];
48
77
  const sidebarCount = summary.sidebarEventCount ?? (summary.sidebarDetected ? 1 : 0);
49
- const base = summary.authoritativeSoftScore ?? summary.totalSoftScore;
78
+ const tabAwayMeaningful =
79
+ (summary.tabAwayLongCount || 0) + (summary.tabAwayMediumCount || 0);
50
80
  return [
51
- ['soft', base],
52
- ['hard', hardTriggered ? 100 : 0],
53
- ['edge', edgeExitCount * 3],
54
- ['ext', aiExtensions.length * 5],
55
- ['sidebar', Math.min(sidebarCount * 3, 15)],
56
- ['kb', (summary.keyboardShortcutCount || 0) * 2],
57
- ['layout', (summary.layoutShiftCount || 0)],
58
- ['zoom', (summary.zoomChangeCount || 0)],
59
- ['synth', summary.totalSyntheticInsertions * 4],
60
- ['foreign', summary.totalForeignInputEvents * 3],
81
+ ['paste', (summary.totalPasteEvents || 0) * 5],
82
+ ['copy', (summary.totalCopyEvents || 0) * 5],
83
+ ['sidebar', sidebarCount * 3],
84
+ ['tabaway', tabAwayMeaningful],
61
85
  ];
62
86
  }
63
87
 
@@ -89,9 +113,14 @@ export function generateTriageReason(summary, edgeExitCount = 0) {
89
113
  const longN = summary.tabAwayLongCount || 0;
90
114
  const midN = summary.tabAwayMediumCount || 0;
91
115
  const flickN = summary.tabAwayFlickerCount || 0;
116
+ // Bin boundaries follow THIS participant's tab-away cutoff (3s by default, 5s
117
+ // for strict) so the reason matches the counts/score, which are now computed
118
+ // against that same per-participant cutoff. Flicker is at-or-below the cutoff
119
+ // (not scored); medium is above the cutoff and under 10s.
120
+ const cutoffS = Math.round((summary.tabAwayCutoffMs ?? 3000) / 1000);
92
121
  if (longN > 0) parts.push(`${longN} tab-away${longN === 1 ? '' : 's'} ≥10s`);
93
- if (midN > 0) parts.push(`${midN} tab-away${midN === 1 ? '' : 's'} 3–10s`);
94
- if (flickN > 0) parts.push(`${flickN} flicker${flickN === 1 ? '' : 's'} <3s`);
122
+ if (midN > 0) parts.push(`${midN} tab-away${midN === 1 ? '' : 's'} ${cutoffS}–10s`);
123
+ if (flickN > 0) parts.push(`${flickN} flicker${flickN === 1 ? '' : 's'} ≤${cutoffS}s`);
95
124
  } else if (summary.totalTabAways > 0) {
96
125
  parts.push(`${summary.totalTabAways} tab-aways`);
97
126
  }
@@ -111,6 +140,9 @@ export function generateTriageReason(summary, edgeExitCount = 0) {
111
140
  if (edgeExitCount > 0) parts.push(`${edgeExitCount} edge-exit patterns`);
112
141
  if (summary.totalSyntheticInsertions > 0) parts.push(`${summary.totalSyntheticInsertions} synthetic insertions`);
113
142
  if (summary.totalForeignInputEvents > 0) parts.push(`${summary.totalForeignInputEvents} foreign inputs`);
143
+ // Guard-honeypot self-disclosure — a strong corroborating signal, surfaced in
144
+ // the reason though it does not feed the four-term score.
145
+ if (summary.honeypotAiUse) parts.push('self-reported AI use (honeypot)');
114
146
  if (parts.length === 0) parts.push('clean');
115
147
  return parts.join('; ');
116
148
  }
package/src/cli/config.js CHANGED
@@ -35,6 +35,7 @@ export function loadConfig(cliArgs) {
35
35
  if (flags.participantIdField) config.participantIdField = flags.participantIdField;
36
36
  if (flags.filePattern) config.filePattern = flags.filePattern;
37
37
  if (flags.integrityField) config.integrityField = flags.integrityField;
38
+ if (flags.sessionIntegrityPath) config.sessionIntegrityPath = flags.sessionIntegrityPath;
38
39
  if (flags['no-visuals']) config.noVisuals = true;
39
40
 
40
41
  // Resolve relative paths to absolute (relative to cwd)
@@ -45,20 +46,42 @@ export function loadConfig(cliArgs) {
45
46
  const warnings = validateConfig(fileConfig);
46
47
  warnings.forEach(w => console.warn(`[cyborg-hunter] ${w}`));
47
48
 
49
+ // Validate config VALUES that would otherwise silently mis-score (a
50
+ // non-numeric softScoreThreshold coerces every `score >= threshold` to
51
+ // false, disabling soft flagging with no error).
52
+ cliConfigWarnings(config).forEach(w => console.warn(`[cyborg-hunter] ${w}`));
53
+
48
54
  return config;
49
55
  }
50
56
 
57
+ // Value-level CLI config checks that catch misconfigurations which would
58
+ // otherwise silently zero out a signal or verdict. Returns warning strings.
59
+ // Exported for testing.
60
+ export function cliConfigWarnings(config) {
61
+ const warnings = [];
62
+ const thr = config?.scoring?.softScoreThreshold;
63
+ if (thr != null && typeof thr !== 'number') {
64
+ warnings.push(
65
+ `scoring.softScoreThreshold is not a number (got ${JSON.stringify(thr)}) — ` +
66
+ `every soft-score comparison coerces to false, so NO participant will be ` +
67
+ `soft-flagged. Set it to a number.`
68
+ );
69
+ }
70
+ return warnings;
71
+ }
72
+
51
73
  // Parses CLI arguments into a flags object.
52
74
  //
53
75
  // Supported flags (kebab-case aliases match the camelCase config keys, so
54
76
  // you can use whichever form you remember):
55
- // --config <path> # alternate config file path
77
+ // --config, --config-file <path> # alternate config file path
56
78
  // --data, --data-dir <path> # config.dataDir
57
79
  // --output, --output-dir <path> # config.outputDir
58
80
  // --participant <id> # filter to single participant
59
81
  // --participant-id-field <name> # config.participantIdField
60
82
  // --file-pattern <glob> # config.filePattern
61
83
  // --integrity-field <name> # config.integrityField
84
+ // --session-integrity-path <p> # config.sessionIntegrityPath (dotted)
62
85
  // --no-visuals # skip canvas-rendered images
63
86
  //
64
87
  // Throws on unknown flags. Earlier behavior was to print a warning and keep
@@ -70,6 +93,7 @@ export function parseFlags(args) {
70
93
  // form so loadConfig can look it up directly.
71
94
  const SINGLE_VALUE = {
72
95
  '--config': 'config',
96
+ '--config-file': 'config',
73
97
  '--data': 'data',
74
98
  '--data-dir': 'data',
75
99
  '--output': 'output',
@@ -78,6 +102,7 @@ export function parseFlags(args) {
78
102
  '--participant-id-field': 'participantIdField',
79
103
  '--file-pattern': 'filePattern',
80
104
  '--integrity-field': 'integrityField',
105
+ '--session-integrity-path': 'sessionIntegrityPath',
81
106
  };
82
107
  for (let i = 0; i < args.length; i++) {
83
108
  const a = args[i];