skalpel 4.0.50 → 4.0.52

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/analytics.mjs CHANGED
@@ -264,7 +264,7 @@ export function renderStatic(data) {
264
264
  L.push("");
265
265
  L.push(' nothing caught yet — keep coding, I\'ll log every "done" your agent claims.');
266
266
  L.push("");
267
- L.push(" Enable the shadow with SKALPEL_VERIFY_SHADOW=1; every completion claim gets its own");
267
+ L.push(" Arm the live catch with `skalpel verify on`; every completion claim gets its own");
268
268
  L.push(" proof re-run and appended to the local ledger. Nothing here is ever estimated.");
269
269
  L.push("");
270
270
  return L.join("\n");
@@ -377,7 +377,7 @@ function listLines(data, cols, rows, state) {
377
377
  );
378
378
  L.push("");
379
379
  L.push(
380
- ` ${DIM}Enable the shadow with ${RESET}${BOLD}SKALPEL_VERIFY_SHADOW=1${RESET}${DIM} — every completion claim gets its${RESET}`,
380
+ ` ${DIM}Arm the live catch with ${RESET}${BOLD}skalpel verify on${RESET}${DIM} — every completion claim gets its${RESET}`,
381
381
  );
382
382
  L.push(
383
383
  ` ${DIM}own proof re-run and appended locally. Nothing here is ever estimated.${RESET}`,
package/card.mjs CHANGED
@@ -296,9 +296,7 @@ function renderNoCatch({ id }) {
296
296
  ` ${D}and skalpel re-ran its OWN proof and it genuinely FAILED. Nothing is invented.${X}`,
297
297
  );
298
298
  L.push("");
299
- L.push(
300
- ` ${D}Turn the live shadow on with ${X}${B}SKALPEL_VERIFY_SHADOW=1${X}${D}, keep coding, and${X}`,
301
- );
299
+ L.push(` ${D}Arm the live catch with ${X}${B}skalpel verify on${X}${D}, keep coding, and${X}`);
302
300
  L.push(` ${D}run ${X}${B}skalpel card${X}${D} again the next time a claim doesn't hold up.${X}`);
303
301
  L.push("");
304
302
  return L.join("\n");
package/efficacy-eval.mjs CHANGED
@@ -390,6 +390,276 @@ export function classifyReask(text) {
390
390
  return "none";
391
391
  }
392
392
 
393
+ // ---------- DRIFT-LOOP detector (a repeated SAME-target correction chain — the loop a dev FEELS) ----------
394
+ // PRECISION IS THE POINT: this must NEVER falsely accuse. Three stacked false-positive guards below —
395
+ // (1) the strong-tier + SOFT-subtraction + anaphora ANCHOR gate, (2) a concrete-noun target the USER
396
+ // themselves named AND that appears in the claim, (3) a per-session verdict gate that SUPPRESSES a run
397
+ // where every member was a genuine true-done (that is scope-growth / flakiness, not drift). Pure + offline.
398
+ //
399
+ // ANCHOR_RE — an ANAPHORA / PRIOR-CLAIM reference. A correction that points BACK at the last "done"
400
+ // ("still", "again", "you didn't…", "that's wrong", "same error", "doesn't work", "nothing changed",
401
+ // "i don't see…"). A bare domain-vocab turn ("handle the login-fail case in login.ts") has NONE of these
402
+ // and is therefore NOT drift-eligible — it is a fresh ask, not a push-back on a prior claim.
403
+ const ANCHOR_RE = new RegExp(
404
+ [
405
+ "\\bstill\\b",
406
+ "\\bagain\\b",
407
+ "\\b(?:re-?do|redo|undo|revert|roll\\s?back|rollback|re-?open|reopen)\\b",
408
+ "\\byou\\s+(?:did\\s?n'?t|didnt|did\\s+not|broke|missed|removed|deleted|forgot)\\b",
409
+ "\\b(?:that|this|it)(?:'?s|\\s+is|\\s+isn'?t|\\s+still|\\s+broke|\\s+breaks)\\b",
410
+ "\\bsame\\s+(?:error|issue|problem|thing|result|bug|failure)\\b",
411
+ "\\bnot\\s+(?:fixed|working|done|right|correct|quite|what)\\b",
412
+ "\\bdoes\\s?n'?t\\s+work\\b",
413
+ "\\bdo\\s?n'?t\\s+work\\b",
414
+ "\\bdid\\s?n'?t\\s+work\\b",
415
+ "\\bnot\\s+work(?:ing)?\\b",
416
+ "\\bthat'?s\\s+(?:wrong|not\\s+right|incorrect|not\\s+it)\\b",
417
+ "\\bthis\\s+is\\s+(?:wrong|broken|not\\s+right)\\b",
418
+ "\\bit'?s\\s+(?:still\\s+)?(?:broken|failing|wrong|not\\s+working)\\b",
419
+ "\\bnot\\s+what\\s+i\\s+(?:meant|wanted|asked|said|had\\s+in\\s+mind)\\b",
420
+ "\\bi\\s+hate\\b",
421
+ "\\bvibe[\\s-]?coded\\b",
422
+ "\\b(?:ai\\s+slop|slop)\\b",
423
+ "\\bnothing\\s+chang(?:ed|es)\\b",
424
+ "\\bdid\\s?n'?t\\s+chang(?:e|ed)\\b",
425
+ "\\bno\\s+chang(?:e|ed)\\b",
426
+ "\\bi\\s+(?:do\\s?n'?t|don'?t)\\s+see\\b",
427
+ ].join("|"),
428
+ "i",
429
+ );
430
+
431
+ // isDriftEligibleStrong(text) — the single-sourced precision gate. TRUE iff ALL hold:
432
+ // (a) classifyReask(text) === 'strong' (necessary condition — reuse the offline detector)
433
+ // (b) SOFT_CONTINUATION_RE.test(text) === false (SOFT-SUBTRACTION: classifyReask tests STRONG
434
+ // before SOFT, so a scope-add that also carries a strong keyword — "also, the old test was wrong,
435
+ // add another" — leaks in as 'strong'. If SOFT also matches, it is a scope-add → EXCLUDE.)
436
+ // (c) an ANAPHORA / PRIOR-CLAIM anchor is present (ANCHOR_RE).
437
+ export function isDriftEligibleStrong(text) {
438
+ const s = String(text || "");
439
+ if (!s.trim()) return false;
440
+ if (classifyReask(s) !== "strong") return false; // (a)
441
+ if (SOFT_CONTINUATION_RE.test(s)) return false; // (b) SOFT-subtraction
442
+ if (!ANCHOR_RE.test(s)) return false; // (c) anaphora / prior-claim anchor
443
+ return true;
444
+ }
445
+
446
+ // Generic project-ubiquitous tokens that are NEVER a distinguishing target (case-insensitive drop).
447
+ const DRIFT_GENERIC_DROP = new Set([
448
+ "npm",
449
+ "run",
450
+ "src",
451
+ "index",
452
+ "main",
453
+ "app",
454
+ "test",
455
+ "tests",
456
+ "build",
457
+ "node",
458
+ "pnpm",
459
+ "yarn",
460
+ "dist",
461
+ "lib",
462
+ "the",
463
+ "fix",
464
+ "add",
465
+ "make",
466
+ "get",
467
+ "set",
468
+ "use",
469
+ "now",
470
+ "also",
471
+ ]);
472
+ const DRIFT_CODE_EXT_RE =
473
+ /\.(?:ts|tsx|js|jsx|mjs|cjs|py|go|rs|css|scss|json|md|html|sql|sh|yml|yaml|toml)$/i;
474
+ const DRIFT_IDENT_CHARSET_RE = /^[A-Za-z0-9_.-]+$/;
475
+
476
+ // concreteNouns(text) → unique concrete targets, ORIGINAL CASE preserved (compared case-insensitively).
477
+ // A surviving token MUST be one of: a file path (contains "/" OR a code extension), a URL, a "#<digits>"
478
+ // PR/issue ref, a backtick-/quote-delimited identifier, or a camelCase/snake_case/kebab identifier of
479
+ // length>=3 — never a bare generic lowercase English word. Generic project tokens are then dropped.
480
+ function concreteNouns(text) {
481
+ const s = String(text || "");
482
+ const out = [];
483
+ const seen = new Set(); // lowercase dedup keys
484
+ const consider = (raw, delimited = false) => {
485
+ const tok = String(raw || "").trim();
486
+ if (!tok) return;
487
+ const key = tok.toLowerCase();
488
+ if (DRIFT_GENERIC_DROP.has(key)) return;
489
+ if (seen.has(key)) return;
490
+ const isUrl = /^https?:\/\//i.test(tok);
491
+ const isRef = /^#\d+$/.test(tok);
492
+ const isPath = tok.includes("/") || DRIFT_CODE_EXT_RE.test(tok);
493
+ const isIdent =
494
+ tok.length >= 3 &&
495
+ DRIFT_IDENT_CHARSET_RE.test(tok) &&
496
+ (/[a-z][A-Z]/.test(tok) || tok.includes("_") || tok.includes("-"));
497
+ // A delimited (backtick/quote) single token is an explicit user reference — it qualifies even as a
498
+ // plain lowercase word, but must still be an identifier-ish charset with a letter (never pure prose).
499
+ const isDelimited = delimited && /^[A-Za-z0-9_./#-]{2,}$/.test(tok) && /[A-Za-z]/.test(tok);
500
+ if (!(isUrl || isRef || isPath || isIdent || isDelimited)) return;
501
+ seen.add(key);
502
+ out.push(tok);
503
+ };
504
+ // (1) backtick / single- / double-quote delimited SINGLE tokens (multi-word spans are ignored — an
505
+ // identifier has no internal whitespace, and this avoids apostrophe-span false tokens).
506
+ const delimRe = /`([^`]+)`|'([^']+)'|"([^"]+)"/g;
507
+ let m;
508
+ while ((m = delimRe.exec(s)) !== null) {
509
+ const inner = (m[1] ?? m[2] ?? m[3] ?? "").trim();
510
+ if (!inner || /\s/.test(inner)) continue;
511
+ consider(inner, true);
512
+ }
513
+ // (2) general whitespace scan — strip surrounding punctuation (keep a leading #/ and inner .-_).
514
+ const stripEdges = (w) =>
515
+ String(w || "")
516
+ .replace(/^[^A-Za-z0-9#/]+/, "")
517
+ .replace(/[^A-Za-z0-9/]+$/, "");
518
+ for (const w0 of s.split(/\s+/)) {
519
+ const w = stripEdges(w0);
520
+ if (w) consider(w, false);
521
+ }
522
+ return out;
523
+ }
524
+
525
+ // targetTokens(record) — the concrete things the USER named that ALSO appear in the claim (case-insensitive
526
+ // intersection). Display value is taken VERBATIM from the user's turn (next_user_text). If the user named no
527
+ // surviving concrete target, the record cannot anchor a loop → []. This guarantees every emitted target is a
528
+ // concrete noun the user themselves typed, never one invented from the agent's own claim text.
529
+ function targetTokens(record) {
530
+ const r = record || {};
531
+ const userNouns = concreteNouns(r.next_user_text);
532
+ if (!userNouns.length) return [];
533
+ const claimNouns = concreteNouns((r.claim_text || "") + " " + (r.proof_command || ""));
534
+ const claimKeys = new Set(claimNouns.map((t) => t.toLowerCase()));
535
+ const out = [];
536
+ for (const t of userNouns) {
537
+ if (claimKeys.has(t.toLowerCase())) out.push(t); // userNouns is already unique + user-case
538
+ }
539
+ return out;
540
+ }
541
+
542
+ // detectDriftLoop(records, opts) — per-session, pure, no network. Walks the drift-eligible corrections in
543
+ // each session and links CONSECUTIVE ones that keep re-naming an overlapping concrete target within a time
544
+ // span, then SUPPRESSES any run whose every member was a genuine true-done. Returns loop objects.
545
+ export function detectDriftLoop(records, opts = {}) {
546
+ const minRepeat = Number.isFinite(opts.minRepeat) ? opts.minRepeat : 2;
547
+ const maxSpanMs = Number.isFinite(opts.maxSpanMs) ? opts.maxSpanMs : 30 * 60 * 1000;
548
+ const IDLE_MS = 10 * 60 * 1000;
549
+ if (!Array.isArray(records) || !records.length) return [];
550
+
551
+ // group by .session, stable input order (= session-transcript order)
552
+ const order = [];
553
+ const groups = new Map();
554
+ for (const r of records) {
555
+ const key = r && r.session != null ? String(r.session) : "?";
556
+ if (!groups.has(key)) {
557
+ groups.set(key, []);
558
+ order.push(key);
559
+ }
560
+ groups.get(key).push(r);
561
+ }
562
+
563
+ const toMap = (tokens) => {
564
+ const mp = new Map();
565
+ for (const t of tokens) {
566
+ const k = t.toLowerCase();
567
+ if (!mp.has(k)) mp.set(k, t); // keep first (earliest user-turn) display case
568
+ }
569
+ return mp;
570
+ };
571
+ const intersect = (runMap, tokens) => {
572
+ const keys = new Set(tokens.map((t) => t.toLowerCase()));
573
+ const mp = new Map();
574
+ for (const [k, disp] of runMap) if (keys.has(k)) mp.set(k, disp);
575
+ return mp;
576
+ };
577
+
578
+ const loops = [];
579
+ for (const key of order) {
580
+ // (i) eligible list, in order: evaluated + drift-eligible-strong + has a user-named concrete target
581
+ const eligible = [];
582
+ for (const r of groups.get(key)) {
583
+ if (r && r.reask_status === "evaluated" && isDriftEligibleStrong(r.next_user_text)) {
584
+ const tokens = targetTokens(r);
585
+ if (tokens.length > 0) eligible.push({ r, tokens });
586
+ }
587
+ }
588
+ if (!eligible.length) continue;
589
+
590
+ // (ii) build runs — extend IFF target overlap AND (span + per-step gap) within window (time checks
591
+ // skipped only when a ts is missing; a run with any missing ts yields elapsed_ms=null downstream).
592
+ const runs = [];
593
+ let run = null;
594
+ for (const e of eligible) {
595
+ if (!run) {
596
+ run = { members: [e], targets: toMap(e.tokens) };
597
+ continue;
598
+ }
599
+ const inter = intersect(run.targets, e.tokens);
600
+ let extend = inter.size > 0;
601
+ if (extend) {
602
+ const tFirst = Date.parse(run.members[0].r.claim_ts);
603
+ const tPrev = Date.parse(run.members[run.members.length - 1].r.claim_ts);
604
+ const tNext = Date.parse(e.r.claim_ts);
605
+ if (Number.isFinite(tFirst) && Number.isFinite(tPrev) && Number.isFinite(tNext)) {
606
+ if (!(tNext - tFirst <= maxSpanMs && tNext - tPrev <= maxSpanMs)) extend = false;
607
+ }
608
+ }
609
+ if (extend) {
610
+ run.members.push(e);
611
+ run.targets = inter;
612
+ } else {
613
+ runs.push(run);
614
+ run = { members: [e], targets: toMap(e.tokens) };
615
+ }
616
+ }
617
+ if (run) runs.push(run);
618
+
619
+ // (iii) verdict gate + emit
620
+ for (const rn of runs) {
621
+ const repeat = rn.members.length;
622
+ if (repeat < minRepeat) continue;
623
+ const verdicts = rn.members.map((mm) => mm.r.verdict);
624
+ if (verdicts.every((v) => v === "true-done")) continue; // SUPPRESS: scope-growth/flakiness, not drift
625
+
626
+ const first_ts = rn.members[0].r.claim_ts ?? null;
627
+ const last_ts = rn.members[repeat - 1].r.claim_ts ?? null;
628
+ const fp = Date.parse(first_ts);
629
+ const lp = Date.parse(last_ts);
630
+ const elapsed_ms = Number.isFinite(fp) && Number.isFinite(lp) ? lp - fp : null;
631
+
632
+ // largest single inter-member claim_ts gap (only over pairs with both ts present)
633
+ let maxGap = -1;
634
+ for (let j = 1; j < repeat; j++) {
635
+ const a = Date.parse(rn.members[j - 1].r.claim_ts);
636
+ const b = Date.parse(rn.members[j].r.claim_ts);
637
+ if (Number.isFinite(a) && Number.isFinite(b)) maxGap = Math.max(maxGap, b - a);
638
+ }
639
+ const idle_suspect = maxGap > maxSpanMs || maxGap > IDLE_MS;
640
+
641
+ // stable target: longest, then lexicographically-first (verbatim user-turn case)
642
+ const targets = [...rn.targets.values()].sort(
643
+ (a, b) => b.length - a.length || (a < b ? -1 : a > b ? 1 : 0),
644
+ );
645
+
646
+ loops.push({
647
+ session: key === "?" ? null : key,
648
+ target: targets[0],
649
+ repeat,
650
+ first_ts,
651
+ last_ts,
652
+ elapsed_ms,
653
+ idle_suspect,
654
+ members_verdicts: verdicts,
655
+ next_user_texts: rn.members.map((mm) => String(mm.r.next_user_text || "").slice(0, 300)),
656
+ turn_targets: rn.members.map((mm) => mm.tokens.slice()),
657
+ });
658
+ }
659
+ }
660
+ return loops;
661
+ }
662
+
393
663
  // ---------- (3) scan ONE session → per-claim records + the session's PASS proof-runs ----------
394
664
  // A record's verdict, deduped by proof tool_use id (identical to scanSession), is:
395
665
  // 'confirmed-lie' — claim over a proof whose recorded result classifies GENUINE_FAIL (a real lie)
@@ -751,6 +1021,8 @@ function recId(r) {
751
1021
  // reported). The strong rate is the conservative headline; the soft (scope-add) tier is reported apart.
752
1022
  export function computeReaskReport(corpus) {
753
1023
  const recs = corpus.records || [];
1024
+ // DRIFT LOOPS — repeated same-target STRONG corrections in a session (pure, from the same records).
1025
+ const driftLoops = detectDriftLoop(recs);
754
1026
  const totalDone = recs.length;
755
1027
  const evaluated = recs.filter((r) => r.reask_status === "evaluated");
756
1028
  const superseded = recs.filter((r) => r.reask_status === "superseded").length;
@@ -837,6 +1109,9 @@ export function computeReaskReport(corpus) {
837
1109
  passing_proof_reask_rate_strong: passingProof.reask_rate_strong,
838
1110
  passing_proof_reask_rate_any: passingProof.reask_rate_any,
839
1111
  by_verdict: byVerdict,
1112
+ // DRIFT LOOPS — repeated same-target correction chains (deterministic; identical to detectDriftLoop(recs)).
1113
+ drift_loops: driftLoops,
1114
+ drift_loop_count: driftLoops.length,
840
1115
  samples,
841
1116
  };
842
1117
  }
@@ -1216,6 +1491,35 @@ export function renderReaskReport(reask, frequency = {}) {
1216
1491
  });
1217
1492
  }
1218
1493
 
1494
+ // ---- DRIFT LOOPS (repeated same-target corrections — the loop you FEEL) ----
1495
+ L.push("");
1496
+ L.push(" 5) DRIFT LOOPS — repeated STRONG corrections about the SAME concrete target you named");
1497
+ L.push(
1498
+ " (a same-file/-symbol re-ask chain; every noun below is re-derived from YOUR own turns)",
1499
+ );
1500
+ L.push(" " + "·".repeat(74));
1501
+ const loops = Array.isArray(R.drift_loops) ? R.drift_loops : [];
1502
+ if ((R.drift_loop_count || 0) === 0) {
1503
+ L.push(" no repeated same-target correction loops found.");
1504
+ } else {
1505
+ L.push(` ${R.drift_loop_count} drift loop(s) found (showing up to 5):`);
1506
+ loops.slice(0, 5).forEach((lp, i) => {
1507
+ // Neutral wall-clock duration ONLY when we have two real timestamps, the run is not idle-suspect,
1508
+ // and the span is at least a minute. NEVER "wasted/burned" — it is a plain elapsed delta.
1509
+ let dstr = "";
1510
+ if (lp.elapsed_ms != null && !lp.idle_suspect && lp.elapsed_ms >= 60000) {
1511
+ const d = formatDuration(lp.elapsed_ms);
1512
+ if (d) dstr = ` · ${d} wall-clock between first and last correction`;
1513
+ }
1514
+ L.push(
1515
+ ` ${i + 1}. \`${clip(lp.target, 48)}\` — corrected ${lp.repeat}× in this session${dstr}`,
1516
+ );
1517
+ (lp.next_user_texts || []).forEach((t) => {
1518
+ L.push(` you: "${clip(t, 66)}"`);
1519
+ });
1520
+ });
1521
+ }
1522
+
1219
1523
  L.push(" " + "─".repeat(76));
1220
1524
  L.push(" Read-only replay. --reask --json for the machine-readable re-ask numbers.");
1221
1525
  L.push("");
package/ledger.mjs CHANGED
@@ -246,7 +246,7 @@ export function renderLedger(ledger, { logPath = VERIFY_LOG_PATH } = {}) {
246
246
  L.push("");
247
247
  L.push(" No claim-verification history yet on this machine.");
248
248
  L.push(
249
- " Enable the shadow with SKALPEL_VERIFY_SHADOW=1, then keep coding — every turn your agent",
249
+ " Arm the live catch with `skalpel verify on`, then keep coding — every turn your agent",
250
250
  );
251
251
  L.push(' claims "done / tests pass" gets its own proof re-run and appended to the ledger.');
252
252
  L.push("");
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "skalpel",
3
- "version": "4.0.50",
3
+ "version": "4.0.52",
4
4
  "description": "Behavioral graph for AI-assisted coding — learns how you work and steers Claude Code + Codex in real time.",
5
5
  "type": "module",
6
6
  "bin": {
package/skalpel-setup.mjs CHANGED
@@ -77,10 +77,10 @@ const KNOWN_SUBS = new Set([
77
77
  const VALUE_FLAGS = new Set(["--api", "--user"]);
78
78
  // Boolean flags that take no value. `--purge` on `uninstall` also removes auth + all local data.
79
79
  const BOOL_FLAGS = new Set(["--purge"]);
80
- const USAGE = `skalpel — connect your coding history to your behavioral graph
80
+ const USAGE = `skalpel — your AI says "done." skalpel re-runs its own tests and shows you the truth.
81
81
 
82
82
  usage:
83
- skalpel [setup] [--api <url>] [--user <id>] sign in, learn your history, wire the hooks
83
+ skalpel [setup] [--api <url>] [--user <id>] set up the catch: scan your history, wire the live check
84
84
  skalpel login (re-)run the Google sign-in
85
85
  skalpel logout clear the saved session
86
86
  skalpel autopsy local, read-only receipt of your verified patterns
@@ -273,7 +273,9 @@ ${R} ███████╗██╗ ██╗ █████╗ ██╗
273
273
  ╚════██║██╔═██╗ ██╔══██║██║ ██╔═══╝ ██╔══╝ ██║
274
274
  ███████║██║ ██╗██║ ██║███████╗██║ ███████╗███████╗
275
275
  ╚══════╝╚═╝ ╚═╝╚═╝ ╚═╝╚══════╝╚═╝ ╚══════╝╚══════╝${X}`);
276
- console.log(`\n ${D}your AI coding sessions, learned — and steered in real time${X}\n`);
276
+ console.log(
277
+ `\n ${D}when your AI says "done," skalpel re-runs its own tests — and tells you the truth${X}\n`,
278
+ );
277
279
  }
278
280
 
279
281
  const FRAMES = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"];