@ngockhoale/ukit 3.3.3 → 3.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/CHANGELOG.md +44 -0
  2. package/manifests/engineConformance.yaml +17 -1
  3. package/manifests/hostCapabilities.yaml +68 -1
  4. package/manifests/platform.full.yaml +138 -0
  5. package/manifests/platform.user.yaml +255 -3
  6. package/package.json +1 -1
  7. package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
  8. package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
  9. package/scripts/probe/codex-capability-probe.mjs +169 -0
  10. package/src/cli/commands/doctor.js +168 -0
  11. package/src/cli/commands/indexTools.js +7 -0
  12. package/src/cli/commands/metrics.js +66 -2
  13. package/src/cli/commands/playbook.js +4 -4
  14. package/src/cli/commands/vm.js +49 -8
  15. package/src/core/agentRuntime/adapters.js +328 -27
  16. package/src/core/agentRuntime/artifacts.js +89 -0
  17. package/src/core/agentRuntime/context.js +345 -1
  18. package/src/core/agentRuntime/contract.js +296 -0
  19. package/src/core/agentRuntime/eventStore.js +176 -0
  20. package/src/core/agentRuntime/shadowRun.js +481 -5
  21. package/src/core/agentRuntime/telemetry.js +121 -0
  22. package/src/core/observability/emit/lifecycle.js +68 -1
  23. package/src/core/observability/emit/sessionBoot.js +393 -0
  24. package/src/core/observability/privacy/allowlist.js +10 -1
  25. package/src/core/observability/schema/registry.js +10 -0
  26. package/src/core/runtimeConfig.js +133 -0
  27. package/src/core/userPlaybooks.js +18 -3
  28. package/src/decision/registry.js +19 -0
  29. package/src/diagnostics/feedbackEvents.js +7 -4
  30. package/src/diagnostics/routeOutcomes.js +51 -6
  31. package/src/diagnostics/skillAccuracy.js +43 -3
  32. package/src/index/crossCheckMatrix.js +412 -0
  33. package/src/index/fixLoopEscalation.js +453 -0
  34. package/src/index/playbookRegistry.js +691 -0
  35. package/src/index/reviewPolicy.js +368 -0
  36. package/src/index/routeResolver.js +915 -0
  37. package/src/index/sessionHistoryExtractor.js +359 -0
  38. package/src/index/taskRouting.js +764 -581
  39. package/src/index/tierSelection.js +308 -0
  40. package/src/index/verificationMap.js +404 -0
  41. package/template_project/.claude/hooks/observability-emit.mjs +14 -0
  42. package/template_project/.claude/hooks/record-execution.mjs +19 -1
  43. package/template_project/.claude/hooks/skill-router.sh +691 -25
  44. package/template_project/.claude/hooks/verification-guard.sh +230 -1
  45. package/template_project/.claude/settings.json +2 -2
  46. package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
  47. package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
  48. package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
  49. package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
  50. package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
  51. package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
  52. package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
  53. package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
  54. package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
  55. package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
  56. package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
  57. package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
  58. package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
  59. package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
  60. package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
  61. package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
  62. package/template_project/.codex/README.md +8 -0
  63. package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
  64. package/template_project/ukit/README.md +1 -1
  65. package/template_project/ukit/storage/config.json +20 -0
  66. package/template_user/playbooks/architecture-decision.md +28 -0
  67. package/template_user/playbooks/autonomous-run.md +43 -0
  68. package/template_user/playbooks/autopilot-full.md +59 -0
  69. package/template_user/playbooks/autopilot-stack.md +54 -0
  70. package/template_user/playbooks/babysit.md +39 -0
  71. package/template_user/playbooks/bug-fix.md +3 -1
  72. package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
  73. package/template_user/playbooks/hillclimb.md +44 -0
  74. package/template_user/playbooks/investigation.md +21 -0
  75. package/template_user/playbooks/migration.md +21 -0
  76. package/template_user/playbooks/open-pr.md +48 -0
  77. package/template_user/playbooks/orchestrate.md +45 -0
  78. package/template_user/playbooks/performance.md +33 -0
  79. package/template_user/playbooks/prototype.md +28 -0
  80. package/template_user/playbooks/refactor.md +19 -0
  81. package/template_user/playbooks/release.md +28 -0
  82. package/template_user/playbooks/runtime-forensics.md +23 -0
  83. package/template_user/playbooks/session-pickup.md +31 -0
  84. package/template_user/playbooks/shipping.md +53 -0
  85. package/template_user/playbooks/skill-evaluation.md +48 -0
  86. package/template_user/playbooks/small-feature.md +20 -0
  87. package/template_user/playbooks/verification-map.json +153 -0
  88. package/template_user/playbooks/verification.md +22 -0
  89. package/template_user/playbooks/worktree-cleanup.md +37 -0
@@ -17,6 +17,10 @@ const LEDGER_VERSION = 1;
17
17
  const RESUME_INTENT_VERSION = 1;
18
18
  const RESUME_INTENT_TTL_MS = 30 * 60 * 1000;
19
19
  const MAX_RECEIPTS = 24;
20
+ // TASK-005 (FR-005): bounded OutcomeRecord dedupe slot — a resumable-run
21
+ // pointer (interrupted run) emits `outcome.observed` once per task; replays
22
+ // and resume folds are seen via this list, never re-emitted.
23
+ const MAX_LEDGER_OUTCOMES = 16;
20
24
  const MAX_SOURCE_FILES = 16;
21
25
  const MAX_CONTINUATIONS = 6;
22
26
  // A malformed stdin payload crashes `--evaluate-stop` before a route/session can be read, so
@@ -91,6 +95,16 @@ function sessionIdentity(payload = {}) {
91
95
  return 'default';
92
96
  }
93
97
 
98
+ // TASK-C85-016 (BL-016): bounded identity string for the per-pair receipt
99
+ // fields (implementer/reviewer/reviewVerdict). Model/role ids are short
100
+ // strings; anything else collapses to null so OutcomeRecord fields stay
101
+ // nullable-clean.
102
+ function pairIdentity(value) {
103
+ return typeof value === 'string' && value.trim()
104
+ ? value.trim().slice(0, 96)
105
+ : null;
106
+ }
107
+
94
108
  function ledgerPath(projectRoot, payload = {}) {
95
109
  return path.join(
96
110
  projectRoot,
@@ -312,7 +326,14 @@ function hasUnfinishedCompletion(state = {}, ledger = {}) {
312
326
  if (!IMPLEMENT_MODES.has(mode)) return false;
313
327
  const required = requiredEvidence(state);
314
328
  if (required.length === 0) return false;
315
- return required.some((item) => !evidenceSatisfied(item, ledger, state));
329
+ const unfinished = required.some((item) => !evidenceSatisfied(item, ledger, state));
330
+ if (unfinished) return true;
331
+ // TASK-009 (BL-011): a bug-fix route is also unfinished while its floor is
332
+ // missing — the generic evidence alone never releases a bug fix.
333
+ if (state?.routeSummary?.playbookId === BUGFIX_PLAYBOOK_ID) {
334
+ return bugFixFloorMissing(bugFixFloorStatus(ledger, BUGFIX_PLAYBOOK_ID)).length > 0;
335
+ }
336
+ return false;
316
337
  }
317
338
 
318
339
  function resumeSessionHash(sessionId) {
@@ -480,7 +501,7 @@ function compactReceipt(receipt) {
480
501
  kind: receipt.kind,
481
502
  success: receipt.success,
482
503
  };
483
- for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error', 'verdict']) {
504
+ for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error', 'verdict', 'evidence', 'evidenceSurface', 'class', 'status', 'detail']) {
484
505
  if (receipt[key] !== undefined && receipt[key] !== null && receipt[key] !== '') {
485
506
  compact[key] = receipt[key];
486
507
  }
@@ -894,10 +915,20 @@ function carriedEvidenceLedger(fresh, current) {
894
915
  // Verification-loop tracking and any minted blocker belong to the same logical request
895
916
  // too — dropping a blocker on re-key would resume the exact loop it recorded.
896
917
  failedVerificationStreak: current.failedVerificationStreak || null,
918
+ playbookId: fresh.playbookId || current.playbookId || null,
919
+ // The bug-fix floor bank is monotone request-scoped evidence — a same-prompt
920
+ // re-key must keep it or the gate would re-demand repro receipts mid-request.
921
+ bugFixFloor: current.bugFixFloor || fresh.bugFixFloor || null,
922
+ // TASK-008: the playbook-finding bank is request-scoped like the floor —
923
+ // a same-prompt re-key must keep the latest per-class statuses.
924
+ playbookFindings: current.playbookFindings || fresh.playbookFindings || null,
897
925
  verificationFailureCounts: current.verificationFailureCounts || {},
898
926
  blocker: current.blocker || null,
899
927
  // Journal bookkeeping belongs to the file, not the request: the dedupe memory must
900
928
  // survive a re-key or replayed journal events would be applied twice (TASK-027).
929
+ // Outcome-emit dedupe rides the same request carry — a same-prompt re-key
930
+ // must not re-emit an INTERRUPTED record it already wrote.
931
+ outcomes: [...(current.outcomes || []), ...(fresh.outcomes || [])].slice(-MAX_LEDGER_OUTCOMES),
901
932
  journalSeen: current.journalSeen || [],
902
933
  journalQuarantined: current.journalQuarantined || 0,
903
934
  lastJournalQuarantineAt: current.lastJournalQuarantineAt || null,
@@ -927,6 +958,16 @@ function freshLedger(payload, routeState, harness) {
927
958
  requestKey: routeState?.requestKey || null,
928
959
  promptKey: evidencePromptKey(routeState),
929
960
  routeFingerprint: routeState?.fingerprint || null,
961
+ // TASK-009 (BL-011): the routed playbook rides the ledger so the bug-fix
962
+ // completion floor can classify repro receipts at apply time — including
963
+ // journaled replays, where no route state is passed.
964
+ playbookId: routeState?.routeSummary?.playbookId || null,
965
+ // Bug-fix completion floor bank: monotone, request-scoped, never evicted by
966
+ // the receipt window. null until the first floor receipt lands.
967
+ bugFixFloor: null,
968
+ // TASK-008 (BL-010): latest-per-class playbook-finding statuses, banked
969
+ // outside the evictable receipt window.
970
+ playbookFindings: null,
930
971
  sourceSucceeded: false,
931
972
  sourceFiles: [],
932
973
  writeAttempted: false,
@@ -946,6 +987,9 @@ function freshLedger(payload, routeState, harness) {
946
987
  bankedWrites: {},
947
988
  bankedVerifications: {},
948
989
  notified: false,
990
+ // TASK-005 (FR-005): OutcomeRecord dedupe ledger ({verdict, runKey, at}).
991
+ // Emitted-once for INTERRUPTED survives re-keys via carriedEvidenceLedger.
992
+ outcomes: [],
949
993
  updatedAt: Date.now(),
950
994
  };
951
995
  }
@@ -971,7 +1015,7 @@ const MAX_JOURNAL_SEEN = 64;
971
1015
  const MAX_JOURNAL_RECORDS = 128;
972
1016
  const MAX_JOURNAL_QUARANTINE_LINES = 32;
973
1017
  const LEDGER_EVENT_TYPES = new Set(['receipt', 'continuation', 'notified', 'stop-progress', 'resumable-run']);
974
- const RECEIPT_KINDS = new Set(['source', 'write', 'verification']);
1018
+ const RECEIPT_KINDS = new Set(['source', 'write', 'verification', 'playbook-finding']);
975
1019
 
976
1020
  function journalPathFor(target) {
977
1021
  return `${target}.journal`;
@@ -1077,6 +1121,24 @@ function newEventId() {
1077
1121
 
1078
1122
  const JOURNAL_RECEIPT_FIELDS = ['ts', 'kind', 'toolName', 'toolUseId', 'success', 'exitCode', 'file', 'command', 'scope'];
1079
1123
 
1124
+ // TASK-009 (BL-011): evidence attestations are whitelisted verbatim — class /
1125
+ // surface / result / claim, the ARCH §Evidence Schema fields a receipt may
1126
+ // carry — because a journaled bug-fix receipt must replay with its floor
1127
+ // evidence intact. The nested whitelist bounds the journal record the same way
1128
+ // the flat one bounds the receipt.
1129
+ const JOURNAL_EVIDENCE_FIELDS = ['class', 'surface', 'result', 'claim'];
1130
+
1131
+ function sanitizeEvidenceForJournal(evidence) {
1132
+ if (!evidence || typeof evidence !== 'object' || Array.isArray(evidence)) return null;
1133
+ const clean = {};
1134
+ for (const key of JOURNAL_EVIDENCE_FIELDS) {
1135
+ const value = evidence[key];
1136
+ if (value === undefined || value === null || value === '') continue;
1137
+ clean[key] = String(value).slice(0, 400);
1138
+ }
1139
+ return Object.keys(clean).length > 0 ? clean : null;
1140
+ }
1141
+
1080
1142
  function sanitizeReceiptForJournal(receipt) {
1081
1143
  const clean = {};
1082
1144
  for (const key of JOURNAL_RECEIPT_FIELDS) {
@@ -1084,6 +1146,8 @@ function sanitizeReceiptForJournal(receipt) {
1084
1146
  clean[key] = receipt[key];
1085
1147
  }
1086
1148
  }
1149
+ const evidence = sanitizeEvidenceForJournal(receipt && receipt.evidence);
1150
+ if (evidence) clean.evidence = evidence;
1087
1151
  return clean;
1088
1152
  }
1089
1153
 
@@ -1312,6 +1376,203 @@ async function drainJournal(target, { ledger, payload, signal, deadlineMs }) {
1312
1376
  return { ledger: next, applied, quarantined: malformed.length, commit };
1313
1377
  }
1314
1378
 
1379
+ // --- TASK-009 (BL-011): bug-fix completion floor --------------------------------
1380
+ // A green verdict on a `playbookId: bug-fix` route requires verbatim
1381
+ // `repro-before` (failing) + `repro-after` (passing) evidence receipts on the
1382
+ // SAME surface plus a named root cause in the exec-ledger — a symptom patch
1383
+ // without receipts cannot pass the gate. A defect that honest attempts cannot
1384
+ // reproduce ends INCONCLUSIVE with the attempt count and the runtime-forensics
1385
+ // follow-on named; the floor never fabricates a pass.
1386
+ //
1387
+ // Receipts may carry an explicit ARCH §Evidence Schema attestation
1388
+ // (`receipt.evidence = {class, surface, result, claim}`) — minted from
1389
+ // `payload.evidence`, `payload.tool_input.evidence`, or a leading
1390
+ // `UKIT_EVIDENCE=class=…;surface=…` env assignment on a Bash command. On a
1391
+ // bug-fix ledger the apply step additionally classifies every un-attested
1392
+ // verification receipt itself: a failed verification is a `repro-before`
1393
+ // (the old failure reproduced verbatim), and a success on a surface with a
1394
+ // recorded failure is the `repro-after` — so the live hook path produces the
1395
+ // floor without any attestation at all. Derived classes are stamped back onto
1396
+ // the receipt, which keeps them verbatim inside the receipt window.
1397
+ //
1398
+ // The bank lives in `ledger.bugFixFloor` and is monotone per request: receipts
1399
+ // evicted by the MAX_RECEIPTS window can never un-bank evidence. `seen` dedupes
1400
+ // per surface+class so replayed journal records and eval-time receipt folds
1401
+ // are idempotent.
1402
+ const BUGFIX_PLAYBOOK_ID = 'bug-fix';
1403
+ const BUGFIX_REPRO_BEFORE = 'repro-before';
1404
+ const BUGFIX_REPRO_AFTER = 'repro-after';
1405
+ const BUGFIX_FLOOR_SEEN_CAP = 32;
1406
+ // Root-cause classes: the mechanism attestation carries the named cause in
1407
+ // `claim` (ARCH §Done Criteria: instrumentation/behavioral-check for the named
1408
+ // mechanism; `root-cause` attested directly is accepted too).
1409
+ const BUGFIX_ROOT_CAUSE_CLASSES = new Set(['root-cause', 'instrumentation', 'behavioral-check']);
1410
+ const BUGFIX_FLOOR_CLASSES = new Set([
1411
+ BUGFIX_REPRO_BEFORE,
1412
+ BUGFIX_REPRO_AFTER,
1413
+ ...BUGFIX_ROOT_CAUSE_CLASSES,
1414
+ ]);
1415
+ const EVIDENCE_RESULTS = new Set(['pass', 'fail', 'inconclusive']);
1416
+ const EVIDENCE_FIELD_MAX = 400;
1417
+
1418
+ function emptyBugFixFloor() {
1419
+ return {
1420
+ seen: [],
1421
+ reproAttempts: 0,
1422
+ // surface -> true: the original failure reproduced verbatim on it.
1423
+ reproBefore: {},
1424
+ // surface -> true: the same surface went green after the fix.
1425
+ reproAfter: {},
1426
+ rootCause: null,
1427
+ };
1428
+ }
1429
+
1430
+ // A receipt's floor surface is its attested surface verbatim, else the
1431
+ // terminal-shell identity of its command — env-var prefixes (the
1432
+ // UKIT_EVIDENCE attestation channel) never change the surface of a command.
1433
+ function bugFixReceiptSurface(receipt) {
1434
+ const attested = receipt?.evidence?.surface;
1435
+ if (typeof attested === 'string' && attested) return attested;
1436
+ const command = String(receipt?.command || '').trim();
1437
+ if (!command) return null;
1438
+ const stripped = command.replace(/^(?:[A-Za-z_][A-Za-z0-9_]*=\S+\s+)+/, '');
1439
+ return terminalShellCommandUnit(stripped) || stripped;
1440
+ }
1441
+
1442
+ // The class a receipt carries for the floor: an attested floor class wins;
1443
+ // otherwise, on a bug-fix ledger only, the verification outcome itself is the
1444
+ // evidence — a failed run is repro-before, a pass over a failed surface is
1445
+ // repro-after. `failureCounts` must be the per-surface failure map observed
1446
+ // BEFORE this receipt applies (the live path passes
1447
+ // `ledger.verificationFailureCounts`; the eval fold passes a counter it builds
1448
+ // while walking the receipt window in order).
1449
+ function bugFixReceiptClass(receipt, floor, failureCounts = {}, playbookId = null) {
1450
+ const attested = receipt?.evidence?.class;
1451
+ if (typeof attested === 'string' && BUGFIX_FLOOR_CLASSES.has(attested)) return attested;
1452
+ if (playbookId !== BUGFIX_PLAYBOOK_ID || receipt?.kind !== 'verification' || receipt?.evidence) {
1453
+ return null;
1454
+ }
1455
+ if (receipt.success === true) {
1456
+ const surface = bugFixReceiptSurface(receipt);
1457
+ if (
1458
+ surface
1459
+ && (floor.reproBefore[surface] === true || Number(failureCounts[surface]) > 0)
1460
+ ) {
1461
+ return BUGFIX_REPRO_AFTER;
1462
+ }
1463
+ return null;
1464
+ }
1465
+ // A failed verification on a bug-fix ledger IS the reproduced old failure —
1466
+ // verbatim, no attestation needed.
1467
+ return BUGFIX_REPRO_BEFORE;
1468
+ }
1469
+
1470
+ // Fold one receipt into the floor bank. Idempotent per surface+class: replayed
1471
+ // journal records and the eval-time window fold can run over the same receipt
1472
+ // without double counting. A `repro-after` only banks on a surface with a
1473
+ // confirmed `repro-before` — a passing run on an unrelated surface is not the
1474
+ // fix-proof the playbook demands (same-surface rule, verbatim).
1475
+ function foldBugFixReceipt(floor, receipt, playbookId = null, failureCounts = {}) {
1476
+ const cls = bugFixReceiptClass(receipt, floor, failureCounts, playbookId);
1477
+ if (!cls) return floor;
1478
+ const surface = bugFixReceiptSurface(receipt);
1479
+ // The dedupe stamp identifies one receipt, not one surface: every recorded
1480
+ // repro attempt counts toward the INCONCLUSIVE report, while a replayed
1481
+ // journal record (same receipt) folds idempotently.
1482
+ const stamp = `${cls}:${surface || '-'}:${receipt?.toolUseId ?? receipt?.ts ?? ''}`;
1483
+ if (floor.seen.includes(stamp)) return floor;
1484
+ const next = { ...floor, seen: [...floor.seen, stamp].slice(-BUGFIX_FLOOR_SEEN_CAP) };
1485
+ if (cls === BUGFIX_REPRO_BEFORE) {
1486
+ // Every recorded attempt counts toward the INCONCLUSIVE report; only a
1487
+ // verbatim FAIL confirms the defect reproduced.
1488
+ next.reproAttempts += 1;
1489
+ const result = receipt?.evidence?.result;
1490
+ if (result === 'fail' || (receipt?.success !== true && result !== 'pass' && result !== 'inconclusive')) {
1491
+ next.reproBefore = { ...next.reproBefore, ...(surface ? { [surface]: true } : {}) };
1492
+ }
1493
+ return next;
1494
+ }
1495
+ if (cls === BUGFIX_REPRO_AFTER) {
1496
+ const result = receipt?.evidence?.result;
1497
+ const reproduced = surface !== null && next.reproBefore[surface] === true;
1498
+ if (
1499
+ surface !== null
1500
+ && reproduced
1501
+ && (result === 'pass' || (receipt?.success === true && result !== 'fail' && result !== 'inconclusive'))
1502
+ ) {
1503
+ next.reproAfter = { ...next.reproAfter, [surface]: true };
1504
+ }
1505
+ return next;
1506
+ }
1507
+ // Root-cause classes: the claim names the mechanism; an unattributed
1508
+ // attestation does not satisfy the floor.
1509
+ const claim = receipt?.evidence?.claim;
1510
+ if (typeof claim === 'string' && claim.trim()) {
1511
+ next.rootCause = String(claim).slice(0, EVIDENCE_FIELD_MAX);
1512
+ }
1513
+ return next;
1514
+ }
1515
+
1516
+ // The eval-side view: the bank is authoritative, and receipts still inside the
1517
+ // window are folded through the same rules so a fixture ledger that never ran
1518
+ // recordExecutionReceipt (no bank) is judged identically. Per-surface failure
1519
+ // counts are rebuilt by walking the window in order.
1520
+ function bugFixFloorStatus(ledger = {}, playbookId = null) {
1521
+ const floor = ledger.bugFixFloor || emptyBugFixFloor();
1522
+ const counts = {};
1523
+ let merged = {
1524
+ ...floor,
1525
+ seen: [...floor.seen],
1526
+ reproBefore: { ...floor.reproBefore },
1527
+ reproAfter: { ...floor.reproAfter },
1528
+ };
1529
+ for (const receipt of Array.isArray(ledger.receipts) ? ledger.receipts : []) {
1530
+ merged = foldBugFixReceipt(merged, receipt, playbookId, counts);
1531
+ if (receipt?.kind === 'verification') {
1532
+ const surface = bugFixReceiptSurface(receipt);
1533
+ if (surface) {
1534
+ if (receipt.success === true) delete counts[surface];
1535
+ else counts[surface] = Number(counts[surface] || 0) + 1;
1536
+ }
1537
+ }
1538
+ }
1539
+ return merged;
1540
+ }
1541
+
1542
+ function bugFixFloorMissing(floor) {
1543
+ const missing = [];
1544
+ if (Object.values(floor.reproBefore).every((value) => value !== true)) {
1545
+ missing.push('repro-before');
1546
+ }
1547
+ if (Object.values(floor.reproAfter).every((value) => value !== true)) {
1548
+ missing.push('repro-after');
1549
+ }
1550
+ if (!floor.rootCause) missing.push('root-cause');
1551
+ return missing;
1552
+ }
1553
+
1554
+ // Mint the class onto the receipt (verbatim evidence) and fold it. Runs before
1555
+ // appendReceipt so the derived class is kept inside the receipt window too.
1556
+ function applyBugFixFloorReceipt(ledger, receipt) {
1557
+ if (ledger?.playbookId !== BUGFIX_PLAYBOOK_ID) return receipt;
1558
+ const cls = bugFixReceiptClass(
1559
+ receipt,
1560
+ ledger.bugFixFloor || emptyBugFixFloor(),
1561
+ ledger.verificationFailureCounts || {},
1562
+ ledger.playbookId,
1563
+ );
1564
+ if (cls && !receipt.evidence?.class) {
1565
+ receipt = {
1566
+ ...receipt,
1567
+ evidence: {
1568
+ ...(receipt.evidence && typeof receipt.evidence === 'object' ? receipt.evidence : {}),
1569
+ class: cls,
1570
+ },
1571
+ };
1572
+ }
1573
+ return receipt;
1574
+ }
1575
+
1315
1576
  function applyReceiptToLedger(ledger, receipt, { vibecode = false } = {}) {
1316
1577
  if (!receipt || typeof receipt !== 'object' || !RECEIPT_KINDS.has(receipt.kind)) return null;
1317
1578
  const next = { ...ledger };
@@ -1406,6 +1667,51 @@ function applyReceiptToLedger(ledger, receipt, { vibecode = false } = {}) {
1406
1667
  }
1407
1668
  }
1408
1669
  }
1670
+ // TASK-008 (BL-010): advisory playbook-finding receipts — the latest status
1671
+ // per receipt class is banked on `playbookFindings` so consumers never have
1672
+ // to re-fold an evictable receipt window. An identical-status repeat is a
1673
+ // no-op (the window is not flooded); a status change replaces the entry.
1674
+ if (receipt.kind === 'playbook-finding') {
1675
+ const cls = typeof receipt.class === 'string' ? receipt.class : null;
1676
+ if (!cls) {
1677
+ next.receipts = appendReceipt(next.receipts, receipt);
1678
+ next.updatedAt = Date.now();
1679
+ return { ledger: next, value: null };
1680
+ }
1681
+ const prior = ledger?.playbookFindings?.[cls];
1682
+ if (prior && prior.status === receipt.status) {
1683
+ return { ledger: next, value: null };
1684
+ }
1685
+ next.playbookFindings = {
1686
+ ...(ledger?.playbookFindings && typeof ledger.playbookFindings === 'object'
1687
+ ? ledger.playbookFindings
1688
+ : {}),
1689
+ [cls]: {
1690
+ status: receipt.status ?? null,
1691
+ detail: receipt.detail ?? null,
1692
+ file: receipt.file ?? null,
1693
+ ts: receipt.ts ?? Date.now(),
1694
+ },
1695
+ };
1696
+ next.receipts = appendReceipt(next.receipts, receipt);
1697
+ next.updatedAt = Date.now();
1698
+ return { ledger: next, value: null };
1699
+ }
1700
+ // TASK-009 (BL-011): fold the receipt into the bug-fix floor bank before it
1701
+ // enters the (evictable) receipt window. The fold runs on the pre-update
1702
+ // verificationFailureCounts: a pass over a surface whose failure is still
1703
+ // recorded is the repro-after the floor demands. Bug-fix ledgers only —
1704
+ // non-bug-fix ledgers never grow the field.
1705
+ if (ledger?.playbookId === BUGFIX_PLAYBOOK_ID) {
1706
+ receipt = applyBugFixFloorReceipt(ledger, receipt);
1707
+ next.bugFixFloor = foldBugFixReceipt(
1708
+ ledger.bugFixFloor || emptyBugFixFloor(),
1709
+ receipt,
1710
+ ledger.playbookId,
1711
+ ledger.verificationFailureCounts || {},
1712
+ );
1713
+ }
1714
+
1409
1715
  next.receipts = appendReceipt(next.receipts, receipt);
1410
1716
  next.updatedAt = Date.now();
1411
1717
  return { ledger: next, value: null };
@@ -1490,6 +1796,68 @@ function liveBaseLedger(event, current, payload, routeState, harness, fallbackLe
1490
1796
  return current || fallbackLedger || freshLedger(payload, null, 'unknown');
1491
1797
  }
1492
1798
 
1799
+ // --- TASK-005 (FR-005, BL-007): outcome.observed terminal emissions ---------
1800
+ // OutcomeRecord = ARCH §Data Contracts: outcomeId, runId, routeId,
1801
+ // routeFingerprint, ledgerKey, verdict ∈ OUTCOME_VERDICTS, evidenceIds[],
1802
+ // cost{latencyMs,rounds}, reviewActions[], enforcement: enforced|advisory,
1803
+ // ts — plus the TASK-003 session-boot `request_key` join id. The emit module
1804
+ // (observability-emit.mjs, TASK-003) owns record construction; the ledger
1805
+ // owns WHEN a terminal event fires.
1806
+ let observabilityEmitModule = null;
1807
+
1808
+ async function loadObservabilityEmit() {
1809
+ if (observabilityEmitModule === null) {
1810
+ try {
1811
+ observabilityEmitModule = await import(
1812
+ new URL('./observability-emit.mjs', import.meta.url).href
1813
+ );
1814
+ } catch {
1815
+ observabilityEmitModule = false;
1816
+ }
1817
+ }
1818
+ return observabilityEmitModule === false ? null : observabilityEmitModule;
1819
+ }
1820
+
1821
+ /**
1822
+ * Fire-and-forget `outcome.observed` emission (FR-005). `outcome` is the
1823
+ * emitter's terminal descriptor ({verdict, source, routeId?, routeFingerprint?,
1824
+ * ledgerKey?, runId?, evidenceIds?, cost?, reviewActions?}); the builder stamps
1825
+ * outcomeId/ts/enforcement/request_key. NEVER throws and returns null on any
1826
+ * failure — a dropped telemetry write must never block a receipt, the Stop
1827
+ * gate, or session end (acceptance: emission is fire-and-forget).
1828
+ */
1829
+ export async function emitTerminalOutcome({
1830
+ projectRoot,
1831
+ outcome,
1832
+ payload = {},
1833
+ harness = null,
1834
+ engine = null,
1835
+ env = null,
1836
+ deadlineMs,
1837
+ } = {}) {
1838
+ try {
1839
+ const emit = await loadObservabilityEmit();
1840
+ if (!emit || typeof emit.emitOutcomeObserved !== 'function'
1841
+ || typeof emit.buildOutcomeObservedRecord !== 'function') {
1842
+ return null;
1843
+ }
1844
+ const descriptor = outcome && typeof outcome === 'object' ? outcome : {};
1845
+ const record = emit.buildOutcomeObservedRecord({
1846
+ outcome: { ...descriptor, engine: engine ?? harness ?? descriptor.engine ?? null },
1847
+ projectRoot,
1848
+ payload: payload && typeof payload === 'object' ? payload : {},
1849
+ env: env === null ? process.env : env,
1850
+ });
1851
+ return await emit.emitOutcomeObserved({
1852
+ projectRoot,
1853
+ record,
1854
+ deadlineMs,
1855
+ });
1856
+ } catch {
1857
+ return null;
1858
+ }
1859
+ }
1860
+
1493
1861
  /**
1494
1862
  * The one locked mutation protocol for the execution ledger (TASK-027). Never mutates
1495
1863
  * the ledger without lock ownership; journals the event when the lock cannot be taken.
@@ -1513,8 +1881,15 @@ export async function recordLedgerEvent(event, {
1513
1881
  }
1514
1882
  const eventId = newEventId();
1515
1883
  const target = ledgerPath(projectRoot, payload);
1884
+ // Hoisted so the post-commit outcome emitter sees the same state snapshot
1885
+ // the lock callback evaluated (a fresh read could observe a re-route).
1886
+ let lockedRouteState = routeState || null;
1887
+ // The descriptor the post-commit emitter must fire (lock-local `value` is
1888
+ // nested two levels deep in the lock result — a hoisted ref is clearer).
1889
+ let emittedOutcome = null;
1516
1890
  const outcome = await withLedgerLock(target, { signal, deadlineMs }, async () => {
1517
1891
  const state = routeState || await readRouteState(projectRoot, payload);
1892
+ lockedRouteState = state || null;
1518
1893
  const current = await readExecutionLedger(projectRoot, payload);
1519
1894
  let ledger = liveBaseLedger(event, current, payload, state, harness, fallbackLedger);
1520
1895
  // Reconcile pending journaled events first — inside this caller's acquired lock and
@@ -1523,15 +1898,91 @@ export async function recordLedgerEvent(event, {
1523
1898
  ledger = drained.ledger;
1524
1899
  let value;
1525
1900
  if (event.type === 'receipt') {
1901
+ // TASK-009 (BL-011): stamp the routed playbook on the ledger so the floor
1902
+ // fold can classify repro receipts at apply time — including journaled
1903
+ // replays, where the ledger's own stamp is the only playbook carrier.
1904
+ // A lost route state (state null) keeps the stamp the request already had.
1905
+ const playbookId = state?.routeSummary?.playbookId || null;
1906
+ if (state && ledger.playbookId !== playbookId) {
1907
+ ledger = { ...ledger, playbookId };
1908
+ }
1526
1909
  const applied = applyReceiptToLedger(ledger, event.receipt, { vibecode: event.vibecode === true });
1527
1910
  if (!applied) return { committed: false, eventId, reason: 'invalid-receipt' };
1528
1911
  ledger = applied.ledger;
1912
+ // TASK-005 (FR-005): completion receipts (write/verification terminal
1913
+ // fields — source receipts carry no outcome signal) mint a descriptor
1914
+ // the CALLER emits: record-execution.mjs for the hook path, the
1915
+ // `--record` CLI for direct invocation. Emitted via the shared seam,
1916
+ // never inside the lock.
1917
+ if (event.receipt?.kind === 'write' || event.receipt?.kind === 'verification') {
1918
+ value = {
1919
+ outcome: {
1920
+ verdict: event.receipt.success === true ? 'VERIFIED' : 'FAILED',
1921
+ source: `receipt-${event.receipt.kind}`,
1922
+ routeId: state?.requestKey ?? ledger?.requestKey ?? null,
1923
+ routeFingerprint: ledger?.routeFingerprint ?? state?.fingerprint ?? null,
1924
+ ledgerKey: `sess-${sessionIdentity(payload)}`,
1925
+ evidenceIds: event.receipt.toolUseId ? [String(event.receipt.toolUseId).slice(0, 160)] : [],
1926
+ cost: { latencyMs: null, rounds: Number.isFinite(ledger?.continuationCount) ? ledger.continuationCount : null },
1927
+ reviewActions: [],
1928
+ // TASK-C85-016 (BL-016): per-pair tracking — nullable, additive.
1929
+ implementer: pairIdentity(event.receipt?.implementer ?? payload.implementer),
1930
+ reviewer: pairIdentity(event.receipt?.reviewer ?? payload.reviewer),
1931
+ reviewVerdict: pairIdentity(event.receipt?.reviewVerdict ?? payload.reviewVerdict),
1932
+ },
1933
+ };
1934
+ }
1529
1935
  } else if (event.type === 'continuation') {
1530
1936
  ledger = applyContinuationToLedger(ledger, event);
1531
1937
  } else if (event.type === 'notified') {
1532
1938
  ledger = { ...ledger, notified: true, updatedAt: Date.now() };
1533
1939
  } else if (event.type === 'resumable-run') {
1940
+ // TASK-005 (FR-005): a resumable-run pointer is the interrupted-run
1941
+ // terminal event — emit `outcome.observed` verdict INTERRUPTED exactly
1942
+ // ONCE per task. The ledger's `outcomes` dedupe slot (carried across
1943
+ // re-keys by carriedEvidenceLedger) suppresses journal replays and
1944
+ // resume-fold re-emissions; `run` pointers that fail to apply never
1945
+ // mint an outcome.
1946
+ const resumableBefore = ledger?.resumableRun || null;
1534
1947
  ledger = applyResumableRunToLedger(ledger, event.run);
1948
+ const resumableApplied = ledger.resumableRun !== resumableBefore;
1949
+ const runKey = typeof event.run?.taskId === 'string' && event.run.taskId
1950
+ ? String(event.run.taskId).slice(0, 160)
1951
+ : null;
1952
+ if (resumableApplied && runKey) {
1953
+ const prior = Array.isArray(ledger.outcomes) ? ledger.outcomes : [];
1954
+ if (!prior.some((entry) => entry?.verdict === 'INTERRUPTED' && entry?.runKey === runKey)) {
1955
+ ledger = {
1956
+ ...ledger,
1957
+ outcomes: [...prior, { verdict: 'INTERRUPTED', runKey, at: Date.now() }]
1958
+ .slice(-MAX_LEDGER_OUTCOMES),
1959
+ };
1960
+ emittedOutcome = {
1961
+ verdict: 'INTERRUPTED',
1962
+ source: 'resumable-run',
1963
+ runId: `run-${runKey.replace(/[^a-zA-Z0-9._-]/g, '_')}`,
1964
+ routeId: state?.requestKey ?? ledger?.requestKey ?? null,
1965
+ routeFingerprint: ledger?.routeFingerprint ?? state?.fingerprint ?? null,
1966
+ ledgerKey: `sess-${sessionIdentity(payload)}`,
1967
+ evidenceIds: [],
1968
+ cost: { latencyMs: null, rounds: null },
1969
+ reviewActions: [],
1970
+ };
1971
+ value = {
1972
+ outcome: {
1973
+ verdict: 'INTERRUPTED',
1974
+ source: 'resumable-run',
1975
+ runId: `run-${runKey.replace(/[^a-zA-Z0-9._-]/g, '_')}`,
1976
+ routeId: state?.requestKey ?? ledger?.requestKey ?? null,
1977
+ routeFingerprint: ledger?.routeFingerprint ?? state?.fingerprint ?? null,
1978
+ ledgerKey: `sess-${sessionIdentity(payload)}`,
1979
+ evidenceIds: [],
1980
+ cost: { latencyMs: null, rounds: null },
1981
+ reviewActions: [],
1982
+ },
1983
+ };
1984
+ }
1985
+ }
1535
1986
  } else if (event.type === 'stop-progress') {
1536
1987
  if (!current && drained.applied === 0) {
1537
1988
  // A stop with no ledger at all creates nothing (historic noteStopProgress shape).
@@ -1553,6 +2004,20 @@ export async function recordLedgerEvent(event, {
1553
2004
  return { committed: true, eventId, value, drained: drained.applied, quarantined: drained.quarantined };
1554
2005
  });
1555
2006
  if (outcome.ok) {
2007
+ // TASK-005 (FR-005): ledger-owned terminal events emit their outcome
2008
+ // descriptor after the commit — never inside the lock (the emit runs its
2009
+ // own bounded write) and never for the caller-owned receipt path
2010
+ // (record-execution.mjs / --record emit those).
2011
+ if (event.type === 'resumable-run' && emittedOutcome) {
2012
+ try {
2013
+ await emitTerminalOutcome({
2014
+ projectRoot,
2015
+ outcome: emittedOutcome,
2016
+ payload,
2017
+ harness,
2018
+ });
2019
+ } catch { /* telemetry must never block the ledger */ }
2020
+ }
1556
2021
  // Sampled bounded dir sweep (BUG-C21-11): runs outside the ledger lock so a
1557
2022
  // contended acquisition never pays sweep latency. Advisory — never throws.
1558
2023
  // BUG-C23-10: a persistent failure is still surfaced as a bounded degrade.
@@ -1576,6 +2041,63 @@ export async function recordLedgerEvent(event, {
1576
2041
  return { rejected: true, eventId, reason: journalResult.reason };
1577
2042
  }
1578
2043
 
2044
+ // TASK-009 (BL-011): an evidence attestation rides a receipt as the bounded
2045
+ // ARCH §Evidence Schema tuple {class, surface, result, claim}. Attestation
2046
+ // surfaces, checked in order: `payload.evidence` (the `--record` CLI and
2047
+ // in-proc callers), `payload.tool_input.evidence` (tools that pass extra
2048
+ // input fields through), and a leading `UKIT_EVIDENCE=class=…;surface=…` env
2049
+ // assignment on a Bash command (the one channel a model can mint through an
2050
+ // ordinary Bash call). Values are bounded strings only — no tool output.
2051
+ function normalizeEvidenceAttestation(value) {
2052
+ if (!value || typeof value !== 'object' || Array.isArray(value)) return null;
2053
+ const clean = {};
2054
+ for (const key of JOURNAL_EVIDENCE_FIELDS) {
2055
+ const field = value[key];
2056
+ if (field === undefined || field === null || field === '') continue;
2057
+ const text = String(field).slice(0, EVIDENCE_FIELD_MAX);
2058
+ if (key === 'result' && !EVIDENCE_RESULTS.has(text)) continue;
2059
+ clean[key] = text;
2060
+ }
2061
+ return Object.keys(clean).length > 0 ? clean : null;
2062
+ }
2063
+
2064
+ function parseCommandEvidence(command) {
2065
+ const text = String(command || '').trim();
2066
+ const match = text.match(/^UKIT_EVIDENCE=((?:"[^"]*")|(?:'[^']*')|(?:\S+))/);
2067
+ if (!match) return null;
2068
+ let token = match[1];
2069
+ if (
2070
+ (token.startsWith('"') && token.endsWith('"'))
2071
+ || (token.startsWith("'") && token.endsWith("'"))
2072
+ ) {
2073
+ token = token.slice(1, -1);
2074
+ }
2075
+ const fields = {};
2076
+ for (const part of token.split(';')) {
2077
+ const trimmed = part.trim();
2078
+ if (!trimmed) continue;
2079
+ const eq = trimmed.indexOf('=');
2080
+ if (eq === -1) {
2081
+ if (!('class' in fields)) fields.class = trimmed;
2082
+ continue;
2083
+ }
2084
+ const key = trimmed.slice(0, eq).trim();
2085
+ const fieldValue = trimmed.slice(eq + 1).trim();
2086
+ if (JOURNAL_EVIDENCE_FIELDS.includes(key) && fieldValue) {
2087
+ fields[key] = fieldValue;
2088
+ }
2089
+ }
2090
+ return normalizeEvidenceAttestation(fields);
2091
+ }
2092
+
2093
+ function receiptEvidenceAttestation(payload, toolInput) {
2094
+ return (
2095
+ normalizeEvidenceAttestation(payload?.evidence)
2096
+ || normalizeEvidenceAttestation(toolInput?.evidence)
2097
+ || parseCommandEvidence(toolInput?.command)
2098
+ );
2099
+ }
2100
+
1579
2101
  // TASK-027: the receipt is classified (kind, file, command, targeted/broad scope) from the
1580
2102
  // advisory route state BEFORE locking — that read is read-only — and then applied through
1581
2103
  // recordLedgerEvent, which owns the fail-closed lock/journal protocol.
@@ -1593,6 +2115,7 @@ export async function recordExecutionReceipt({
1593
2115
  const exitCode = extractExitCode(payload);
1594
2116
  const success = !failed && (exitCode === null || exitCode === 0);
1595
2117
  const toolInput = payload.tool_input || {};
2118
+ const evidence = receiptEvidenceAttestation(payload, toolInput);
1596
2119
  const receipt = {
1597
2120
  ts: Date.now(),
1598
2121
  toolName,
@@ -1607,7 +2130,12 @@ export async function recordExecutionReceipt({
1607
2130
  } else if (toolName === 'Edit' || toolName === 'Write') {
1608
2131
  receipt.kind = 'write';
1609
2132
  receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
1610
- } else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
2133
+ } else if (toolName === 'Bash' && (isVerificationCommand(toolInput.command) || evidence)) {
2134
+ // TASK-009 (BL-011): a Bash command carrying an evidence attestation
2135
+ // records as a verification receipt even when the command is not a known
2136
+ // test runner — a repro command, a trace, a one-off check. The typed
2137
+ // verdict stays restricted to real verification commands: attestation
2138
+ // alone must never mint verification evidence.
1611
2139
  receipt.kind = 'verification';
1612
2140
  receipt.command = String(toolInput.command || '').trim();
1613
2141
  // WS-C routed-verification receipt: a command counts as "targeted" when it matches the
@@ -1622,11 +2150,24 @@ export async function recordExecutionReceipt({
1622
2150
  // SPEC-typed-verdicts §2.1: every verification receipt mints a typed verdict record
1623
2151
  // (kind/evidence/verifier/headSha/baseSha/patchId/ts). Explicit payload.verdict
1624
2152
  // fields win so a hook or subagent verifier can attest its own kind and identity.
1625
- receipt.verdict = mintVerificationVerdict(receipt, payload, projectRoot);
2153
+ if (isVerificationCommand(toolInput.command)) {
2154
+ receipt.verdict = mintVerificationVerdict(receipt, payload, projectRoot);
2155
+ }
1626
2156
  } else {
1627
2157
  // Untracked tool: no event, no lock, no write — same as the old unlocked early return.
1628
2158
  return { rejected: true, eventId: null };
1629
2159
  }
2160
+ if (evidence) {
2161
+ receipt.evidence = evidence;
2162
+ }
2163
+
2164
+ // TASK-C85-016 (BL-016): additive per-pair tracking fields — the caller
2165
+ // (hook payload or a future review lane) attests who implemented, who
2166
+ // reviewed, and the verdict. Nullable + bounded; absent → null on the
2167
+ // outcome descriptor so old ledgers stay valid.
2168
+ receipt.implementer = pairIdentity(payload.implementer ?? toolInput.implementer);
2169
+ receipt.reviewer = pairIdentity(payload.reviewer ?? toolInput.reviewer);
2170
+ receipt.reviewVerdict = pairIdentity(payload.reviewVerdict ?? toolInput.reviewVerdict);
1630
2171
 
1631
2172
  return recordLedgerEvent(
1632
2173
  {
@@ -1691,6 +2232,17 @@ function requiredEvidence(state = {}) {
1691
2232
  if (routeSummary.riskEscalation?.level === 'high' && routeSummary.riskEscalation?.stage !== 'shadow') {
1692
2233
  required.push('verification-evidence');
1693
2234
  }
2235
+ // TASK-C85-014 (BL-014): a resolved verification-map recipe rides the route
2236
+ // summary (stamped by stop-coordinator's recipe consult). Its
2237
+ // evidenceRequired[] classes join the required set verbatim — a `|` token
2238
+ // is an anyOf list satisfied by ANY alternative receipt class. Absent or
2239
+ // malformed recipe data adds nothing: the gate stays exactly as before.
2240
+ const recipeRequired = routeSummary.verificationRecipe?.evidenceRequired;
2241
+ if (Array.isArray(recipeRequired)) {
2242
+ for (const item of recipeRequired) {
2243
+ if (typeof item === 'string' && item.trim()) required.push(item.trim());
2244
+ }
2245
+ }
1694
2246
  return [...new Set(required)];
1695
2247
  }
1696
2248
 
@@ -1733,12 +2285,83 @@ function evidenceSatisfied(evidence, ledger = {}, state = {}, { cwd } = {}) {
1733
2285
  }
1734
2286
  return ledger.sourceSucceeded === true;
1735
2287
  }
1736
- return false;
2288
+ // TASK-C85-014 (BL-014): recipe evidence classes (verification-map.json
2289
+ // artifactClasses.<cls>.evidenceRequired) satisfy by receipt attestation /
2290
+ // playbook-finding bank — see receiptEvidenceSatisfied below.
2291
+ return receiptEvidenceSatisfied(evidence, ledger);
2292
+ }
2293
+
2294
+ // ─── Receipt-class satisfaction (TASK-C85-014 fix round) ────────────────────
2295
+ // Semantics identical to receiptEvidenceSatisfied in
2296
+ // ukit/index/verification-map.mjs (canonical: src/index/verificationMap.js).
2297
+ // A shared leaf is impossible — runtime/ must not static-import index/
2298
+ // (partial installs break: the ledger runs with the reader absent) — so parity
2299
+ // is enforced by tests/consistency/receiptSatisfactionParity.test.js instead.
2300
+ // Any `|` alternative satisfies; the playbook-finding bank is latest-wins (a
2301
+ // later 'missing' finding revokes an earlier 'satisfied' one, and a satisfied
2302
+ // class survives receipt eviction via the bank). A failed attestation is not
2303
+ // proof of the class.
2304
+ export function receiptEvidenceSatisfied(requiredClass, ledger = {}) {
2305
+ const token = String(requiredClass || '').trim();
2306
+ const alternatives = token
2307
+ .split('|')
2308
+ .map((entry) => entry.trim())
2309
+ .filter(Boolean);
2310
+ if (alternatives.length === 0) return false;
2311
+ const matchesClass = (value) => {
2312
+ const cls = typeof value === 'string' ? value.trim() : '';
2313
+ return cls === token || alternatives.includes(cls);
2314
+ };
2315
+ // The banked latest-per-class status governs the receipt window; several
2316
+ // banked keys can match one anyOf token, so the newest entry wins. A
2317
+ // playbook-finding receipt folds into the bank regardless of its success
2318
+ // flag — the receipt scan below mirrors that.
2319
+ let bankedTs = -1;
2320
+ let bankedStatus = null;
2321
+ const bank = ledger?.playbookFindings;
2322
+ if (bank && typeof bank === 'object') {
2323
+ for (const [key, entry] of Object.entries(bank)) {
2324
+ if (!matchesClass(key)) continue;
2325
+ const ts = typeof entry?.ts === 'number' ? entry.ts : 0;
2326
+ if (ts >= bankedTs) {
2327
+ bankedTs = ts;
2328
+ bankedStatus = typeof entry?.status === 'string' ? entry.status : null;
2329
+ }
2330
+ }
2331
+ }
2332
+ const receipts = Array.isArray(ledger?.receipts) ? ledger.receipts : [];
2333
+ let attested = false;
2334
+ let scannedStatus = null;
2335
+ for (const receipt of receipts) {
2336
+ if (!receipt || typeof receipt !== 'object') continue;
2337
+ if (receipt.kind === 'playbook-finding') {
2338
+ if (matchesClass(receipt.class) && typeof receipt.status === 'string') {
2339
+ scannedStatus = receipt.status; // chronological — last write wins
2340
+ }
2341
+ continue;
2342
+ }
2343
+ if (receipt.success === false) continue;
2344
+ const cls = receipt?.evidence?.class;
2345
+ if (typeof cls === 'string' && alternatives.includes(cls.trim())) attested = true;
2346
+ }
2347
+ if (attested) return true;
2348
+ const findingStatus = bankedTs >= 0 ? bankedStatus : scannedStatus;
2349
+ return findingStatus === 'satisfied';
1737
2350
  }
1738
2351
 
1739
2352
  function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}, { cwd } = {}) {
1740
2353
  let instruction = null;
1741
- if (missingEvidence.includes('write-evidence')) {
2354
+ // TASK-009 (BL-011): bug-fix floor instructions take precedence — the fix
2355
+ // loop needs the verbatim repro pair and the named mechanism, not more
2356
+ // generic write/verify pushes.
2357
+ if (missingEvidence.includes('repro-before')) {
2358
+ instruction = 'Reproduce the original failure verbatim on the defect surface and record it as the repro-before receipt (failing run). No fix evidence counts before the failure is proven.';
2359
+ } else if (missingEvidence.includes('repro-after')) {
2360
+ instruction = 'Re-run the original repro on the same surface after the fix and record the passing repro-after receipt — a different surface or unrelated green check is not the fix-proof.';
2361
+ } else if (missingEvidence.includes('root-cause')) {
2362
+ instruction = 'Record the root cause: attest the named mechanism (UKIT_EVIDENCE root-cause/instrumentation/behavioral-check with claim) — the rejected hypotheses and the surviving mechanism, one line each.';
2363
+ }
2364
+ if (!instruction && missingEvidence.includes('write-evidence')) {
1742
2365
  if (!ledger.sourceSucceeded) {
1743
2366
  instruction = 'Pull one bounded indexed source slice, then make the requested Edit/Write in this continuation.';
1744
2367
  } else if (ledger.writeAttempted && !ledger.writeSucceeded) {
@@ -1772,6 +2395,21 @@ function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}, {
1772
2395
  }
1773
2396
  }
1774
2397
  }
2398
+ // TASK-C85-014 (BL-014): a missing recipe receipt names the recipe's own
2399
+ // check and its capability-negotiated fallback receipt spec — never a
2400
+ // fabricated command.
2401
+ if (!instruction) {
2402
+ const recipe = routeSummary?.verificationRecipe;
2403
+ const recipeMissing = missingEvidence.filter((item) =>
2404
+ Array.isArray(recipe?.evidenceRequired) && recipe.evidenceRequired.includes(item));
2405
+ if (recipeMissing.length > 0) {
2406
+ instruction = [
2407
+ `The verification-map recipe for artifact class '${recipe.artifactClass || 'unknown'}' requires receipt class ${recipeMissing.join(', ')}.`,
2408
+ recipe.check ? `Check: ${recipe.check}` : null,
2409
+ recipe.fallback ? `Fallback receipt: ${recipe.fallback}` : null,
2410
+ ].filter(Boolean).join(' ');
2411
+ }
2412
+ }
1775
2413
  // WS-C: the target hint travels with the reason whenever impact evidence is missing and
1776
2414
  // the route names expected files — it must survive branch precedence above.
1777
2415
  if (missingEvidence.includes('impact-evidence')) {
@@ -1858,6 +2496,44 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1858
2496
  || (ledger.promptKey && evidencePromptKey(state) === ledger.promptKey);
1859
2497
  const effectiveLedger = sameRequest ? ledger : {};
1860
2498
  const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, state, { cwd }));
2499
+ // TASK-009 (BL-011): the bug-fix floor. On a `playbookId: bug-fix` route a
2500
+ // green verdict requires verbatim repro-before (fail) + repro-after (pass)
2501
+ // receipts on the same surface plus a named root cause in the exec-ledger —
2502
+ // symptom-patching without receipts can never release green. Scoped to
2503
+ // bug-fix only: every other playbook and plain-mode route keeps the generic
2504
+ // completion behavior (false-block = 0 on the P2 gate).
2505
+ const playbookId = routeSummary.playbookId || null;
2506
+ if (playbookId === BUGFIX_PLAYBOOK_ID) {
2507
+ const floor = bugFixFloorStatus(effectiveLedger, playbookId);
2508
+ missingEvidence.push(...bugFixFloorMissing(floor));
2509
+ // An unreproducible defect ends INCONCLUSIVE, never a fabricated pass and
2510
+ // never an unbounded recovery loop: the defect could not be reproduced in
2511
+ // the recorded attempts, so the fix floor is unreachable. The reason names
2512
+ // the attempt count and the runtime-forensics follow-on (playbook step 5).
2513
+ if (
2514
+ floor.reproAttempts > 0
2515
+ && Object.values(floor.reproBefore).every((value) => value !== true)
2516
+ && Object.values(floor.reproAfter).every((value) => value !== true)
2517
+ ) {
2518
+ return {
2519
+ continue: false,
2520
+ notify: true,
2521
+ inconclusive: true,
2522
+ outcome: 'inconclusive',
2523
+ missingEvidence,
2524
+ reason: `UKit completion gate: bug-fix INCONCLUSIVE — the defect did not reproduce in ${floor.reproAttempts} recorded repro attempt(s), so the fix floor (verbatim repro-before fail + repro-after pass on the same surface) is unreachable. Report verified-vs-not verbatim, the attempts made, and the runtime-forensics follow-on lane (live symptom). Never claim a fix without the repro pair.`,
2525
+ };
2526
+ }
2527
+ if (missingEvidence.length === 0) {
2528
+ return {
2529
+ continue: false,
2530
+ notify: false,
2531
+ complete: true,
2532
+ outcome: 'verified',
2533
+ missingEvidence: [],
2534
+ };
2535
+ }
2536
+ }
1861
2537
  if (missingEvidence.length === 0) {
1862
2538
  // Silent success: the route is present and every required evidence is satisfied. Marked
1863
2539
  // `complete` so the CLI dispatch recognizes it BEFORE the loud final else — otherwise a
@@ -1873,15 +2549,26 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1873
2549
  // recovery gate — only a no-mutation, recommend-only investigation that has not already
1874
2550
  // observed a failed verification gets this release valve. A route that named an actionable
1875
2551
  // command, or a failing check, has concrete unfinished work and must keep recovering.
2552
+ // TASK-C85-019 fix B: the release criterion is the find-cause contract's
2553
+ // completionRule ('never-claim-fixed-without-write-and-verification' — a FIX
2554
+ // claim needs evidence, a mutation is never mandatory), NOT the verification
2555
+ // policyMode. policyMode derives from the verification recommendation, so any
2556
+ // route that resolved targeted commands lands on auto-run-* and the old
2557
+ // recommend-only check never fired in a real project. An absent rule keeps
2558
+ // the pre-C85-019 states and the canonical find-cause contract on the valve.
2559
+ const findCauseNoMutationContract = mode === 'find-cause'
2560
+ && (routeSummary?.executionContract?.completionRule == null
2561
+ || routeSummary.executionContract.completionRule === 'never-claim-fixed-without-write-and-verification');
1876
2562
  if (
1877
- mode === 'find-cause'
1878
- && routeSummary.policyMode === 'recommend-only'
2563
+ findCauseNoMutationContract
2564
+ && playbookId !== BUGFIX_PLAYBOOK_ID
1879
2565
  && !effectiveLedger.writeAttempted
1880
2566
  && !effectiveLedger.verificationFailed
1881
2567
  ) {
1882
2568
  return {
1883
2569
  continue: false,
1884
2570
  notify: true,
2571
+ gateRelease: 'find-cause-no-mutation',
1885
2572
  missingEvidence,
1886
2573
  reason: 'UKit investigation ended without a mutation. A clean audit is valid; report whether no actionable defect was found or a concrete blocker remains. Do not claim a bug was fixed without write and verification evidence.',
1887
2574
  };
@@ -1904,6 +2591,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1904
2591
  return {
1905
2592
  continue: false,
1906
2593
  notify: true,
2594
+ gateRelease: 'map-impact-no-mutation',
1907
2595
  missingEvidence,
1908
2596
  reason: 'UKit impact analysis ended without a mutation. A completed impact map with no warranted change is valid; report the findings and whether a follow-up edit is needed. Do not claim any fix without write and verification evidence.',
1909
2597
  };
@@ -1915,6 +2603,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1915
2603
  continue: false,
1916
2604
  notify: true,
1917
2605
  missingEvidence,
2606
+ ...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'inconclusive' } : {}),
1918
2607
  reason: `UKit completion gate: missing ${missingEvidence.join(', ')}. This mode does not auto-continue; tell the user what is unfinished.`,
1919
2608
  };
1920
2609
  }
@@ -1937,6 +2626,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1937
2626
  capped: true,
1938
2627
  notify: true,
1939
2628
  missingEvidence,
2629
+ ...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'inconclusive' } : {}),
1940
2630
  reason: `UKit continuation cap reached with missing evidence: ${missingEvidence.join(', ')}.`,
1941
2631
  };
1942
2632
  }
@@ -1946,6 +2636,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1946
2636
  notify: true,
1947
2637
  capped: true,
1948
2638
  missingEvidence,
2639
+ ...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'blocked' } : {}),
1949
2640
  reason: `UKit stopping with unfinished work: ${missingEvidence.join(', ')}. Tell the user what is unfinished and stop; do not continue further.`,
1950
2641
  };
1951
2642
  }
@@ -1964,6 +2655,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1964
2655
  notify: true,
1965
2656
  noProgressCount,
1966
2657
  missingEvidence,
2658
+ ...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'inconclusive' } : {}),
1967
2659
  reason: `UKit vibecode liveness breaker: ${noProgressCount} continuations produced no new verifiable progress (missing evidence: ${missingEvidence.join(', ')}). Stop and tell the user what is unfinished; do not keep continuing without new evidence.`,
1968
2660
  };
1969
2661
  }
@@ -1974,6 +2666,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1974
2666
  capped: true,
1975
2667
  noProgressCount,
1976
2668
  missingEvidence,
2669
+ ...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'blocked' } : {}),
1977
2670
  reason: `UKit stopping with unfinished work: ${noProgressCount} continuations produced no new verifiable progress (${missingEvidence.join(', ')}). Tell the user what is unfinished and stop; do not continue further.`,
1978
2671
  };
1979
2672
  }
@@ -1984,6 +2677,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
1984
2677
  continue: true,
1985
2678
  missingEvidence,
1986
2679
  noProgressCount,
2680
+ ...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'blocked' } : {}),
1987
2681
  reason: [
1988
2682
  `UKit completion gate: missing ${missingEvidence.join(', ')}.`,
1989
2683
  instruction,
@@ -2125,6 +2819,71 @@ async function runEvaluateStop() {
2125
2819
 
2126
2820
  const result = evaluateCompletion({ state, ledger, cwd: projectRoot });
2127
2821
 
2822
+ // TASK-C85-019 fix C: a mutating-mode route released by a no-mutation valve
2823
+ // while evidence is missing is the anomaly signal for misrouted advisory
2824
+ // turns — measure it. One bounded gate-release row lands in decisions.tsv
2825
+ // (mode + missing evidence + prompt hash, never prompt text), and
2826
+ // UKIT_GATE_DEBUG=1 mirrors the same signal on stderr.
2827
+ if (result.gateRelease) {
2828
+ const gateMode = state?.routeSummary?.executionMode
2829
+ || state?.routeSummary?.approachSelector?.executionMode
2830
+ || 'unknown';
2831
+ try {
2832
+ await appendDecisionReceipt(projectRoot, {
2833
+ kind: 'gate-release',
2834
+ boundary: 'completion-gate',
2835
+ stage: 'no-mutation',
2836
+ outcomeClass: gateMode,
2837
+ checkpoint: evidencePromptKey(state) ?? 'no-prompt-hash',
2838
+ decisionKeys: Array.isArray(result.missingEvidence) ? result.missingEvidence : [],
2839
+ agreement: 'released',
2840
+ fallbackCode: result.gateRelease,
2841
+ });
2842
+ } catch { /* advisory */ }
2843
+ if (process.env.UKIT_GATE_DEBUG === '1') {
2844
+ process.stderr.write(
2845
+ `[ukit-completion] gate-release mode=${gateMode} valve=${result.gateRelease} missing=${(result.missingEvidence || []).join(',') || 'none'} prompt=${evidencePromptKey(state) || 'none'}\n`,
2846
+ );
2847
+ }
2848
+ }
2849
+
2850
+ // TASK-005 (FR-005): the completion evaluator's terminal verdict emits one
2851
+ // `outcome.observed` record per evaluated stop. The verdict maps the gate's
2852
+ // own decision shape — a still-blocked stop is BLOCKED, a capped/unfinished
2853
+ // release FAILED, a notice-only or undecidable release INCONCLUSIVE, and a
2854
+ // satisfied route VERIFIED. Fire-and-forget: awaited inside this child's
2855
+ // own deadline so the record is durable before exit, but a failed emit
2856
+ // never changes the decision below.
2857
+ {
2858
+ // continue → still blocked; satisfied → VERIFIED; an explicit inconclusive
2859
+ // release (the bug-fix unreproducible path) → INCONCLUSIVE; a
2860
+ // capped/missing-evidence release → FAILED; every other release is
2861
+ // INCONCLUSIVE (a notice-only or undecidable end is an honest non-verdict).
2862
+ const stopVerdict = result.continue === true ? 'BLOCKED'
2863
+ : result.complete === true ? 'VERIFIED'
2864
+ : result.inconclusive === true ? 'INCONCLUSIVE'
2865
+ : result.capped === true || (Array.isArray(result.missingEvidence) && result.missingEvidence.length > 0) ? 'FAILED'
2866
+ : 'INCONCLUSIVE';
2867
+ try {
2868
+ await emitTerminalOutcome({
2869
+ projectRoot,
2870
+ outcome: {
2871
+ verdict: stopVerdict,
2872
+ source: 'stop-evaluate',
2873
+ routeId: state?.requestKey ?? ledger?.requestKey ?? null,
2874
+ routeFingerprint: ledger?.routeFingerprint ?? state?.fingerprint ?? null,
2875
+ ledgerKey: `sess-${sessionIdentity(payload)}`,
2876
+ evidenceIds: [],
2877
+ cost: { latencyMs: null, rounds: Number.isFinite(ledger?.continuationCount) ? ledger.continuationCount : null },
2878
+ reviewActions: [],
2879
+ },
2880
+ payload,
2881
+ harness: process.env.UKIT_HARNESS || null,
2882
+ deadlineMs: 200,
2883
+ });
2884
+ } catch { /* advisory */ }
2885
+ }
2886
+
2128
2887
  // A reentrant Stop (Claude Code re-fires Stop after this hook already blocked once) is
2129
2888
  // gated exactly like any other Stop: while evidence is still missing it blocks again with
2130
2889
  // the actionable reason instead of silently releasing the recovery turn. Loop termination
@@ -2162,12 +2921,24 @@ async function main() {
2162
2921
  const payload = JSON.parse((await readStdin()) || '{}');
2163
2922
  const projectRoot = process.env.CLAUDE_PROJECT_DIR || payload.cwd || process.cwd();
2164
2923
  if (process.argv.includes('--record')) {
2165
- await recordExecutionReceipt({
2924
+ const receiptResult = await recordExecutionReceipt({
2166
2925
  projectRoot,
2167
2926
  payload,
2168
2927
  toolName: payload.tool_name,
2169
2928
  harness: process.env.UKIT_HARNESS || 'claude-code',
2170
2929
  });
2930
+ // TASK-005 (FR-005): emit the committed completion receipt's outcome —
2931
+ // same terminal contract record-execution.mjs performs for hook callers.
2932
+ if (receiptResult?.value?.outcome) {
2933
+ try {
2934
+ await emitTerminalOutcome({
2935
+ projectRoot,
2936
+ outcome: receiptResult.value.outcome,
2937
+ payload,
2938
+ harness: process.env.UKIT_HARNESS || 'claude-code',
2939
+ });
2940
+ } catch { /* advisory: never block the receipt path */ }
2941
+ }
2171
2942
  }
2172
2943
  }
2173
2944
 
@@ -2234,7 +3005,14 @@ export async function appendDecision(projectRoot, {
2234
3005
  // 'shadow' (advisory run, deterministic policy stayed authoritative),
2235
3006
  // 'applied' (canary/default accepted answer), 'fallback' (adapter failed and
2236
3007
  // the deterministic baseline stayed in force).
2237
- const DECISION_RECEIPT_KINDS = new Set(['shadow', 'applied', 'fallback']);
3008
+ // TASK-C85-015 (BL-015): 'review-policy' receipts carry the review decision
3009
+ // evaluator's advisory verdicts (action/namedCheck/signalsHit/round per SPEC §7)
3010
+ // — additive kind; the shadow/applied/fallback semantics are unchanged.
3011
+ // TASK-C85-019 fix C: 'gate-release' receipts carry the completion gate's
3012
+ // no-mutation valve releases — a mutating-mode route let go while evidence is
3013
+ // missing is the anomaly signal for misrouted advisory turns, made measurable
3014
+ // in decisions.tsv (mode + missingEvidence + prompt hash; never prompt text).
3015
+ const DECISION_RECEIPT_KINDS = new Set(['shadow', 'applied', 'fallback', 'review-policy', 'gate-release']);
2238
3016
  const DECISION_RECEIPT_CELL_MAX = 160;
2239
3017
 
2240
3018
  function receiptCell(value) {