@ngockhoale/ukit 3.3.3 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/manifests/engineConformance.yaml +17 -1
- package/manifests/hostCapabilities.yaml +68 -1
- package/manifests/platform.full.yaml +138 -0
- package/manifests/platform.user.yaml +255 -3
- package/package.json +1 -1
- package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
- package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
- package/scripts/probe/codex-capability-probe.mjs +169 -0
- package/src/cli/commands/doctor.js +168 -0
- package/src/cli/commands/indexTools.js +7 -0
- package/src/cli/commands/metrics.js +66 -2
- package/src/cli/commands/playbook.js +4 -4
- package/src/cli/commands/vm.js +49 -8
- package/src/core/agentRuntime/adapters.js +328 -27
- package/src/core/agentRuntime/artifacts.js +89 -0
- package/src/core/agentRuntime/context.js +345 -1
- package/src/core/agentRuntime/contract.js +296 -0
- package/src/core/agentRuntime/eventStore.js +176 -0
- package/src/core/agentRuntime/shadowRun.js +481 -5
- package/src/core/agentRuntime/telemetry.js +121 -0
- package/src/core/observability/emit/lifecycle.js +68 -1
- package/src/core/observability/emit/sessionBoot.js +393 -0
- package/src/core/observability/privacy/allowlist.js +10 -1
- package/src/core/observability/schema/registry.js +10 -0
- package/src/core/runtimeConfig.js +133 -0
- package/src/core/userPlaybooks.js +18 -3
- package/src/decision/registry.js +19 -0
- package/src/diagnostics/feedbackEvents.js +7 -4
- package/src/diagnostics/routeOutcomes.js +51 -6
- package/src/diagnostics/skillAccuracy.js +43 -3
- package/src/index/crossCheckMatrix.js +412 -0
- package/src/index/fixLoopEscalation.js +453 -0
- package/src/index/playbookRegistry.js +691 -0
- package/src/index/reviewPolicy.js +368 -0
- package/src/index/routeResolver.js +915 -0
- package/src/index/sessionHistoryExtractor.js +359 -0
- package/src/index/taskRouting.js +764 -581
- package/src/index/tierSelection.js +308 -0
- package/src/index/verificationMap.js +404 -0
- package/template_project/.claude/hooks/observability-emit.mjs +14 -0
- package/template_project/.claude/hooks/record-execution.mjs +19 -1
- package/template_project/.claude/hooks/skill-router.sh +691 -25
- package/template_project/.claude/hooks/verification-guard.sh +230 -1
- package/template_project/.claude/settings.json +2 -2
- package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
- package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
- package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
- package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
- package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
- package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
- package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
- package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
- package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
- package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
- package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
- package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
- package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
- package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
- package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
- package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
- package/template_project/.codex/README.md +8 -0
- package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
- package/template_project/ukit/README.md +1 -1
- package/template_project/ukit/storage/config.json +20 -0
- package/template_user/playbooks/architecture-decision.md +28 -0
- package/template_user/playbooks/autonomous-run.md +43 -0
- package/template_user/playbooks/autopilot-full.md +59 -0
- package/template_user/playbooks/autopilot-stack.md +54 -0
- package/template_user/playbooks/babysit.md +39 -0
- package/template_user/playbooks/bug-fix.md +3 -1
- package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
- package/template_user/playbooks/hillclimb.md +44 -0
- package/template_user/playbooks/investigation.md +21 -0
- package/template_user/playbooks/migration.md +21 -0
- package/template_user/playbooks/open-pr.md +48 -0
- package/template_user/playbooks/orchestrate.md +45 -0
- package/template_user/playbooks/performance.md +33 -0
- package/template_user/playbooks/prototype.md +28 -0
- package/template_user/playbooks/refactor.md +19 -0
- package/template_user/playbooks/release.md +28 -0
- package/template_user/playbooks/runtime-forensics.md +23 -0
- package/template_user/playbooks/session-pickup.md +31 -0
- package/template_user/playbooks/shipping.md +53 -0
- package/template_user/playbooks/skill-evaluation.md +48 -0
- package/template_user/playbooks/small-feature.md +20 -0
- package/template_user/playbooks/verification-map.json +153 -0
- package/template_user/playbooks/verification.md +22 -0
- package/template_user/playbooks/worktree-cleanup.md +37 -0
|
@@ -17,6 +17,10 @@ const LEDGER_VERSION = 1;
|
|
|
17
17
|
const RESUME_INTENT_VERSION = 1;
|
|
18
18
|
const RESUME_INTENT_TTL_MS = 30 * 60 * 1000;
|
|
19
19
|
const MAX_RECEIPTS = 24;
|
|
20
|
+
// TASK-005 (FR-005): bounded OutcomeRecord dedupe slot — a resumable-run
|
|
21
|
+
// pointer (interrupted run) emits `outcome.observed` once per task; replays
|
|
22
|
+
// and resume folds are seen via this list, never re-emitted.
|
|
23
|
+
const MAX_LEDGER_OUTCOMES = 16;
|
|
20
24
|
const MAX_SOURCE_FILES = 16;
|
|
21
25
|
const MAX_CONTINUATIONS = 6;
|
|
22
26
|
// A malformed stdin payload crashes `--evaluate-stop` before a route/session can be read, so
|
|
@@ -91,6 +95,16 @@ function sessionIdentity(payload = {}) {
|
|
|
91
95
|
return 'default';
|
|
92
96
|
}
|
|
93
97
|
|
|
98
|
+
// TASK-C85-016 (BL-016): bounded identity string for the per-pair receipt
|
|
99
|
+
// fields (implementer/reviewer/reviewVerdict). Model/role ids are short
|
|
100
|
+
// strings; anything else collapses to null so OutcomeRecord fields stay
|
|
101
|
+
// nullable-clean.
|
|
102
|
+
function pairIdentity(value) {
|
|
103
|
+
return typeof value === 'string' && value.trim()
|
|
104
|
+
? value.trim().slice(0, 96)
|
|
105
|
+
: null;
|
|
106
|
+
}
|
|
107
|
+
|
|
94
108
|
function ledgerPath(projectRoot, payload = {}) {
|
|
95
109
|
return path.join(
|
|
96
110
|
projectRoot,
|
|
@@ -312,7 +326,14 @@ function hasUnfinishedCompletion(state = {}, ledger = {}) {
|
|
|
312
326
|
if (!IMPLEMENT_MODES.has(mode)) return false;
|
|
313
327
|
const required = requiredEvidence(state);
|
|
314
328
|
if (required.length === 0) return false;
|
|
315
|
-
|
|
329
|
+
const unfinished = required.some((item) => !evidenceSatisfied(item, ledger, state));
|
|
330
|
+
if (unfinished) return true;
|
|
331
|
+
// TASK-009 (BL-011): a bug-fix route is also unfinished while its floor is
|
|
332
|
+
// missing — the generic evidence alone never releases a bug fix.
|
|
333
|
+
if (state?.routeSummary?.playbookId === BUGFIX_PLAYBOOK_ID) {
|
|
334
|
+
return bugFixFloorMissing(bugFixFloorStatus(ledger, BUGFIX_PLAYBOOK_ID)).length > 0;
|
|
335
|
+
}
|
|
336
|
+
return false;
|
|
316
337
|
}
|
|
317
338
|
|
|
318
339
|
function resumeSessionHash(sessionId) {
|
|
@@ -480,7 +501,7 @@ function compactReceipt(receipt) {
|
|
|
480
501
|
kind: receipt.kind,
|
|
481
502
|
success: receipt.success,
|
|
482
503
|
};
|
|
483
|
-
for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error', 'verdict']) {
|
|
504
|
+
for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error', 'verdict', 'evidence', 'evidenceSurface', 'class', 'status', 'detail']) {
|
|
484
505
|
if (receipt[key] !== undefined && receipt[key] !== null && receipt[key] !== '') {
|
|
485
506
|
compact[key] = receipt[key];
|
|
486
507
|
}
|
|
@@ -894,10 +915,20 @@ function carriedEvidenceLedger(fresh, current) {
|
|
|
894
915
|
// Verification-loop tracking and any minted blocker belong to the same logical request
|
|
895
916
|
// too — dropping a blocker on re-key would resume the exact loop it recorded.
|
|
896
917
|
failedVerificationStreak: current.failedVerificationStreak || null,
|
|
918
|
+
playbookId: fresh.playbookId || current.playbookId || null,
|
|
919
|
+
// The bug-fix floor bank is monotone request-scoped evidence — a same-prompt
|
|
920
|
+
// re-key must keep it or the gate would re-demand repro receipts mid-request.
|
|
921
|
+
bugFixFloor: current.bugFixFloor || fresh.bugFixFloor || null,
|
|
922
|
+
// TASK-008: the playbook-finding bank is request-scoped like the floor —
|
|
923
|
+
// a same-prompt re-key must keep the latest per-class statuses.
|
|
924
|
+
playbookFindings: current.playbookFindings || fresh.playbookFindings || null,
|
|
897
925
|
verificationFailureCounts: current.verificationFailureCounts || {},
|
|
898
926
|
blocker: current.blocker || null,
|
|
899
927
|
// Journal bookkeeping belongs to the file, not the request: the dedupe memory must
|
|
900
928
|
// survive a re-key or replayed journal events would be applied twice (TASK-027).
|
|
929
|
+
// Outcome-emit dedupe rides the same request carry — a same-prompt re-key
|
|
930
|
+
// must not re-emit an INTERRUPTED record it already wrote.
|
|
931
|
+
outcomes: [...(current.outcomes || []), ...(fresh.outcomes || [])].slice(-MAX_LEDGER_OUTCOMES),
|
|
901
932
|
journalSeen: current.journalSeen || [],
|
|
902
933
|
journalQuarantined: current.journalQuarantined || 0,
|
|
903
934
|
lastJournalQuarantineAt: current.lastJournalQuarantineAt || null,
|
|
@@ -927,6 +958,16 @@ function freshLedger(payload, routeState, harness) {
|
|
|
927
958
|
requestKey: routeState?.requestKey || null,
|
|
928
959
|
promptKey: evidencePromptKey(routeState),
|
|
929
960
|
routeFingerprint: routeState?.fingerprint || null,
|
|
961
|
+
// TASK-009 (BL-011): the routed playbook rides the ledger so the bug-fix
|
|
962
|
+
// completion floor can classify repro receipts at apply time — including
|
|
963
|
+
// journaled replays, where no route state is passed.
|
|
964
|
+
playbookId: routeState?.routeSummary?.playbookId || null,
|
|
965
|
+
// Bug-fix completion floor bank: monotone, request-scoped, never evicted by
|
|
966
|
+
// the receipt window. null until the first floor receipt lands.
|
|
967
|
+
bugFixFloor: null,
|
|
968
|
+
// TASK-008 (BL-010): latest-per-class playbook-finding statuses, banked
|
|
969
|
+
// outside the evictable receipt window.
|
|
970
|
+
playbookFindings: null,
|
|
930
971
|
sourceSucceeded: false,
|
|
931
972
|
sourceFiles: [],
|
|
932
973
|
writeAttempted: false,
|
|
@@ -946,6 +987,9 @@ function freshLedger(payload, routeState, harness) {
|
|
|
946
987
|
bankedWrites: {},
|
|
947
988
|
bankedVerifications: {},
|
|
948
989
|
notified: false,
|
|
990
|
+
// TASK-005 (FR-005): OutcomeRecord dedupe ledger ({verdict, runKey, at}).
|
|
991
|
+
// Emitted-once for INTERRUPTED survives re-keys via carriedEvidenceLedger.
|
|
992
|
+
outcomes: [],
|
|
949
993
|
updatedAt: Date.now(),
|
|
950
994
|
};
|
|
951
995
|
}
|
|
@@ -971,7 +1015,7 @@ const MAX_JOURNAL_SEEN = 64;
|
|
|
971
1015
|
const MAX_JOURNAL_RECORDS = 128;
|
|
972
1016
|
const MAX_JOURNAL_QUARANTINE_LINES = 32;
|
|
973
1017
|
const LEDGER_EVENT_TYPES = new Set(['receipt', 'continuation', 'notified', 'stop-progress', 'resumable-run']);
|
|
974
|
-
const RECEIPT_KINDS = new Set(['source', 'write', 'verification']);
|
|
1018
|
+
const RECEIPT_KINDS = new Set(['source', 'write', 'verification', 'playbook-finding']);
|
|
975
1019
|
|
|
976
1020
|
function journalPathFor(target) {
|
|
977
1021
|
return `${target}.journal`;
|
|
@@ -1077,6 +1121,24 @@ function newEventId() {
|
|
|
1077
1121
|
|
|
1078
1122
|
const JOURNAL_RECEIPT_FIELDS = ['ts', 'kind', 'toolName', 'toolUseId', 'success', 'exitCode', 'file', 'command', 'scope'];
|
|
1079
1123
|
|
|
1124
|
+
// TASK-009 (BL-011): evidence attestations are whitelisted verbatim — class /
|
|
1125
|
+
// surface / result / claim, the ARCH §Evidence Schema fields a receipt may
|
|
1126
|
+
// carry — because a journaled bug-fix receipt must replay with its floor
|
|
1127
|
+
// evidence intact. The nested whitelist bounds the journal record the same way
|
|
1128
|
+
// the flat one bounds the receipt.
|
|
1129
|
+
const JOURNAL_EVIDENCE_FIELDS = ['class', 'surface', 'result', 'claim'];
|
|
1130
|
+
|
|
1131
|
+
function sanitizeEvidenceForJournal(evidence) {
|
|
1132
|
+
if (!evidence || typeof evidence !== 'object' || Array.isArray(evidence)) return null;
|
|
1133
|
+
const clean = {};
|
|
1134
|
+
for (const key of JOURNAL_EVIDENCE_FIELDS) {
|
|
1135
|
+
const value = evidence[key];
|
|
1136
|
+
if (value === undefined || value === null || value === '') continue;
|
|
1137
|
+
clean[key] = String(value).slice(0, 400);
|
|
1138
|
+
}
|
|
1139
|
+
return Object.keys(clean).length > 0 ? clean : null;
|
|
1140
|
+
}
|
|
1141
|
+
|
|
1080
1142
|
function sanitizeReceiptForJournal(receipt) {
|
|
1081
1143
|
const clean = {};
|
|
1082
1144
|
for (const key of JOURNAL_RECEIPT_FIELDS) {
|
|
@@ -1084,6 +1146,8 @@ function sanitizeReceiptForJournal(receipt) {
|
|
|
1084
1146
|
clean[key] = receipt[key];
|
|
1085
1147
|
}
|
|
1086
1148
|
}
|
|
1149
|
+
const evidence = sanitizeEvidenceForJournal(receipt && receipt.evidence);
|
|
1150
|
+
if (evidence) clean.evidence = evidence;
|
|
1087
1151
|
return clean;
|
|
1088
1152
|
}
|
|
1089
1153
|
|
|
@@ -1312,6 +1376,203 @@ async function drainJournal(target, { ledger, payload, signal, deadlineMs }) {
|
|
|
1312
1376
|
return { ledger: next, applied, quarantined: malformed.length, commit };
|
|
1313
1377
|
}
|
|
1314
1378
|
|
|
1379
|
+
// --- TASK-009 (BL-011): bug-fix completion floor --------------------------------
|
|
1380
|
+
// A green verdict on a `playbookId: bug-fix` route requires verbatim
|
|
1381
|
+
// `repro-before` (failing) + `repro-after` (passing) evidence receipts on the
|
|
1382
|
+
// SAME surface plus a named root cause in the exec-ledger — a symptom patch
|
|
1383
|
+
// without receipts cannot pass the gate. A defect that honest attempts cannot
|
|
1384
|
+
// reproduce ends INCONCLUSIVE with the attempt count and the runtime-forensics
|
|
1385
|
+
// follow-on named; the floor never fabricates a pass.
|
|
1386
|
+
//
|
|
1387
|
+
// Receipts may carry an explicit ARCH §Evidence Schema attestation
|
|
1388
|
+
// (`receipt.evidence = {class, surface, result, claim}`) — minted from
|
|
1389
|
+
// `payload.evidence`, `payload.tool_input.evidence`, or a leading
|
|
1390
|
+
// `UKIT_EVIDENCE=class=…;surface=…` env assignment on a Bash command. On a
|
|
1391
|
+
// bug-fix ledger the apply step additionally classifies every un-attested
|
|
1392
|
+
// verification receipt itself: a failed verification is a `repro-before`
|
|
1393
|
+
// (the old failure reproduced verbatim), and a success on a surface with a
|
|
1394
|
+
// recorded failure is the `repro-after` — so the live hook path produces the
|
|
1395
|
+
// floor without any attestation at all. Derived classes are stamped back onto
|
|
1396
|
+
// the receipt, which keeps them verbatim inside the receipt window.
|
|
1397
|
+
//
|
|
1398
|
+
// The bank lives in `ledger.bugFixFloor` and is monotone per request: receipts
|
|
1399
|
+
// evicted by the MAX_RECEIPTS window can never un-bank evidence. `seen` dedupes
|
|
1400
|
+
// per surface+class so replayed journal records and eval-time receipt folds
|
|
1401
|
+
// are idempotent.
|
|
1402
|
+
const BUGFIX_PLAYBOOK_ID = 'bug-fix';
|
|
1403
|
+
const BUGFIX_REPRO_BEFORE = 'repro-before';
|
|
1404
|
+
const BUGFIX_REPRO_AFTER = 'repro-after';
|
|
1405
|
+
const BUGFIX_FLOOR_SEEN_CAP = 32;
|
|
1406
|
+
// Root-cause classes: the mechanism attestation carries the named cause in
|
|
1407
|
+
// `claim` (ARCH §Done Criteria: instrumentation/behavioral-check for the named
|
|
1408
|
+
// mechanism; `root-cause` attested directly is accepted too).
|
|
1409
|
+
const BUGFIX_ROOT_CAUSE_CLASSES = new Set(['root-cause', 'instrumentation', 'behavioral-check']);
|
|
1410
|
+
const BUGFIX_FLOOR_CLASSES = new Set([
|
|
1411
|
+
BUGFIX_REPRO_BEFORE,
|
|
1412
|
+
BUGFIX_REPRO_AFTER,
|
|
1413
|
+
...BUGFIX_ROOT_CAUSE_CLASSES,
|
|
1414
|
+
]);
|
|
1415
|
+
const EVIDENCE_RESULTS = new Set(['pass', 'fail', 'inconclusive']);
|
|
1416
|
+
const EVIDENCE_FIELD_MAX = 400;
|
|
1417
|
+
|
|
1418
|
+
function emptyBugFixFloor() {
|
|
1419
|
+
return {
|
|
1420
|
+
seen: [],
|
|
1421
|
+
reproAttempts: 0,
|
|
1422
|
+
// surface -> true: the original failure reproduced verbatim on it.
|
|
1423
|
+
reproBefore: {},
|
|
1424
|
+
// surface -> true: the same surface went green after the fix.
|
|
1425
|
+
reproAfter: {},
|
|
1426
|
+
rootCause: null,
|
|
1427
|
+
};
|
|
1428
|
+
}
|
|
1429
|
+
|
|
1430
|
+
// A receipt's floor surface is its attested surface verbatim, else the
|
|
1431
|
+
// terminal-shell identity of its command — env-var prefixes (the
|
|
1432
|
+
// UKIT_EVIDENCE attestation channel) never change the surface of a command.
|
|
1433
|
+
function bugFixReceiptSurface(receipt) {
|
|
1434
|
+
const attested = receipt?.evidence?.surface;
|
|
1435
|
+
if (typeof attested === 'string' && attested) return attested;
|
|
1436
|
+
const command = String(receipt?.command || '').trim();
|
|
1437
|
+
if (!command) return null;
|
|
1438
|
+
const stripped = command.replace(/^(?:[A-Za-z_][A-Za-z0-9_]*=\S+\s+)+/, '');
|
|
1439
|
+
return terminalShellCommandUnit(stripped) || stripped;
|
|
1440
|
+
}
|
|
1441
|
+
|
|
1442
|
+
// The class a receipt carries for the floor: an attested floor class wins;
|
|
1443
|
+
// otherwise, on a bug-fix ledger only, the verification outcome itself is the
|
|
1444
|
+
// evidence — a failed run is repro-before, a pass over a failed surface is
|
|
1445
|
+
// repro-after. `failureCounts` must be the per-surface failure map observed
|
|
1446
|
+
// BEFORE this receipt applies (the live path passes
|
|
1447
|
+
// `ledger.verificationFailureCounts`; the eval fold passes a counter it builds
|
|
1448
|
+
// while walking the receipt window in order).
|
|
1449
|
+
function bugFixReceiptClass(receipt, floor, failureCounts = {}, playbookId = null) {
|
|
1450
|
+
const attested = receipt?.evidence?.class;
|
|
1451
|
+
if (typeof attested === 'string' && BUGFIX_FLOOR_CLASSES.has(attested)) return attested;
|
|
1452
|
+
if (playbookId !== BUGFIX_PLAYBOOK_ID || receipt?.kind !== 'verification' || receipt?.evidence) {
|
|
1453
|
+
return null;
|
|
1454
|
+
}
|
|
1455
|
+
if (receipt.success === true) {
|
|
1456
|
+
const surface = bugFixReceiptSurface(receipt);
|
|
1457
|
+
if (
|
|
1458
|
+
surface
|
|
1459
|
+
&& (floor.reproBefore[surface] === true || Number(failureCounts[surface]) > 0)
|
|
1460
|
+
) {
|
|
1461
|
+
return BUGFIX_REPRO_AFTER;
|
|
1462
|
+
}
|
|
1463
|
+
return null;
|
|
1464
|
+
}
|
|
1465
|
+
// A failed verification on a bug-fix ledger IS the reproduced old failure —
|
|
1466
|
+
// verbatim, no attestation needed.
|
|
1467
|
+
return BUGFIX_REPRO_BEFORE;
|
|
1468
|
+
}
|
|
1469
|
+
|
|
1470
|
+
// Fold one receipt into the floor bank. Idempotent per surface+class: replayed
|
|
1471
|
+
// journal records and the eval-time window fold can run over the same receipt
|
|
1472
|
+
// without double counting. A `repro-after` only banks on a surface with a
|
|
1473
|
+
// confirmed `repro-before` — a passing run on an unrelated surface is not the
|
|
1474
|
+
// fix-proof the playbook demands (same-surface rule, verbatim).
|
|
1475
|
+
function foldBugFixReceipt(floor, receipt, playbookId = null, failureCounts = {}) {
|
|
1476
|
+
const cls = bugFixReceiptClass(receipt, floor, failureCounts, playbookId);
|
|
1477
|
+
if (!cls) return floor;
|
|
1478
|
+
const surface = bugFixReceiptSurface(receipt);
|
|
1479
|
+
// The dedupe stamp identifies one receipt, not one surface: every recorded
|
|
1480
|
+
// repro attempt counts toward the INCONCLUSIVE report, while a replayed
|
|
1481
|
+
// journal record (same receipt) folds idempotently.
|
|
1482
|
+
const stamp = `${cls}:${surface || '-'}:${receipt?.toolUseId ?? receipt?.ts ?? ''}`;
|
|
1483
|
+
if (floor.seen.includes(stamp)) return floor;
|
|
1484
|
+
const next = { ...floor, seen: [...floor.seen, stamp].slice(-BUGFIX_FLOOR_SEEN_CAP) };
|
|
1485
|
+
if (cls === BUGFIX_REPRO_BEFORE) {
|
|
1486
|
+
// Every recorded attempt counts toward the INCONCLUSIVE report; only a
|
|
1487
|
+
// verbatim FAIL confirms the defect reproduced.
|
|
1488
|
+
next.reproAttempts += 1;
|
|
1489
|
+
const result = receipt?.evidence?.result;
|
|
1490
|
+
if (result === 'fail' || (receipt?.success !== true && result !== 'pass' && result !== 'inconclusive')) {
|
|
1491
|
+
next.reproBefore = { ...next.reproBefore, ...(surface ? { [surface]: true } : {}) };
|
|
1492
|
+
}
|
|
1493
|
+
return next;
|
|
1494
|
+
}
|
|
1495
|
+
if (cls === BUGFIX_REPRO_AFTER) {
|
|
1496
|
+
const result = receipt?.evidence?.result;
|
|
1497
|
+
const reproduced = surface !== null && next.reproBefore[surface] === true;
|
|
1498
|
+
if (
|
|
1499
|
+
surface !== null
|
|
1500
|
+
&& reproduced
|
|
1501
|
+
&& (result === 'pass' || (receipt?.success === true && result !== 'fail' && result !== 'inconclusive'))
|
|
1502
|
+
) {
|
|
1503
|
+
next.reproAfter = { ...next.reproAfter, [surface]: true };
|
|
1504
|
+
}
|
|
1505
|
+
return next;
|
|
1506
|
+
}
|
|
1507
|
+
// Root-cause classes: the claim names the mechanism; an unattributed
|
|
1508
|
+
// attestation does not satisfy the floor.
|
|
1509
|
+
const claim = receipt?.evidence?.claim;
|
|
1510
|
+
if (typeof claim === 'string' && claim.trim()) {
|
|
1511
|
+
next.rootCause = String(claim).slice(0, EVIDENCE_FIELD_MAX);
|
|
1512
|
+
}
|
|
1513
|
+
return next;
|
|
1514
|
+
}
|
|
1515
|
+
|
|
1516
|
+
// The eval-side view: the bank is authoritative, and receipts still inside the
|
|
1517
|
+
// window are folded through the same rules so a fixture ledger that never ran
|
|
1518
|
+
// recordExecutionReceipt (no bank) is judged identically. Per-surface failure
|
|
1519
|
+
// counts are rebuilt by walking the window in order.
|
|
1520
|
+
function bugFixFloorStatus(ledger = {}, playbookId = null) {
|
|
1521
|
+
const floor = ledger.bugFixFloor || emptyBugFixFloor();
|
|
1522
|
+
const counts = {};
|
|
1523
|
+
let merged = {
|
|
1524
|
+
...floor,
|
|
1525
|
+
seen: [...floor.seen],
|
|
1526
|
+
reproBefore: { ...floor.reproBefore },
|
|
1527
|
+
reproAfter: { ...floor.reproAfter },
|
|
1528
|
+
};
|
|
1529
|
+
for (const receipt of Array.isArray(ledger.receipts) ? ledger.receipts : []) {
|
|
1530
|
+
merged = foldBugFixReceipt(merged, receipt, playbookId, counts);
|
|
1531
|
+
if (receipt?.kind === 'verification') {
|
|
1532
|
+
const surface = bugFixReceiptSurface(receipt);
|
|
1533
|
+
if (surface) {
|
|
1534
|
+
if (receipt.success === true) delete counts[surface];
|
|
1535
|
+
else counts[surface] = Number(counts[surface] || 0) + 1;
|
|
1536
|
+
}
|
|
1537
|
+
}
|
|
1538
|
+
}
|
|
1539
|
+
return merged;
|
|
1540
|
+
}
|
|
1541
|
+
|
|
1542
|
+
function bugFixFloorMissing(floor) {
|
|
1543
|
+
const missing = [];
|
|
1544
|
+
if (Object.values(floor.reproBefore).every((value) => value !== true)) {
|
|
1545
|
+
missing.push('repro-before');
|
|
1546
|
+
}
|
|
1547
|
+
if (Object.values(floor.reproAfter).every((value) => value !== true)) {
|
|
1548
|
+
missing.push('repro-after');
|
|
1549
|
+
}
|
|
1550
|
+
if (!floor.rootCause) missing.push('root-cause');
|
|
1551
|
+
return missing;
|
|
1552
|
+
}
|
|
1553
|
+
|
|
1554
|
+
// Mint the class onto the receipt (verbatim evidence) and fold it. Runs before
|
|
1555
|
+
// appendReceipt so the derived class is kept inside the receipt window too.
|
|
1556
|
+
function applyBugFixFloorReceipt(ledger, receipt) {
|
|
1557
|
+
if (ledger?.playbookId !== BUGFIX_PLAYBOOK_ID) return receipt;
|
|
1558
|
+
const cls = bugFixReceiptClass(
|
|
1559
|
+
receipt,
|
|
1560
|
+
ledger.bugFixFloor || emptyBugFixFloor(),
|
|
1561
|
+
ledger.verificationFailureCounts || {},
|
|
1562
|
+
ledger.playbookId,
|
|
1563
|
+
);
|
|
1564
|
+
if (cls && !receipt.evidence?.class) {
|
|
1565
|
+
receipt = {
|
|
1566
|
+
...receipt,
|
|
1567
|
+
evidence: {
|
|
1568
|
+
...(receipt.evidence && typeof receipt.evidence === 'object' ? receipt.evidence : {}),
|
|
1569
|
+
class: cls,
|
|
1570
|
+
},
|
|
1571
|
+
};
|
|
1572
|
+
}
|
|
1573
|
+
return receipt;
|
|
1574
|
+
}
|
|
1575
|
+
|
|
1315
1576
|
function applyReceiptToLedger(ledger, receipt, { vibecode = false } = {}) {
|
|
1316
1577
|
if (!receipt || typeof receipt !== 'object' || !RECEIPT_KINDS.has(receipt.kind)) return null;
|
|
1317
1578
|
const next = { ...ledger };
|
|
@@ -1406,6 +1667,51 @@ function applyReceiptToLedger(ledger, receipt, { vibecode = false } = {}) {
|
|
|
1406
1667
|
}
|
|
1407
1668
|
}
|
|
1408
1669
|
}
|
|
1670
|
+
// TASK-008 (BL-010): advisory playbook-finding receipts — the latest status
|
|
1671
|
+
// per receipt class is banked on `playbookFindings` so consumers never have
|
|
1672
|
+
// to re-fold an evictable receipt window. An identical-status repeat is a
|
|
1673
|
+
// no-op (the window is not flooded); a status change replaces the entry.
|
|
1674
|
+
if (receipt.kind === 'playbook-finding') {
|
|
1675
|
+
const cls = typeof receipt.class === 'string' ? receipt.class : null;
|
|
1676
|
+
if (!cls) {
|
|
1677
|
+
next.receipts = appendReceipt(next.receipts, receipt);
|
|
1678
|
+
next.updatedAt = Date.now();
|
|
1679
|
+
return { ledger: next, value: null };
|
|
1680
|
+
}
|
|
1681
|
+
const prior = ledger?.playbookFindings?.[cls];
|
|
1682
|
+
if (prior && prior.status === receipt.status) {
|
|
1683
|
+
return { ledger: next, value: null };
|
|
1684
|
+
}
|
|
1685
|
+
next.playbookFindings = {
|
|
1686
|
+
...(ledger?.playbookFindings && typeof ledger.playbookFindings === 'object'
|
|
1687
|
+
? ledger.playbookFindings
|
|
1688
|
+
: {}),
|
|
1689
|
+
[cls]: {
|
|
1690
|
+
status: receipt.status ?? null,
|
|
1691
|
+
detail: receipt.detail ?? null,
|
|
1692
|
+
file: receipt.file ?? null,
|
|
1693
|
+
ts: receipt.ts ?? Date.now(),
|
|
1694
|
+
},
|
|
1695
|
+
};
|
|
1696
|
+
next.receipts = appendReceipt(next.receipts, receipt);
|
|
1697
|
+
next.updatedAt = Date.now();
|
|
1698
|
+
return { ledger: next, value: null };
|
|
1699
|
+
}
|
|
1700
|
+
// TASK-009 (BL-011): fold the receipt into the bug-fix floor bank before it
|
|
1701
|
+
// enters the (evictable) receipt window. The fold runs on the pre-update
|
|
1702
|
+
// verificationFailureCounts: a pass over a surface whose failure is still
|
|
1703
|
+
// recorded is the repro-after the floor demands. Bug-fix ledgers only —
|
|
1704
|
+
// non-bug-fix ledgers never grow the field.
|
|
1705
|
+
if (ledger?.playbookId === BUGFIX_PLAYBOOK_ID) {
|
|
1706
|
+
receipt = applyBugFixFloorReceipt(ledger, receipt);
|
|
1707
|
+
next.bugFixFloor = foldBugFixReceipt(
|
|
1708
|
+
ledger.bugFixFloor || emptyBugFixFloor(),
|
|
1709
|
+
receipt,
|
|
1710
|
+
ledger.playbookId,
|
|
1711
|
+
ledger.verificationFailureCounts || {},
|
|
1712
|
+
);
|
|
1713
|
+
}
|
|
1714
|
+
|
|
1409
1715
|
next.receipts = appendReceipt(next.receipts, receipt);
|
|
1410
1716
|
next.updatedAt = Date.now();
|
|
1411
1717
|
return { ledger: next, value: null };
|
|
@@ -1490,6 +1796,68 @@ function liveBaseLedger(event, current, payload, routeState, harness, fallbackLe
|
|
|
1490
1796
|
return current || fallbackLedger || freshLedger(payload, null, 'unknown');
|
|
1491
1797
|
}
|
|
1492
1798
|
|
|
1799
|
+
// --- TASK-005 (FR-005, BL-007): outcome.observed terminal emissions ---------
|
|
1800
|
+
// OutcomeRecord = ARCH §Data Contracts: outcomeId, runId, routeId,
|
|
1801
|
+
// routeFingerprint, ledgerKey, verdict ∈ OUTCOME_VERDICTS, evidenceIds[],
|
|
1802
|
+
// cost{latencyMs,rounds}, reviewActions[], enforcement: enforced|advisory,
|
|
1803
|
+
// ts — plus the TASK-003 session-boot `request_key` join id. The emit module
|
|
1804
|
+
// (observability-emit.mjs, TASK-003) owns record construction; the ledger
|
|
1805
|
+
// owns WHEN a terminal event fires.
|
|
1806
|
+
let observabilityEmitModule = null;
|
|
1807
|
+
|
|
1808
|
+
async function loadObservabilityEmit() {
|
|
1809
|
+
if (observabilityEmitModule === null) {
|
|
1810
|
+
try {
|
|
1811
|
+
observabilityEmitModule = await import(
|
|
1812
|
+
new URL('./observability-emit.mjs', import.meta.url).href
|
|
1813
|
+
);
|
|
1814
|
+
} catch {
|
|
1815
|
+
observabilityEmitModule = false;
|
|
1816
|
+
}
|
|
1817
|
+
}
|
|
1818
|
+
return observabilityEmitModule === false ? null : observabilityEmitModule;
|
|
1819
|
+
}
|
|
1820
|
+
|
|
1821
|
+
/**
|
|
1822
|
+
* Fire-and-forget `outcome.observed` emission (FR-005). `outcome` is the
|
|
1823
|
+
* emitter's terminal descriptor ({verdict, source, routeId?, routeFingerprint?,
|
|
1824
|
+
* ledgerKey?, runId?, evidenceIds?, cost?, reviewActions?}); the builder stamps
|
|
1825
|
+
* outcomeId/ts/enforcement/request_key. NEVER throws and returns null on any
|
|
1826
|
+
* failure — a dropped telemetry write must never block a receipt, the Stop
|
|
1827
|
+
* gate, or session end (acceptance: emission is fire-and-forget).
|
|
1828
|
+
*/
|
|
1829
|
+
export async function emitTerminalOutcome({
|
|
1830
|
+
projectRoot,
|
|
1831
|
+
outcome,
|
|
1832
|
+
payload = {},
|
|
1833
|
+
harness = null,
|
|
1834
|
+
engine = null,
|
|
1835
|
+
env = null,
|
|
1836
|
+
deadlineMs,
|
|
1837
|
+
} = {}) {
|
|
1838
|
+
try {
|
|
1839
|
+
const emit = await loadObservabilityEmit();
|
|
1840
|
+
if (!emit || typeof emit.emitOutcomeObserved !== 'function'
|
|
1841
|
+
|| typeof emit.buildOutcomeObservedRecord !== 'function') {
|
|
1842
|
+
return null;
|
|
1843
|
+
}
|
|
1844
|
+
const descriptor = outcome && typeof outcome === 'object' ? outcome : {};
|
|
1845
|
+
const record = emit.buildOutcomeObservedRecord({
|
|
1846
|
+
outcome: { ...descriptor, engine: engine ?? harness ?? descriptor.engine ?? null },
|
|
1847
|
+
projectRoot,
|
|
1848
|
+
payload: payload && typeof payload === 'object' ? payload : {},
|
|
1849
|
+
env: env === null ? process.env : env,
|
|
1850
|
+
});
|
|
1851
|
+
return await emit.emitOutcomeObserved({
|
|
1852
|
+
projectRoot,
|
|
1853
|
+
record,
|
|
1854
|
+
deadlineMs,
|
|
1855
|
+
});
|
|
1856
|
+
} catch {
|
|
1857
|
+
return null;
|
|
1858
|
+
}
|
|
1859
|
+
}
|
|
1860
|
+
|
|
1493
1861
|
/**
|
|
1494
1862
|
* The one locked mutation protocol for the execution ledger (TASK-027). Never mutates
|
|
1495
1863
|
* the ledger without lock ownership; journals the event when the lock cannot be taken.
|
|
@@ -1513,8 +1881,15 @@ export async function recordLedgerEvent(event, {
|
|
|
1513
1881
|
}
|
|
1514
1882
|
const eventId = newEventId();
|
|
1515
1883
|
const target = ledgerPath(projectRoot, payload);
|
|
1884
|
+
// Hoisted so the post-commit outcome emitter sees the same state snapshot
|
|
1885
|
+
// the lock callback evaluated (a fresh read could observe a re-route).
|
|
1886
|
+
let lockedRouteState = routeState || null;
|
|
1887
|
+
// The descriptor the post-commit emitter must fire (lock-local `value` is
|
|
1888
|
+
// nested two levels deep in the lock result — a hoisted ref is clearer).
|
|
1889
|
+
let emittedOutcome = null;
|
|
1516
1890
|
const outcome = await withLedgerLock(target, { signal, deadlineMs }, async () => {
|
|
1517
1891
|
const state = routeState || await readRouteState(projectRoot, payload);
|
|
1892
|
+
lockedRouteState = state || null;
|
|
1518
1893
|
const current = await readExecutionLedger(projectRoot, payload);
|
|
1519
1894
|
let ledger = liveBaseLedger(event, current, payload, state, harness, fallbackLedger);
|
|
1520
1895
|
// Reconcile pending journaled events first — inside this caller's acquired lock and
|
|
@@ -1523,15 +1898,91 @@ export async function recordLedgerEvent(event, {
|
|
|
1523
1898
|
ledger = drained.ledger;
|
|
1524
1899
|
let value;
|
|
1525
1900
|
if (event.type === 'receipt') {
|
|
1901
|
+
// TASK-009 (BL-011): stamp the routed playbook on the ledger so the floor
|
|
1902
|
+
// fold can classify repro receipts at apply time — including journaled
|
|
1903
|
+
// replays, where the ledger's own stamp is the only playbook carrier.
|
|
1904
|
+
// A lost route state (state null) keeps the stamp the request already had.
|
|
1905
|
+
const playbookId = state?.routeSummary?.playbookId || null;
|
|
1906
|
+
if (state && ledger.playbookId !== playbookId) {
|
|
1907
|
+
ledger = { ...ledger, playbookId };
|
|
1908
|
+
}
|
|
1526
1909
|
const applied = applyReceiptToLedger(ledger, event.receipt, { vibecode: event.vibecode === true });
|
|
1527
1910
|
if (!applied) return { committed: false, eventId, reason: 'invalid-receipt' };
|
|
1528
1911
|
ledger = applied.ledger;
|
|
1912
|
+
// TASK-005 (FR-005): completion receipts (write/verification terminal
|
|
1913
|
+
// fields — source receipts carry no outcome signal) mint a descriptor
|
|
1914
|
+
// the CALLER emits: record-execution.mjs for the hook path, the
|
|
1915
|
+
// `--record` CLI for direct invocation. Emitted via the shared seam,
|
|
1916
|
+
// never inside the lock.
|
|
1917
|
+
if (event.receipt?.kind === 'write' || event.receipt?.kind === 'verification') {
|
|
1918
|
+
value = {
|
|
1919
|
+
outcome: {
|
|
1920
|
+
verdict: event.receipt.success === true ? 'VERIFIED' : 'FAILED',
|
|
1921
|
+
source: `receipt-${event.receipt.kind}`,
|
|
1922
|
+
routeId: state?.requestKey ?? ledger?.requestKey ?? null,
|
|
1923
|
+
routeFingerprint: ledger?.routeFingerprint ?? state?.fingerprint ?? null,
|
|
1924
|
+
ledgerKey: `sess-${sessionIdentity(payload)}`,
|
|
1925
|
+
evidenceIds: event.receipt.toolUseId ? [String(event.receipt.toolUseId).slice(0, 160)] : [],
|
|
1926
|
+
cost: { latencyMs: null, rounds: Number.isFinite(ledger?.continuationCount) ? ledger.continuationCount : null },
|
|
1927
|
+
reviewActions: [],
|
|
1928
|
+
// TASK-C85-016 (BL-016): per-pair tracking — nullable, additive.
|
|
1929
|
+
implementer: pairIdentity(event.receipt?.implementer ?? payload.implementer),
|
|
1930
|
+
reviewer: pairIdentity(event.receipt?.reviewer ?? payload.reviewer),
|
|
1931
|
+
reviewVerdict: pairIdentity(event.receipt?.reviewVerdict ?? payload.reviewVerdict),
|
|
1932
|
+
},
|
|
1933
|
+
};
|
|
1934
|
+
}
|
|
1529
1935
|
} else if (event.type === 'continuation') {
|
|
1530
1936
|
ledger = applyContinuationToLedger(ledger, event);
|
|
1531
1937
|
} else if (event.type === 'notified') {
|
|
1532
1938
|
ledger = { ...ledger, notified: true, updatedAt: Date.now() };
|
|
1533
1939
|
} else if (event.type === 'resumable-run') {
|
|
1940
|
+
// TASK-005 (FR-005): a resumable-run pointer is the interrupted-run
|
|
1941
|
+
// terminal event — emit `outcome.observed` verdict INTERRUPTED exactly
|
|
1942
|
+
// ONCE per task. The ledger's `outcomes` dedupe slot (carried across
|
|
1943
|
+
// re-keys by carriedEvidenceLedger) suppresses journal replays and
|
|
1944
|
+
// resume-fold re-emissions; `run` pointers that fail to apply never
|
|
1945
|
+
// mint an outcome.
|
|
1946
|
+
const resumableBefore = ledger?.resumableRun || null;
|
|
1534
1947
|
ledger = applyResumableRunToLedger(ledger, event.run);
|
|
1948
|
+
const resumableApplied = ledger.resumableRun !== resumableBefore;
|
|
1949
|
+
const runKey = typeof event.run?.taskId === 'string' && event.run.taskId
|
|
1950
|
+
? String(event.run.taskId).slice(0, 160)
|
|
1951
|
+
: null;
|
|
1952
|
+
if (resumableApplied && runKey) {
|
|
1953
|
+
const prior = Array.isArray(ledger.outcomes) ? ledger.outcomes : [];
|
|
1954
|
+
if (!prior.some((entry) => entry?.verdict === 'INTERRUPTED' && entry?.runKey === runKey)) {
|
|
1955
|
+
ledger = {
|
|
1956
|
+
...ledger,
|
|
1957
|
+
outcomes: [...prior, { verdict: 'INTERRUPTED', runKey, at: Date.now() }]
|
|
1958
|
+
.slice(-MAX_LEDGER_OUTCOMES),
|
|
1959
|
+
};
|
|
1960
|
+
emittedOutcome = {
|
|
1961
|
+
verdict: 'INTERRUPTED',
|
|
1962
|
+
source: 'resumable-run',
|
|
1963
|
+
runId: `run-${runKey.replace(/[^a-zA-Z0-9._-]/g, '_')}`,
|
|
1964
|
+
routeId: state?.requestKey ?? ledger?.requestKey ?? null,
|
|
1965
|
+
routeFingerprint: ledger?.routeFingerprint ?? state?.fingerprint ?? null,
|
|
1966
|
+
ledgerKey: `sess-${sessionIdentity(payload)}`,
|
|
1967
|
+
evidenceIds: [],
|
|
1968
|
+
cost: { latencyMs: null, rounds: null },
|
|
1969
|
+
reviewActions: [],
|
|
1970
|
+
};
|
|
1971
|
+
value = {
|
|
1972
|
+
outcome: {
|
|
1973
|
+
verdict: 'INTERRUPTED',
|
|
1974
|
+
source: 'resumable-run',
|
|
1975
|
+
runId: `run-${runKey.replace(/[^a-zA-Z0-9._-]/g, '_')}`,
|
|
1976
|
+
routeId: state?.requestKey ?? ledger?.requestKey ?? null,
|
|
1977
|
+
routeFingerprint: ledger?.routeFingerprint ?? state?.fingerprint ?? null,
|
|
1978
|
+
ledgerKey: `sess-${sessionIdentity(payload)}`,
|
|
1979
|
+
evidenceIds: [],
|
|
1980
|
+
cost: { latencyMs: null, rounds: null },
|
|
1981
|
+
reviewActions: [],
|
|
1982
|
+
},
|
|
1983
|
+
};
|
|
1984
|
+
}
|
|
1985
|
+
}
|
|
1535
1986
|
} else if (event.type === 'stop-progress') {
|
|
1536
1987
|
if (!current && drained.applied === 0) {
|
|
1537
1988
|
// A stop with no ledger at all creates nothing (historic noteStopProgress shape).
|
|
@@ -1553,6 +2004,20 @@ export async function recordLedgerEvent(event, {
|
|
|
1553
2004
|
return { committed: true, eventId, value, drained: drained.applied, quarantined: drained.quarantined };
|
|
1554
2005
|
});
|
|
1555
2006
|
if (outcome.ok) {
|
|
2007
|
+
// TASK-005 (FR-005): ledger-owned terminal events emit their outcome
|
|
2008
|
+
// descriptor after the commit — never inside the lock (the emit runs its
|
|
2009
|
+
// own bounded write) and never for the caller-owned receipt path
|
|
2010
|
+
// (record-execution.mjs / --record emit those).
|
|
2011
|
+
if (event.type === 'resumable-run' && emittedOutcome) {
|
|
2012
|
+
try {
|
|
2013
|
+
await emitTerminalOutcome({
|
|
2014
|
+
projectRoot,
|
|
2015
|
+
outcome: emittedOutcome,
|
|
2016
|
+
payload,
|
|
2017
|
+
harness,
|
|
2018
|
+
});
|
|
2019
|
+
} catch { /* telemetry must never block the ledger */ }
|
|
2020
|
+
}
|
|
1556
2021
|
// Sampled bounded dir sweep (BUG-C21-11): runs outside the ledger lock so a
|
|
1557
2022
|
// contended acquisition never pays sweep latency. Advisory — never throws.
|
|
1558
2023
|
// BUG-C23-10: a persistent failure is still surfaced as a bounded degrade.
|
|
@@ -1576,6 +2041,63 @@ export async function recordLedgerEvent(event, {
|
|
|
1576
2041
|
return { rejected: true, eventId, reason: journalResult.reason };
|
|
1577
2042
|
}
|
|
1578
2043
|
|
|
2044
|
+
// TASK-009 (BL-011): an evidence attestation rides a receipt as the bounded
|
|
2045
|
+
// ARCH §Evidence Schema tuple {class, surface, result, claim}. Attestation
|
|
2046
|
+
// surfaces, checked in order: `payload.evidence` (the `--record` CLI and
|
|
2047
|
+
// in-proc callers), `payload.tool_input.evidence` (tools that pass extra
|
|
2048
|
+
// input fields through), and a leading `UKIT_EVIDENCE=class=…;surface=…` env
|
|
2049
|
+
// assignment on a Bash command (the one channel a model can mint through an
|
|
2050
|
+
// ordinary Bash call). Values are bounded strings only — no tool output.
|
|
2051
|
+
function normalizeEvidenceAttestation(value) {
|
|
2052
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) return null;
|
|
2053
|
+
const clean = {};
|
|
2054
|
+
for (const key of JOURNAL_EVIDENCE_FIELDS) {
|
|
2055
|
+
const field = value[key];
|
|
2056
|
+
if (field === undefined || field === null || field === '') continue;
|
|
2057
|
+
const text = String(field).slice(0, EVIDENCE_FIELD_MAX);
|
|
2058
|
+
if (key === 'result' && !EVIDENCE_RESULTS.has(text)) continue;
|
|
2059
|
+
clean[key] = text;
|
|
2060
|
+
}
|
|
2061
|
+
return Object.keys(clean).length > 0 ? clean : null;
|
|
2062
|
+
}
|
|
2063
|
+
|
|
2064
|
+
function parseCommandEvidence(command) {
|
|
2065
|
+
const text = String(command || '').trim();
|
|
2066
|
+
const match = text.match(/^UKIT_EVIDENCE=((?:"[^"]*")|(?:'[^']*')|(?:\S+))/);
|
|
2067
|
+
if (!match) return null;
|
|
2068
|
+
let token = match[1];
|
|
2069
|
+
if (
|
|
2070
|
+
(token.startsWith('"') && token.endsWith('"'))
|
|
2071
|
+
|| (token.startsWith("'") && token.endsWith("'"))
|
|
2072
|
+
) {
|
|
2073
|
+
token = token.slice(1, -1);
|
|
2074
|
+
}
|
|
2075
|
+
const fields = {};
|
|
2076
|
+
for (const part of token.split(';')) {
|
|
2077
|
+
const trimmed = part.trim();
|
|
2078
|
+
if (!trimmed) continue;
|
|
2079
|
+
const eq = trimmed.indexOf('=');
|
|
2080
|
+
if (eq === -1) {
|
|
2081
|
+
if (!('class' in fields)) fields.class = trimmed;
|
|
2082
|
+
continue;
|
|
2083
|
+
}
|
|
2084
|
+
const key = trimmed.slice(0, eq).trim();
|
|
2085
|
+
const fieldValue = trimmed.slice(eq + 1).trim();
|
|
2086
|
+
if (JOURNAL_EVIDENCE_FIELDS.includes(key) && fieldValue) {
|
|
2087
|
+
fields[key] = fieldValue;
|
|
2088
|
+
}
|
|
2089
|
+
}
|
|
2090
|
+
return normalizeEvidenceAttestation(fields);
|
|
2091
|
+
}
|
|
2092
|
+
|
|
2093
|
+
function receiptEvidenceAttestation(payload, toolInput) {
|
|
2094
|
+
return (
|
|
2095
|
+
normalizeEvidenceAttestation(payload?.evidence)
|
|
2096
|
+
|| normalizeEvidenceAttestation(toolInput?.evidence)
|
|
2097
|
+
|| parseCommandEvidence(toolInput?.command)
|
|
2098
|
+
);
|
|
2099
|
+
}
|
|
2100
|
+
|
|
1579
2101
|
// TASK-027: the receipt is classified (kind, file, command, targeted/broad scope) from the
|
|
1580
2102
|
// advisory route state BEFORE locking — that read is read-only — and then applied through
|
|
1581
2103
|
// recordLedgerEvent, which owns the fail-closed lock/journal protocol.
|
|
@@ -1593,6 +2115,7 @@ export async function recordExecutionReceipt({
|
|
|
1593
2115
|
const exitCode = extractExitCode(payload);
|
|
1594
2116
|
const success = !failed && (exitCode === null || exitCode === 0);
|
|
1595
2117
|
const toolInput = payload.tool_input || {};
|
|
2118
|
+
const evidence = receiptEvidenceAttestation(payload, toolInput);
|
|
1596
2119
|
const receipt = {
|
|
1597
2120
|
ts: Date.now(),
|
|
1598
2121
|
toolName,
|
|
@@ -1607,7 +2130,12 @@ export async function recordExecutionReceipt({
|
|
|
1607
2130
|
} else if (toolName === 'Edit' || toolName === 'Write') {
|
|
1608
2131
|
receipt.kind = 'write';
|
|
1609
2132
|
receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
|
|
1610
|
-
} else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
|
|
2133
|
+
} else if (toolName === 'Bash' && (isVerificationCommand(toolInput.command) || evidence)) {
|
|
2134
|
+
// TASK-009 (BL-011): a Bash command carrying an evidence attestation
|
|
2135
|
+
// records as a verification receipt even when the command is not a known
|
|
2136
|
+
// test runner — a repro command, a trace, a one-off check. The typed
|
|
2137
|
+
// verdict stays restricted to real verification commands: attestation
|
|
2138
|
+
// alone must never mint verification evidence.
|
|
1611
2139
|
receipt.kind = 'verification';
|
|
1612
2140
|
receipt.command = String(toolInput.command || '').trim();
|
|
1613
2141
|
// WS-C routed-verification receipt: a command counts as "targeted" when it matches the
|
|
@@ -1622,11 +2150,24 @@ export async function recordExecutionReceipt({
|
|
|
1622
2150
|
// SPEC-typed-verdicts §2.1: every verification receipt mints a typed verdict record
|
|
1623
2151
|
// (kind/evidence/verifier/headSha/baseSha/patchId/ts). Explicit payload.verdict
|
|
1624
2152
|
// fields win so a hook or subagent verifier can attest its own kind and identity.
|
|
1625
|
-
|
|
2153
|
+
if (isVerificationCommand(toolInput.command)) {
|
|
2154
|
+
receipt.verdict = mintVerificationVerdict(receipt, payload, projectRoot);
|
|
2155
|
+
}
|
|
1626
2156
|
} else {
|
|
1627
2157
|
// Untracked tool: no event, no lock, no write — same as the old unlocked early return.
|
|
1628
2158
|
return { rejected: true, eventId: null };
|
|
1629
2159
|
}
|
|
2160
|
+
if (evidence) {
|
|
2161
|
+
receipt.evidence = evidence;
|
|
2162
|
+
}
|
|
2163
|
+
|
|
2164
|
+
// TASK-C85-016 (BL-016): additive per-pair tracking fields — the caller
|
|
2165
|
+
// (hook payload or a future review lane) attests who implemented, who
|
|
2166
|
+
// reviewed, and the verdict. Nullable + bounded; absent → null on the
|
|
2167
|
+
// outcome descriptor so old ledgers stay valid.
|
|
2168
|
+
receipt.implementer = pairIdentity(payload.implementer ?? toolInput.implementer);
|
|
2169
|
+
receipt.reviewer = pairIdentity(payload.reviewer ?? toolInput.reviewer);
|
|
2170
|
+
receipt.reviewVerdict = pairIdentity(payload.reviewVerdict ?? toolInput.reviewVerdict);
|
|
1630
2171
|
|
|
1631
2172
|
return recordLedgerEvent(
|
|
1632
2173
|
{
|
|
@@ -1691,6 +2232,17 @@ function requiredEvidence(state = {}) {
|
|
|
1691
2232
|
if (routeSummary.riskEscalation?.level === 'high' && routeSummary.riskEscalation?.stage !== 'shadow') {
|
|
1692
2233
|
required.push('verification-evidence');
|
|
1693
2234
|
}
|
|
2235
|
+
// TASK-C85-014 (BL-014): a resolved verification-map recipe rides the route
|
|
2236
|
+
// summary (stamped by stop-coordinator's recipe consult). Its
|
|
2237
|
+
// evidenceRequired[] classes join the required set verbatim — a `|` token
|
|
2238
|
+
// is an anyOf list satisfied by ANY alternative receipt class. Absent or
|
|
2239
|
+
// malformed recipe data adds nothing: the gate stays exactly as before.
|
|
2240
|
+
const recipeRequired = routeSummary.verificationRecipe?.evidenceRequired;
|
|
2241
|
+
if (Array.isArray(recipeRequired)) {
|
|
2242
|
+
for (const item of recipeRequired) {
|
|
2243
|
+
if (typeof item === 'string' && item.trim()) required.push(item.trim());
|
|
2244
|
+
}
|
|
2245
|
+
}
|
|
1694
2246
|
return [...new Set(required)];
|
|
1695
2247
|
}
|
|
1696
2248
|
|
|
@@ -1733,12 +2285,83 @@ function evidenceSatisfied(evidence, ledger = {}, state = {}, { cwd } = {}) {
|
|
|
1733
2285
|
}
|
|
1734
2286
|
return ledger.sourceSucceeded === true;
|
|
1735
2287
|
}
|
|
1736
|
-
|
|
2288
|
+
// TASK-C85-014 (BL-014): recipe evidence classes (verification-map.json
|
|
2289
|
+
// artifactClasses.<cls>.evidenceRequired) satisfy by receipt attestation /
|
|
2290
|
+
// playbook-finding bank — see receiptEvidenceSatisfied below.
|
|
2291
|
+
return receiptEvidenceSatisfied(evidence, ledger);
|
|
2292
|
+
}
|
|
2293
|
+
|
|
2294
|
+
// ─── Receipt-class satisfaction (TASK-C85-014 fix round) ────────────────────
|
|
2295
|
+
// Semantics identical to receiptEvidenceSatisfied in
|
|
2296
|
+
// ukit/index/verification-map.mjs (canonical: src/index/verificationMap.js).
|
|
2297
|
+
// A shared leaf is impossible — runtime/ must not static-import index/
|
|
2298
|
+
// (partial installs break: the ledger runs with the reader absent) — so parity
|
|
2299
|
+
// is enforced by tests/consistency/receiptSatisfactionParity.test.js instead.
|
|
2300
|
+
// Any `|` alternative satisfies; the playbook-finding bank is latest-wins (a
|
|
2301
|
+
// later 'missing' finding revokes an earlier 'satisfied' one, and a satisfied
|
|
2302
|
+
// class survives receipt eviction via the bank). A failed attestation is not
|
|
2303
|
+
// proof of the class.
|
|
2304
|
+
export function receiptEvidenceSatisfied(requiredClass, ledger = {}) {
|
|
2305
|
+
const token = String(requiredClass || '').trim();
|
|
2306
|
+
const alternatives = token
|
|
2307
|
+
.split('|')
|
|
2308
|
+
.map((entry) => entry.trim())
|
|
2309
|
+
.filter(Boolean);
|
|
2310
|
+
if (alternatives.length === 0) return false;
|
|
2311
|
+
const matchesClass = (value) => {
|
|
2312
|
+
const cls = typeof value === 'string' ? value.trim() : '';
|
|
2313
|
+
return cls === token || alternatives.includes(cls);
|
|
2314
|
+
};
|
|
2315
|
+
// The banked latest-per-class status governs the receipt window; several
|
|
2316
|
+
// banked keys can match one anyOf token, so the newest entry wins. A
|
|
2317
|
+
// playbook-finding receipt folds into the bank regardless of its success
|
|
2318
|
+
// flag — the receipt scan below mirrors that.
|
|
2319
|
+
let bankedTs = -1;
|
|
2320
|
+
let bankedStatus = null;
|
|
2321
|
+
const bank = ledger?.playbookFindings;
|
|
2322
|
+
if (bank && typeof bank === 'object') {
|
|
2323
|
+
for (const [key, entry] of Object.entries(bank)) {
|
|
2324
|
+
if (!matchesClass(key)) continue;
|
|
2325
|
+
const ts = typeof entry?.ts === 'number' ? entry.ts : 0;
|
|
2326
|
+
if (ts >= bankedTs) {
|
|
2327
|
+
bankedTs = ts;
|
|
2328
|
+
bankedStatus = typeof entry?.status === 'string' ? entry.status : null;
|
|
2329
|
+
}
|
|
2330
|
+
}
|
|
2331
|
+
}
|
|
2332
|
+
const receipts = Array.isArray(ledger?.receipts) ? ledger.receipts : [];
|
|
2333
|
+
let attested = false;
|
|
2334
|
+
let scannedStatus = null;
|
|
2335
|
+
for (const receipt of receipts) {
|
|
2336
|
+
if (!receipt || typeof receipt !== 'object') continue;
|
|
2337
|
+
if (receipt.kind === 'playbook-finding') {
|
|
2338
|
+
if (matchesClass(receipt.class) && typeof receipt.status === 'string') {
|
|
2339
|
+
scannedStatus = receipt.status; // chronological — last write wins
|
|
2340
|
+
}
|
|
2341
|
+
continue;
|
|
2342
|
+
}
|
|
2343
|
+
if (receipt.success === false) continue;
|
|
2344
|
+
const cls = receipt?.evidence?.class;
|
|
2345
|
+
if (typeof cls === 'string' && alternatives.includes(cls.trim())) attested = true;
|
|
2346
|
+
}
|
|
2347
|
+
if (attested) return true;
|
|
2348
|
+
const findingStatus = bankedTs >= 0 ? bankedStatus : scannedStatus;
|
|
2349
|
+
return findingStatus === 'satisfied';
|
|
1737
2350
|
}
|
|
1738
2351
|
|
|
1739
2352
|
function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}, { cwd } = {}) {
|
|
1740
2353
|
let instruction = null;
|
|
1741
|
-
|
|
2354
|
+
// TASK-009 (BL-011): bug-fix floor instructions take precedence — the fix
|
|
2355
|
+
// loop needs the verbatim repro pair and the named mechanism, not more
|
|
2356
|
+
// generic write/verify pushes.
|
|
2357
|
+
if (missingEvidence.includes('repro-before')) {
|
|
2358
|
+
instruction = 'Reproduce the original failure verbatim on the defect surface and record it as the repro-before receipt (failing run). No fix evidence counts before the failure is proven.';
|
|
2359
|
+
} else if (missingEvidence.includes('repro-after')) {
|
|
2360
|
+
instruction = 'Re-run the original repro on the same surface after the fix and record the passing repro-after receipt — a different surface or unrelated green check is not the fix-proof.';
|
|
2361
|
+
} else if (missingEvidence.includes('root-cause')) {
|
|
2362
|
+
instruction = 'Record the root cause: attest the named mechanism (UKIT_EVIDENCE root-cause/instrumentation/behavioral-check with claim) — the rejected hypotheses and the surviving mechanism, one line each.';
|
|
2363
|
+
}
|
|
2364
|
+
if (!instruction && missingEvidence.includes('write-evidence')) {
|
|
1742
2365
|
if (!ledger.sourceSucceeded) {
|
|
1743
2366
|
instruction = 'Pull one bounded indexed source slice, then make the requested Edit/Write in this continuation.';
|
|
1744
2367
|
} else if (ledger.writeAttempted && !ledger.writeSucceeded) {
|
|
@@ -1772,6 +2395,21 @@ function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}, {
|
|
|
1772
2395
|
}
|
|
1773
2396
|
}
|
|
1774
2397
|
}
|
|
2398
|
+
// TASK-C85-014 (BL-014): a missing recipe receipt names the recipe's own
|
|
2399
|
+
// check and its capability-negotiated fallback receipt spec — never a
|
|
2400
|
+
// fabricated command.
|
|
2401
|
+
if (!instruction) {
|
|
2402
|
+
const recipe = routeSummary?.verificationRecipe;
|
|
2403
|
+
const recipeMissing = missingEvidence.filter((item) =>
|
|
2404
|
+
Array.isArray(recipe?.evidenceRequired) && recipe.evidenceRequired.includes(item));
|
|
2405
|
+
if (recipeMissing.length > 0) {
|
|
2406
|
+
instruction = [
|
|
2407
|
+
`The verification-map recipe for artifact class '${recipe.artifactClass || 'unknown'}' requires receipt class ${recipeMissing.join(', ')}.`,
|
|
2408
|
+
recipe.check ? `Check: ${recipe.check}` : null,
|
|
2409
|
+
recipe.fallback ? `Fallback receipt: ${recipe.fallback}` : null,
|
|
2410
|
+
].filter(Boolean).join(' ');
|
|
2411
|
+
}
|
|
2412
|
+
}
|
|
1775
2413
|
// WS-C: the target hint travels with the reason whenever impact evidence is missing and
|
|
1776
2414
|
// the route names expected files — it must survive branch precedence above.
|
|
1777
2415
|
if (missingEvidence.includes('impact-evidence')) {
|
|
@@ -1858,6 +2496,44 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1858
2496
|
|| (ledger.promptKey && evidencePromptKey(state) === ledger.promptKey);
|
|
1859
2497
|
const effectiveLedger = sameRequest ? ledger : {};
|
|
1860
2498
|
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, state, { cwd }));
|
|
2499
|
+
// TASK-009 (BL-011): the bug-fix floor. On a `playbookId: bug-fix` route a
|
|
2500
|
+
// green verdict requires verbatim repro-before (fail) + repro-after (pass)
|
|
2501
|
+
// receipts on the same surface plus a named root cause in the exec-ledger —
|
|
2502
|
+
// symptom-patching without receipts can never release green. Scoped to
|
|
2503
|
+
// bug-fix only: every other playbook and plain-mode route keeps the generic
|
|
2504
|
+
// completion behavior (false-block = 0 on the P2 gate).
|
|
2505
|
+
const playbookId = routeSummary.playbookId || null;
|
|
2506
|
+
if (playbookId === BUGFIX_PLAYBOOK_ID) {
|
|
2507
|
+
const floor = bugFixFloorStatus(effectiveLedger, playbookId);
|
|
2508
|
+
missingEvidence.push(...bugFixFloorMissing(floor));
|
|
2509
|
+
// An unreproducible defect ends INCONCLUSIVE, never a fabricated pass and
|
|
2510
|
+
// never an unbounded recovery loop: the defect could not be reproduced in
|
|
2511
|
+
// the recorded attempts, so the fix floor is unreachable. The reason names
|
|
2512
|
+
// the attempt count and the runtime-forensics follow-on (playbook step 5).
|
|
2513
|
+
if (
|
|
2514
|
+
floor.reproAttempts > 0
|
|
2515
|
+
&& Object.values(floor.reproBefore).every((value) => value !== true)
|
|
2516
|
+
&& Object.values(floor.reproAfter).every((value) => value !== true)
|
|
2517
|
+
) {
|
|
2518
|
+
return {
|
|
2519
|
+
continue: false,
|
|
2520
|
+
notify: true,
|
|
2521
|
+
inconclusive: true,
|
|
2522
|
+
outcome: 'inconclusive',
|
|
2523
|
+
missingEvidence,
|
|
2524
|
+
reason: `UKit completion gate: bug-fix INCONCLUSIVE — the defect did not reproduce in ${floor.reproAttempts} recorded repro attempt(s), so the fix floor (verbatim repro-before fail + repro-after pass on the same surface) is unreachable. Report verified-vs-not verbatim, the attempts made, and the runtime-forensics follow-on lane (live symptom). Never claim a fix without the repro pair.`,
|
|
2525
|
+
};
|
|
2526
|
+
}
|
|
2527
|
+
if (missingEvidence.length === 0) {
|
|
2528
|
+
return {
|
|
2529
|
+
continue: false,
|
|
2530
|
+
notify: false,
|
|
2531
|
+
complete: true,
|
|
2532
|
+
outcome: 'verified',
|
|
2533
|
+
missingEvidence: [],
|
|
2534
|
+
};
|
|
2535
|
+
}
|
|
2536
|
+
}
|
|
1861
2537
|
if (missingEvidence.length === 0) {
|
|
1862
2538
|
// Silent success: the route is present and every required evidence is satisfied. Marked
|
|
1863
2539
|
// `complete` so the CLI dispatch recognizes it BEFORE the loud final else — otherwise a
|
|
@@ -1873,15 +2549,26 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1873
2549
|
// recovery gate — only a no-mutation, recommend-only investigation that has not already
|
|
1874
2550
|
// observed a failed verification gets this release valve. A route that named an actionable
|
|
1875
2551
|
// command, or a failing check, has concrete unfinished work and must keep recovering.
|
|
2552
|
+
// TASK-C85-019 fix B: the release criterion is the find-cause contract's
|
|
2553
|
+
// completionRule ('never-claim-fixed-without-write-and-verification' — a FIX
|
|
2554
|
+
// claim needs evidence, a mutation is never mandatory), NOT the verification
|
|
2555
|
+
// policyMode. policyMode derives from the verification recommendation, so any
|
|
2556
|
+
// route that resolved targeted commands lands on auto-run-* and the old
|
|
2557
|
+
// recommend-only check never fired in a real project. An absent rule keeps
|
|
2558
|
+
// the pre-C85-019 states and the canonical find-cause contract on the valve.
|
|
2559
|
+
const findCauseNoMutationContract = mode === 'find-cause'
|
|
2560
|
+
&& (routeSummary?.executionContract?.completionRule == null
|
|
2561
|
+
|| routeSummary.executionContract.completionRule === 'never-claim-fixed-without-write-and-verification');
|
|
1876
2562
|
if (
|
|
1877
|
-
|
|
1878
|
-
&&
|
|
2563
|
+
findCauseNoMutationContract
|
|
2564
|
+
&& playbookId !== BUGFIX_PLAYBOOK_ID
|
|
1879
2565
|
&& !effectiveLedger.writeAttempted
|
|
1880
2566
|
&& !effectiveLedger.verificationFailed
|
|
1881
2567
|
) {
|
|
1882
2568
|
return {
|
|
1883
2569
|
continue: false,
|
|
1884
2570
|
notify: true,
|
|
2571
|
+
gateRelease: 'find-cause-no-mutation',
|
|
1885
2572
|
missingEvidence,
|
|
1886
2573
|
reason: 'UKit investigation ended without a mutation. A clean audit is valid; report whether no actionable defect was found or a concrete blocker remains. Do not claim a bug was fixed without write and verification evidence.',
|
|
1887
2574
|
};
|
|
@@ -1904,6 +2591,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1904
2591
|
return {
|
|
1905
2592
|
continue: false,
|
|
1906
2593
|
notify: true,
|
|
2594
|
+
gateRelease: 'map-impact-no-mutation',
|
|
1907
2595
|
missingEvidence,
|
|
1908
2596
|
reason: 'UKit impact analysis ended without a mutation. A completed impact map with no warranted change is valid; report the findings and whether a follow-up edit is needed. Do not claim any fix without write and verification evidence.',
|
|
1909
2597
|
};
|
|
@@ -1915,6 +2603,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1915
2603
|
continue: false,
|
|
1916
2604
|
notify: true,
|
|
1917
2605
|
missingEvidence,
|
|
2606
|
+
...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'inconclusive' } : {}),
|
|
1918
2607
|
reason: `UKit completion gate: missing ${missingEvidence.join(', ')}. This mode does not auto-continue; tell the user what is unfinished.`,
|
|
1919
2608
|
};
|
|
1920
2609
|
}
|
|
@@ -1937,6 +2626,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1937
2626
|
capped: true,
|
|
1938
2627
|
notify: true,
|
|
1939
2628
|
missingEvidence,
|
|
2629
|
+
...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'inconclusive' } : {}),
|
|
1940
2630
|
reason: `UKit continuation cap reached with missing evidence: ${missingEvidence.join(', ')}.`,
|
|
1941
2631
|
};
|
|
1942
2632
|
}
|
|
@@ -1946,6 +2636,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1946
2636
|
notify: true,
|
|
1947
2637
|
capped: true,
|
|
1948
2638
|
missingEvidence,
|
|
2639
|
+
...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'blocked' } : {}),
|
|
1949
2640
|
reason: `UKit stopping with unfinished work: ${missingEvidence.join(', ')}. Tell the user what is unfinished and stop; do not continue further.`,
|
|
1950
2641
|
};
|
|
1951
2642
|
}
|
|
@@ -1964,6 +2655,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1964
2655
|
notify: true,
|
|
1965
2656
|
noProgressCount,
|
|
1966
2657
|
missingEvidence,
|
|
2658
|
+
...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'inconclusive' } : {}),
|
|
1967
2659
|
reason: `UKit vibecode liveness breaker: ${noProgressCount} continuations produced no new verifiable progress (missing evidence: ${missingEvidence.join(', ')}). Stop and tell the user what is unfinished; do not keep continuing without new evidence.`,
|
|
1968
2660
|
};
|
|
1969
2661
|
}
|
|
@@ -1974,6 +2666,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1974
2666
|
capped: true,
|
|
1975
2667
|
noProgressCount,
|
|
1976
2668
|
missingEvidence,
|
|
2669
|
+
...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'blocked' } : {}),
|
|
1977
2670
|
reason: `UKit stopping with unfinished work: ${noProgressCount} continuations produced no new verifiable progress (${missingEvidence.join(', ')}). Tell the user what is unfinished and stop; do not continue further.`,
|
|
1978
2671
|
};
|
|
1979
2672
|
}
|
|
@@ -1984,6 +2677,7 @@ export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
|
1984
2677
|
continue: true,
|
|
1985
2678
|
missingEvidence,
|
|
1986
2679
|
noProgressCount,
|
|
2680
|
+
...(playbookId === BUGFIX_PLAYBOOK_ID ? { outcome: 'blocked' } : {}),
|
|
1987
2681
|
reason: [
|
|
1988
2682
|
`UKit completion gate: missing ${missingEvidence.join(', ')}.`,
|
|
1989
2683
|
instruction,
|
|
@@ -2125,6 +2819,71 @@ async function runEvaluateStop() {
|
|
|
2125
2819
|
|
|
2126
2820
|
const result = evaluateCompletion({ state, ledger, cwd: projectRoot });
|
|
2127
2821
|
|
|
2822
|
+
// TASK-C85-019 fix C: a mutating-mode route released by a no-mutation valve
|
|
2823
|
+
// while evidence is missing is the anomaly signal for misrouted advisory
|
|
2824
|
+
// turns — measure it. One bounded gate-release row lands in decisions.tsv
|
|
2825
|
+
// (mode + missing evidence + prompt hash, never prompt text), and
|
|
2826
|
+
// UKIT_GATE_DEBUG=1 mirrors the same signal on stderr.
|
|
2827
|
+
if (result.gateRelease) {
|
|
2828
|
+
const gateMode = state?.routeSummary?.executionMode
|
|
2829
|
+
|| state?.routeSummary?.approachSelector?.executionMode
|
|
2830
|
+
|| 'unknown';
|
|
2831
|
+
try {
|
|
2832
|
+
await appendDecisionReceipt(projectRoot, {
|
|
2833
|
+
kind: 'gate-release',
|
|
2834
|
+
boundary: 'completion-gate',
|
|
2835
|
+
stage: 'no-mutation',
|
|
2836
|
+
outcomeClass: gateMode,
|
|
2837
|
+
checkpoint: evidencePromptKey(state) ?? 'no-prompt-hash',
|
|
2838
|
+
decisionKeys: Array.isArray(result.missingEvidence) ? result.missingEvidence : [],
|
|
2839
|
+
agreement: 'released',
|
|
2840
|
+
fallbackCode: result.gateRelease,
|
|
2841
|
+
});
|
|
2842
|
+
} catch { /* advisory */ }
|
|
2843
|
+
if (process.env.UKIT_GATE_DEBUG === '1') {
|
|
2844
|
+
process.stderr.write(
|
|
2845
|
+
`[ukit-completion] gate-release mode=${gateMode} valve=${result.gateRelease} missing=${(result.missingEvidence || []).join(',') || 'none'} prompt=${evidencePromptKey(state) || 'none'}\n`,
|
|
2846
|
+
);
|
|
2847
|
+
}
|
|
2848
|
+
}
|
|
2849
|
+
|
|
2850
|
+
// TASK-005 (FR-005): the completion evaluator's terminal verdict emits one
|
|
2851
|
+
// `outcome.observed` record per evaluated stop. The verdict maps the gate's
|
|
2852
|
+
// own decision shape — a still-blocked stop is BLOCKED, a capped/unfinished
|
|
2853
|
+
// release FAILED, a notice-only or undecidable release INCONCLUSIVE, and a
|
|
2854
|
+
// satisfied route VERIFIED. Fire-and-forget: awaited inside this child's
|
|
2855
|
+
// own deadline so the record is durable before exit, but a failed emit
|
|
2856
|
+
// never changes the decision below.
|
|
2857
|
+
{
|
|
2858
|
+
// continue → still blocked; satisfied → VERIFIED; an explicit inconclusive
|
|
2859
|
+
// release (the bug-fix unreproducible path) → INCONCLUSIVE; a
|
|
2860
|
+
// capped/missing-evidence release → FAILED; every other release is
|
|
2861
|
+
// INCONCLUSIVE (a notice-only or undecidable end is an honest non-verdict).
|
|
2862
|
+
const stopVerdict = result.continue === true ? 'BLOCKED'
|
|
2863
|
+
: result.complete === true ? 'VERIFIED'
|
|
2864
|
+
: result.inconclusive === true ? 'INCONCLUSIVE'
|
|
2865
|
+
: result.capped === true || (Array.isArray(result.missingEvidence) && result.missingEvidence.length > 0) ? 'FAILED'
|
|
2866
|
+
: 'INCONCLUSIVE';
|
|
2867
|
+
try {
|
|
2868
|
+
await emitTerminalOutcome({
|
|
2869
|
+
projectRoot,
|
|
2870
|
+
outcome: {
|
|
2871
|
+
verdict: stopVerdict,
|
|
2872
|
+
source: 'stop-evaluate',
|
|
2873
|
+
routeId: state?.requestKey ?? ledger?.requestKey ?? null,
|
|
2874
|
+
routeFingerprint: ledger?.routeFingerprint ?? state?.fingerprint ?? null,
|
|
2875
|
+
ledgerKey: `sess-${sessionIdentity(payload)}`,
|
|
2876
|
+
evidenceIds: [],
|
|
2877
|
+
cost: { latencyMs: null, rounds: Number.isFinite(ledger?.continuationCount) ? ledger.continuationCount : null },
|
|
2878
|
+
reviewActions: [],
|
|
2879
|
+
},
|
|
2880
|
+
payload,
|
|
2881
|
+
harness: process.env.UKIT_HARNESS || null,
|
|
2882
|
+
deadlineMs: 200,
|
|
2883
|
+
});
|
|
2884
|
+
} catch { /* advisory */ }
|
|
2885
|
+
}
|
|
2886
|
+
|
|
2128
2887
|
// A reentrant Stop (Claude Code re-fires Stop after this hook already blocked once) is
|
|
2129
2888
|
// gated exactly like any other Stop: while evidence is still missing it blocks again with
|
|
2130
2889
|
// the actionable reason instead of silently releasing the recovery turn. Loop termination
|
|
@@ -2162,12 +2921,24 @@ async function main() {
|
|
|
2162
2921
|
const payload = JSON.parse((await readStdin()) || '{}');
|
|
2163
2922
|
const projectRoot = process.env.CLAUDE_PROJECT_DIR || payload.cwd || process.cwd();
|
|
2164
2923
|
if (process.argv.includes('--record')) {
|
|
2165
|
-
await recordExecutionReceipt({
|
|
2924
|
+
const receiptResult = await recordExecutionReceipt({
|
|
2166
2925
|
projectRoot,
|
|
2167
2926
|
payload,
|
|
2168
2927
|
toolName: payload.tool_name,
|
|
2169
2928
|
harness: process.env.UKIT_HARNESS || 'claude-code',
|
|
2170
2929
|
});
|
|
2930
|
+
// TASK-005 (FR-005): emit the committed completion receipt's outcome —
|
|
2931
|
+
// same terminal contract record-execution.mjs performs for hook callers.
|
|
2932
|
+
if (receiptResult?.value?.outcome) {
|
|
2933
|
+
try {
|
|
2934
|
+
await emitTerminalOutcome({
|
|
2935
|
+
projectRoot,
|
|
2936
|
+
outcome: receiptResult.value.outcome,
|
|
2937
|
+
payload,
|
|
2938
|
+
harness: process.env.UKIT_HARNESS || 'claude-code',
|
|
2939
|
+
});
|
|
2940
|
+
} catch { /* advisory: never block the receipt path */ }
|
|
2941
|
+
}
|
|
2171
2942
|
}
|
|
2172
2943
|
}
|
|
2173
2944
|
|
|
@@ -2234,7 +3005,14 @@ export async function appendDecision(projectRoot, {
|
|
|
2234
3005
|
// 'shadow' (advisory run, deterministic policy stayed authoritative),
|
|
2235
3006
|
// 'applied' (canary/default accepted answer), 'fallback' (adapter failed and
|
|
2236
3007
|
// the deterministic baseline stayed in force).
|
|
2237
|
-
|
|
3008
|
+
// TASK-C85-015 (BL-015): 'review-policy' receipts carry the review decision
|
|
3009
|
+
// evaluator's advisory verdicts (action/namedCheck/signalsHit/round per SPEC §7)
|
|
3010
|
+
// — additive kind; the shadow/applied/fallback semantics are unchanged.
|
|
3011
|
+
// TASK-C85-019 fix C: 'gate-release' receipts carry the completion gate's
|
|
3012
|
+
// no-mutation valve releases — a mutating-mode route let go while evidence is
|
|
3013
|
+
// missing is the anomaly signal for misrouted advisory turns, made measurable
|
|
3014
|
+
// in decisions.tsv (mode + missingEvidence + prompt hash; never prompt text).
|
|
3015
|
+
const DECISION_RECEIPT_KINDS = new Set(['shadow', 'applied', 'fallback', 'review-policy', 'gate-release']);
|
|
2238
3016
|
const DECISION_RECEIPT_CELL_MAX = 160;
|
|
2239
3017
|
|
|
2240
3018
|
function receiptCell(value) {
|