@ngockhoale/ukit 3.3.2 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/CHANGELOG.md +44 -0
  2. package/manifests/engineConformance.yaml +17 -1
  3. package/manifests/hostCapabilities.yaml +68 -1
  4. package/manifests/platform.full.yaml +138 -0
  5. package/manifests/platform.user.yaml +255 -3
  6. package/package.json +1 -1
  7. package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
  8. package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
  9. package/scripts/probe/codex-capability-probe.mjs +169 -0
  10. package/src/cli/commands/doctor.js +168 -0
  11. package/src/cli/commands/indexTools.js +7 -0
  12. package/src/cli/commands/metrics.js +66 -2
  13. package/src/cli/commands/playbook.js +4 -4
  14. package/src/cli/commands/vm.js +49 -8
  15. package/src/core/agentRuntime/adapters.js +328 -27
  16. package/src/core/agentRuntime/artifacts.js +89 -0
  17. package/src/core/agentRuntime/context.js +345 -1
  18. package/src/core/agentRuntime/contract.js +296 -0
  19. package/src/core/agentRuntime/eventStore.js +176 -0
  20. package/src/core/agentRuntime/shadowRun.js +481 -5
  21. package/src/core/agentRuntime/telemetry.js +121 -0
  22. package/src/core/observability/emit/lifecycle.js +68 -1
  23. package/src/core/observability/emit/sessionBoot.js +393 -0
  24. package/src/core/observability/privacy/allowlist.js +10 -1
  25. package/src/core/observability/schema/registry.js +10 -0
  26. package/src/core/runtimeConfig.js +133 -0
  27. package/src/core/userPlaybooks.js +18 -3
  28. package/src/decision/registry.js +19 -0
  29. package/src/diagnostics/feedbackEvents.js +7 -4
  30. package/src/diagnostics/routeOutcomes.js +51 -6
  31. package/src/diagnostics/skillAccuracy.js +43 -3
  32. package/src/index/crossCheckMatrix.js +412 -0
  33. package/src/index/fixLoopEscalation.js +453 -0
  34. package/src/index/playbookRegistry.js +691 -0
  35. package/src/index/reviewPolicy.js +368 -0
  36. package/src/index/routeResolver.js +915 -0
  37. package/src/index/sessionHistoryExtractor.js +359 -0
  38. package/src/index/taskRouting.js +764 -581
  39. package/src/index/tierSelection.js +308 -0
  40. package/src/index/verificationMap.js +404 -0
  41. package/template_project/.claude/hooks/observability-emit.mjs +14 -0
  42. package/template_project/.claude/hooks/record-execution.mjs +19 -1
  43. package/template_project/.claude/hooks/skill-router.sh +691 -25
  44. package/template_project/.claude/hooks/verification-guard.sh +230 -1
  45. package/template_project/.claude/settings.json +2 -2
  46. package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
  47. package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
  48. package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
  49. package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
  50. package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
  51. package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
  52. package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
  53. package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
  54. package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
  55. package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
  56. package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
  57. package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
  58. package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
  59. package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
  60. package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
  61. package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
  62. package/template_project/.codex/README.md +8 -0
  63. package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
  64. package/template_project/ukit/README.md +1 -1
  65. package/template_project/ukit/storage/config.json +20 -0
  66. package/template_user/playbooks/architecture-decision.md +28 -0
  67. package/template_user/playbooks/autonomous-run.md +43 -0
  68. package/template_user/playbooks/autopilot-full.md +59 -0
  69. package/template_user/playbooks/autopilot-stack.md +54 -0
  70. package/template_user/playbooks/babysit.md +39 -0
  71. package/template_user/playbooks/bug-fix.md +3 -1
  72. package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
  73. package/template_user/playbooks/hillclimb.md +44 -0
  74. package/template_user/playbooks/investigation.md +21 -0
  75. package/template_user/playbooks/migration.md +21 -0
  76. package/template_user/playbooks/open-pr.md +48 -0
  77. package/template_user/playbooks/orchestrate.md +45 -0
  78. package/template_user/playbooks/performance.md +33 -0
  79. package/template_user/playbooks/prototype.md +28 -0
  80. package/template_user/playbooks/refactor.md +19 -0
  81. package/template_user/playbooks/release.md +28 -0
  82. package/template_user/playbooks/runtime-forensics.md +23 -0
  83. package/template_user/playbooks/session-pickup.md +31 -0
  84. package/template_user/playbooks/shipping.md +53 -0
  85. package/template_user/playbooks/skill-evaluation.md +48 -0
  86. package/template_user/playbooks/small-feature.md +20 -0
  87. package/template_user/playbooks/verification-map.json +153 -0
  88. package/template_user/playbooks/verification.md +22 -0
  89. package/template_user/playbooks/worktree-cleanup.md +37 -0
@@ -100,8 +100,16 @@ async function buildResumableRunLines(projectRoot, state, config) {
100
100
  const record = resumed.record;
101
101
  if (!record) return [];
102
102
  const lines = [];
103
+ // BL-024 (session-pickup): the state-verify receipt — record-vs-worktree
104
+ // fingerprint agreement (or the divergent keys) announced before any
105
+ // resumed plan lines, so a cold pickup never trusts the record silently.
103
106
  if (resumed.status === 'stale') {
104
- lines.push('- Resumed run: stale — source/index changed since persist; dependent plans were invalidated, completed receipts kept.');
107
+ const keys = Array.isArray(resumed.changed) && resumed.changed.length > 0
108
+ ? resumed.changed.join(', ')
109
+ : 'source/index';
110
+ lines.push(`- Resumed state verify: stale — recorded fingerprints diverge (${keys}); dependent plans were invalidated, completed receipts kept.`);
111
+ } else {
112
+ lines.push('- Resumed state verify: fresh — recorded fingerprints match the live index/config.');
105
113
  }
106
114
  const phase = String(record.workflow?.phase ?? '').trim();
107
115
  if (phase) lines.push(`- Resumed run phase: ${phase}`);
@@ -445,7 +445,10 @@ function applyInvalidationCodes(record, codes) {
445
445
  * Read the C10 record for a task. Never throws on corrupt state.
446
446
  *
447
447
  * @returns {Promise<{status:'fresh'|'stale'|'invalid'|'absent', record:object|null,
448
- * invalidated?:string[], warnings:string[]}>}
448
+ * invalidated?:string[], changed?:string[], warnings:string[]}>}
449
+ * `changed` lists the source/index/config snapshot keys that diverged from the
450
+ * live fingerprint — the state-verify evidence the session-pickup playbook
451
+ * consumes (BL-024). Only present once the fingerprint check ran.
449
452
  */
450
453
  export async function readResumableRun(projectRoot, {
451
454
  taskId,
@@ -500,11 +503,93 @@ export async function readResumableRun(projectRoot, {
500
503
  status: 'stale',
501
504
  record,
502
505
  invalidated,
506
+ changed,
503
507
  warnings: [`source/index fingerprint changed (${changed.join(', ')}); dependent plans invalidated`],
504
508
  };
505
509
  }
506
510
 
507
- return { status: 'fresh', record: parsed, warnings: [] };
511
+ return { status: 'fresh', record: parsed, changed, warnings: [] };
512
+ }
513
+
514
+ // --- suspend (TASK-C86-024, BL-024, CENSUS PB-U20) ----------------------------
515
+
516
+ /**
517
+ * The suspend half of the session-pickup contract: checkpoint progress into a
518
+ * record the pickup `read` consumes. Same merge semantics as route finalize —
519
+ * the just-completed phase lands in workflow.completedBlocks, plans and
520
+ * completed receipts carry forward, and a fresh live fingerprint is stamped so
521
+ * the next resume starts clean. An omitted taskBoundary preserves the existing
522
+ * record's boundary; an explicit new one starts a fresh record (a different
523
+ * logical task). Stage 'off' stays fail-closed — suspending writes nothing.
524
+ */
525
+ export async function suspendResumableRun(projectRoot, {
526
+ taskId,
527
+ taskBoundary = null,
528
+ phase = 'suspended',
529
+ nextAction = null,
530
+ completedBlocks = [],
531
+ signal,
532
+ deadlineMs = LOCK_MAX_SLICE_MS,
533
+ } = {}) {
534
+ const boundaryProvided = typeof taskBoundary === 'string' && taskBoundary.length > 0;
535
+ const resumed = boundaryProvided
536
+ ? await readResumableRun(projectRoot, { taskId, taskBoundary })
537
+ : await readResumableRun(projectRoot, { taskId });
538
+ const previous = resumed.status === 'fresh' || resumed.status === 'stale'
539
+ ? resumed.record
540
+ : null;
541
+ const boundary = boundaryProvided
542
+ ? taskBoundary
543
+ : (previous?.taskBoundary ?? 'suspended');
544
+
545
+ const carried = [...(previous?.workflow?.completedBlocks ?? [])];
546
+ const previousPhase = previous?.workflow?.phase;
547
+ if (previousPhase && previousPhase !== phase && !carried.includes(previousPhase)) {
548
+ carried.push(previousPhase);
549
+ }
550
+ const checkpoint = Array.isArray(completedBlocks) ? completedBlocks : [];
551
+ const merged = [...new Set([...carried, ...checkpoint])];
552
+
553
+ const fingerprint = await resumableRunSourceFingerprint(projectRoot);
554
+ const record = {
555
+ schemaVersion: RESUMABLE_RUN_SCHEMA_VERSION,
556
+ taskId,
557
+ taskBoundary: boundary,
558
+ route: previous?.route ?? {
559
+ routeVersion: 'suspend-cli',
560
+ mode: 'suspend',
561
+ rigor: 'unresolved',
562
+ contractVersion: 'suspend-cli',
563
+ },
564
+ budget: previous?.budget ?? { policyVersion: 'suspend-cli', consumed: 0, remaining: 0 },
565
+ workflow: {
566
+ workflowId: previous?.workflow?.workflowId ?? 'session-pickup',
567
+ workflowVersion: previous?.workflow?.workflowVersion ?? 'v1',
568
+ phase: String(phase ?? 'suspended'),
569
+ completedBlocks: merged,
570
+ },
571
+ invariants: previous?.invariants ?? [],
572
+ hypotheses: previous?.hypotheses ?? [],
573
+ decisions: previous?.decisions ?? [],
574
+ evidenceRefs: previous?.evidenceRefs ?? [],
575
+ escalationHistory: previous?.escalationHistory ?? [],
576
+ unresolvedFailures: previous?.unresolvedFailures ?? [],
577
+ nextAction: typeof nextAction === 'string' && nextAction !== ''
578
+ ? nextAction.slice(0, MAX_STRING_LENGTH)
579
+ : null,
580
+ sourceSnapshot: { ...(previous?.sourceSnapshot ?? {}), ...fingerprint },
581
+ };
582
+
583
+ const write = await writeResumableRun(projectRoot, record, { signal, deadlineMs });
584
+ return {
585
+ ok: write.ok === true,
586
+ status: write.ok ? 'written' : (write.journaled ? 'journaled' : 'skipped'),
587
+ reason: write.ok || write.journaled ? null : (write.reason ?? 'unknown'),
588
+ errors: write.errors,
589
+ journaled: write.journaled === true,
590
+ path: write.ok ? write.path : resumableRunPath(projectRoot, taskId),
591
+ record,
592
+ };
508
593
  }
509
594
 
510
595
  /**
@@ -584,15 +669,74 @@ export async function resumableRunSourceFingerprint(projectRoot, { indexGenerate
584
669
 
585
670
  // --- CLI (diagnostics) ---------------------------------------------------------
586
671
 
672
+ function parseCliFlags(argv) {
673
+ const flags = {};
674
+ for (let i = 0; i < argv.length; i += 1) {
675
+ const match = /^--([\w-]+)$/.exec(argv[i]);
676
+ if (match) {
677
+ flags[match[1]] = argv[i + 1] !== undefined ? argv[i + 1] : '';
678
+ i += 1;
679
+ }
680
+ }
681
+ return flags;
682
+ }
683
+
587
684
  async function main() {
588
- const [command, rootArg, taskId] = process.argv.slice(2);
685
+ const [command, rootArg, taskId, ...rest] = process.argv.slice(2);
589
686
  const projectRoot = path.resolve(rootArg || process.cwd());
687
+ const flags = parseCliFlags(rest);
590
688
  if (command === 'read' && taskId) {
591
- const result = await readResumableRun(projectRoot, { taskId });
689
+ // BL-024 (session-pickup): the pickup path's state-verify receipt — compare
690
+ // the recorded fingerprints against the live index/config so a stale
691
+ // record reports the divergence instead of being trusted silently.
692
+ const sourceFingerprint = await resumableRunSourceFingerprint(projectRoot);
693
+ const result = await readResumableRun(projectRoot, {
694
+ taskId,
695
+ taskBoundary: typeof flags.boundary === 'string' && flags.boundary !== ''
696
+ ? flags.boundary
697
+ : undefined,
698
+ sourceFingerprint,
699
+ });
700
+ const output = {
701
+ ...result,
702
+ stateVerify: {
703
+ result: result.status === 'fresh' ? 'pass' : 'fail',
704
+ changed: result.changed ?? [],
705
+ cursor: result.record
706
+ ? {
707
+ phase: result.record.workflow?.phase ?? null,
708
+ completedBlocks: result.record.workflow?.completedBlocks ?? [],
709
+ nextAction: result.record.nextAction ?? null,
710
+ }
711
+ : null,
712
+ },
713
+ };
714
+ process.stdout.write(`${JSON.stringify(output, null, 2)}\n`);
715
+ return;
716
+ }
717
+ if (command === 'suspend' && taskId) {
718
+ // BL-024 (PB-U20 suspend half): checkpoint progress into a record the
719
+ // pickup `read` above consumes — persist + report, never write unlocked.
720
+ const result = await suspendResumableRun(projectRoot, {
721
+ taskId,
722
+ taskBoundary: typeof flags.boundary === 'string' && flags.boundary !== ''
723
+ ? flags.boundary
724
+ : null,
725
+ phase: flags.phase,
726
+ nextAction: typeof flags.next === 'string' && flags.next !== '' ? flags.next : null,
727
+ completedBlocks: typeof flags.completed === 'string' && flags.completed !== ''
728
+ ? flags.completed.split(',').map((entry) => entry.trim()).filter(Boolean)
729
+ : [],
730
+ });
592
731
  process.stdout.write(`${JSON.stringify(result, null, 2)}\n`);
732
+ if (!result.ok && !result.journaled) process.exitCode = 3;
593
733
  return;
594
734
  }
595
- process.stderr.write('usage: resumable-run.mjs read <projectRoot> <taskId>\n');
735
+ process.stderr.write(
736
+ 'usage: resumable-run.mjs read <projectRoot> <taskId> [--boundary <id>]\n'
737
+ + ' resumable-run.mjs suspend <projectRoot> <taskId> [--boundary <id>] '
738
+ + '[--phase <phase>] [--next <action>] [--completed <a,b,...>]\n',
739
+ );
596
740
  process.exitCode = 2;
597
741
  }
598
742
 
@@ -78,6 +78,18 @@ import path from 'node:path';
78
78
  import { spawn, spawnSync } from 'node:child_process';
79
79
 
80
80
  import { withAsyncLock } from './async-lock.mjs';
81
+ // TASK-005 (FR-005): the outcome seam loads LAZILY — a partial install missing
82
+ // observability-emit.mjs must degrade to "no outcome record" (telemetry is
83
+ // advisory), never crash the coordinator at module load.
84
+ let outcomeEmitModulePromise = null;
85
+ function loadOutcomeEmitModule() {
86
+ if (!outcomeEmitModulePromise) {
87
+ outcomeEmitModulePromise = import(
88
+ new URL('./observability-emit.mjs', import.meta.url).href
89
+ ).catch(() => null);
90
+ }
91
+ return outcomeEmitModulePromise;
92
+ }
81
93
 
82
94
  // Orphan-leak guard (2.4.1 / TASK-025): completion-gate.sh spawns this as
83
95
  // `node "$SCRIPT"`, so a host that kills the bash wrapper would reparent a
@@ -357,11 +369,21 @@ export function coordinateStopDecisions({ completion = null, watchdog = null, ha
357
369
  }
358
370
 
359
371
  // ─── Completion evaluator (spawn once) ────────────────────────────────────
360
- function runEvaluatorChild(scriptPath, input, budgetMs) {
372
+ function runEvaluatorChild(scriptPath, input, budgetMs, { cwd = null } = {}) {
361
373
  return new Promise((resolve, reject) => {
362
374
  const child = spawn(process.execPath, [scriptPath, '--evaluate-stop'], {
363
- // The child gets the same environment plus its own (smaller) self-kill budget.
364
- env: { ...process.env, UKIT_HOOK_DEADLINE_MS: String(Math.max(250, budgetMs)) },
375
+ // The child gets the same environment plus its own (smaller) self-kill
376
+ // budget, and the caller's project root is pinned twice — cwd AND an
377
+ // explicit CLAUDE_PROJECT_DIR — because the evaluator resolves its own
378
+ // root from `CLAUDE_PROJECT_DIR || process.cwd()`: an in-proc caller
379
+ // (bridge, tests) whose cwd is another project used to evaluate THAT
380
+ // project's state and silently release the wrong stop.
381
+ cwd: cwd || undefined,
382
+ env: {
383
+ ...process.env,
384
+ CLAUDE_PROJECT_DIR: cwd || process.env.CLAUDE_PROJECT_DIR || '',
385
+ UKIT_HOOK_DEADLINE_MS: String(Math.max(250, budgetMs)),
386
+ },
365
387
  stdio: ['pipe', 'pipe', 'pipe'],
366
388
  });
367
389
  // Tracked so the hook-deadline path can reap a still-running evaluator instead of
@@ -416,10 +438,237 @@ export async function evaluateCompletionPolicy({ projectRoot, rawInput, budgetMs
416
438
  } catch {
417
439
  throw new Error('completion evaluator unavailable: execution-ledger.mjs is missing (run: ukit install)');
418
440
  }
419
- const { stdout } = await runEvaluatorChild(ledgerPath, String(rawInput || ''), budgetMs);
441
+ // TASK-C85-014 (BL-014): consult the verification map BEFORE spawning the
442
+ // evaluator — the resolved recipe is stamped onto the route state the child
443
+ // re-reads, so the single authority (the ledger's evaluateCompletion) owns
444
+ // the verdict either way. Advisory: any failure leaves the gate as before.
445
+ try {
446
+ await stampVerificationRecipe({ projectRoot });
447
+ } catch {}
448
+ const { stdout } = await runEvaluatorChild(ledgerPath, String(rawInput || ''), budgetMs, { cwd: projectRoot });
420
449
  return normalizeEvaluatorStdout(stdout);
421
450
  }
422
451
 
452
+ // ─── Verification-map recipe consult (C85 TASK-014, BL-014) ────────────────
453
+ // The resolved verification recipe is an advisory INPUT to the completion
454
+ // evaluator, not a second gate: the coordinator reads the route state, loads
455
+ // verification-map.json through the shared index reader, classifies the
456
+ // artifact from the change set, and stamps `routeSummary.verificationRecipe`
457
+ // (+ `routeSummary.artifactClass`) back onto the state file before the
458
+ // evaluator child re-reads it. The stamp persists deliberately — the
459
+ // resume-intent lane (`hasUnfinishedCompletion`) then counts recipe evidence
460
+ // identically across compactions. Map absent/malformed or class unknown →
461
+ // no recipe, and a stale stamp is cleared so behavior returns to pre-map.
462
+ let verificationMapModulePromise = null;
463
+ function loadVerificationMapModule() {
464
+ if (!verificationMapModulePromise) {
465
+ verificationMapModulePromise = import(
466
+ new URL('../index/verification-map.mjs', import.meta.url).href
467
+ ).catch(() => null);
468
+ }
469
+ return verificationMapModulePromise;
470
+ }
471
+
472
+ function readChangedPathsForRecipe(projectRoot) {
473
+ try {
474
+ // -z NUL-terminates entries verbatim — the C-quoted porcelain form
475
+ // ("src/My Widget.jsx") would misclassify to null downstream (fix round:
476
+ // review minor). Rename/copy entries emit two fields under -z: `XY to`
477
+ // then the bare `from` path — the from field is skipped positionally.
478
+ const out = spawnSync('git', ['status', '--porcelain=v1', '-z', '--untracked-files=all'], {
479
+ cwd: projectRoot,
480
+ encoding: 'utf8',
481
+ timeout: 800,
482
+ stdio: ['ignore', 'pipe', 'ignore'],
483
+ });
484
+ if (out.error || out.status !== 0) return [];
485
+ const fields = out.stdout.split('\0');
486
+ const entries = [];
487
+ for (let i = 0; i < fields.length; i += 1) {
488
+ const field = fields[i];
489
+ if (!field || !/^.. /.test(field)) continue;
490
+ entries.push(field); // raw "XY path" — callers strip the status prefix
491
+ if (/^[RC]/.test(field)) i += 1; // consume the rename/copy source field
492
+ }
493
+ return entries;
494
+ } catch {
495
+ return [];
496
+ }
497
+ }
498
+
499
+ async function writeJsonAtomicLocal(filePath, value) {
500
+ await fs.mkdir(path.dirname(filePath), { recursive: true });
501
+ const tempPath = `${filePath}.tmp-${process.pid}-${Date.now()}-${Math.random().toString(16).slice(2)}`;
502
+ try {
503
+ await fs.writeFile(tempPath, `${JSON.stringify(value, null, 2)}\n`, 'utf8');
504
+ await fs.rename(tempPath, filePath);
505
+ } catch (error) {
506
+ try { await fs.rm(tempPath, { force: true }); } catch {}
507
+ throw error;
508
+ }
509
+ }
510
+
511
+ // The stamped recipe must round-trip through JSON: a plain data subset, never
512
+ // the reader's full object (which carries no functions anyway, but this keeps
513
+ // the on-disk contract explicit).
514
+ function serializeVerificationRecipe(recipe, artifactClass) {
515
+ if (!recipe) return null;
516
+ return {
517
+ artifactClass,
518
+ name: recipe.name || null,
519
+ check: recipe.check || null,
520
+ harness: recipe.harness || null,
521
+ fallback: recipe.fallback || null,
522
+ evidenceRequired: Array.isArray(recipe.evidenceRequired) ? recipe.evidenceRequired : [],
523
+ };
524
+ }
525
+
526
+ async function stampVerificationRecipe({ projectRoot }) {
527
+ const mapModule = await loadVerificationMapModule();
528
+ if (!mapModule) return; // reader absent (partial install): pre-map behavior
529
+ const statePath = path.join(projectRoot, '.claude', 'ukit', 'skill-router-state.json');
530
+ let state;
531
+ try {
532
+ state = JSON.parse(await fs.readFile(statePath, 'utf8'));
533
+ } catch {
534
+ return; // no route state → nothing to stamp
535
+ }
536
+ const routeSummary = state?.routeSummary;
537
+ if (!routeSummary || typeof routeSummary !== 'object') return;
538
+
539
+ const changedPaths = readChangedPathsForRecipe(projectRoot)
540
+ .map((line) => line.replace(/^(..)\s+/, ''));
541
+ const artifactClass = mapModule.classifyArtifactClass(changedPaths);
542
+ const map = await mapModule.loadVerificationMap({ projectRoot });
543
+ const recipe = artifactClass && map
544
+ ? await mapModule.recipeForArtifact({
545
+ playbookId: routeSummary.playbookId || null,
546
+ artifactClass,
547
+ map,
548
+ })
549
+ : null;
550
+ const serialized = serializeVerificationRecipe(recipe, artifactClass);
551
+
552
+ const before = JSON.stringify(routeSummary.verificationRecipe ?? null);
553
+ const after = JSON.stringify(serialized);
554
+ if (before === after) return; // no stamp change → no rewrite
555
+ routeSummary.verificationRecipe = serialized;
556
+ if (artifactClass) routeSummary.artifactClass = artifactClass;
557
+ else delete routeSummary.artifactClass;
558
+ await writeJsonAtomicLocal(statePath, state);
559
+ }
560
+
561
+ // ─── Review-policy advisory lane (C85 TASK-015, BL-015) ────────────────────
562
+ // "Is one more verification round needed?" becomes a signal decision at the
563
+ // pre-Stop gate. The lane is advisory: the verdict + namedCheck are recorded
564
+ // on the coordinator output and a `kind:'review-policy'` decision receipt is
565
+ // appended for calibration — the completion gate's own block logic and the
566
+ // merged decision are untouched (single Stop authority preserved).
567
+ let reviewPolicyModulePromise = null;
568
+ function loadReviewPolicyModule() {
569
+ if (!reviewPolicyModulePromise) {
570
+ reviewPolicyModulePromise = import(
571
+ new URL('../index/review-policy.mjs', import.meta.url).href
572
+ ).catch(() => null);
573
+ }
574
+ return reviewPolicyModulePromise;
575
+ }
576
+
577
+ let ledgerReceiptModulePromise = null;
578
+ function loadLedgerReceiptModule() {
579
+ if (!ledgerReceiptModulePromise) {
580
+ ledgerReceiptModulePromise = import(
581
+ new URL('./execution-ledger.mjs', import.meta.url).href
582
+ ).catch(() => null);
583
+ }
584
+ return ledgerReceiptModulePromise;
585
+ }
586
+
587
+ /**
588
+ * Evaluate the review decision policy once per stop when a route record
589
+ * exists, append the calibration receipt, and stamp reviewRounds +
590
+ * lastReviewPolicy back onto the route state for the next stop's round
591
+ * accounting. Returns the decision {action, namedCheck?, signalsHit[],
592
+ * reviewRounds, ...} or null when there is no route record / the module is
593
+ * absent (partial install). Never throws — an advisory lane must not wedge
594
+ * the coordinator.
595
+ *
596
+ * `completionKind` ('block'|'advisory'|'complete'|null) is the completion
597
+ * evaluator's verdict, used only to mark which done criteria are evidenced.
598
+ */
599
+ export async function evaluateStopReviewPolicy({ projectRoot, completionKind = null } = {}) {
600
+ try {
601
+ const policyModule = await loadReviewPolicyModule();
602
+ if (!policyModule || typeof policyModule.evaluateReviewPolicy !== 'function') return null;
603
+
604
+ const statePath = path.join(projectRoot, '.claude', 'ukit', 'skill-router-state.json');
605
+ let state;
606
+ try {
607
+ state = JSON.parse(await fs.readFile(statePath, 'utf8'));
608
+ } catch {
609
+ return null; // no route record → nothing to evaluate
610
+ }
611
+ const routeSummary = state?.routeSummary;
612
+ if (!routeSummary || typeof routeSummary !== 'object') return null;
613
+
614
+ // A gate-released stop means the contract's evidence landed ('verified');
615
+ // otherwise the criteria are unproven — the gate still owns the block, the
616
+ // policy only reports what the next round would have to check.
617
+ const evidenceSatisfied = completionKind === 'complete';
618
+ const requiredTokens = [
619
+ ...(Array.isArray(routeSummary?.executionContract?.completionEvidence)
620
+ ? routeSummary.executionContract.completionEvidence
621
+ : []),
622
+ ...(Array.isArray(routeSummary?.verificationRecipe?.evidenceRequired)
623
+ ? routeSummary.verificationRecipe.evidenceRequired
624
+ : []),
625
+ ].filter((token) => typeof token === 'string' && token.trim());
626
+ const doneCriteria = [...new Set(requiredTokens)].map((name) => ({
627
+ name,
628
+ status: evidenceSatisfied ? 'verified' : 'missing',
629
+ }));
630
+
631
+ const historySignals = routeSummary.historySignals && typeof routeSummary.historySignals === 'object'
632
+ ? routeSummary.historySignals
633
+ : {};
634
+ const round = Number.isFinite(routeSummary.reviewRounds)
635
+ ? routeSummary.reviewRounds
636
+ : 0;
637
+
638
+ const ledgerModule = await loadLedgerReceiptModule();
639
+ const appendDecisionReceipt = ledgerModule?.appendDecisionReceipt;
640
+
641
+ const decision = policyModule.evaluateReviewPolicy({
642
+ signals: routeSummary.suspicionSignals ?? {},
643
+ doneCriteria,
644
+ evidenceClasses: [],
645
+ fixLoopCount: Number.isFinite(historySignals.fixLoopCount) ? historySignals.fixLoopCount : 0,
646
+ round,
647
+ cap: policyModule.DEFAULT_REVIEW_ROUND_CAP,
648
+ projectRoot,
649
+ appendDecisionReceipt,
650
+ });
651
+ if (!decision || typeof decision !== 'object') return null;
652
+
653
+ // Persist the round counter + last verdict for the next stop (same atomic
654
+ // write as the recipe stamp). A stale verdict is honest — it is the last
655
+ // recorded policy decision, keyed by its action.
656
+ routeSummary.reviewRounds = Number.isFinite(decision.reviewRounds) ? decision.reviewRounds : round;
657
+ routeSummary.lastReviewPolicy = {
658
+ action: decision.action,
659
+ namedCheck: decision.namedCheck ?? null,
660
+ signalsHit: Array.isArray(decision.signalsHit) ? decision.signalsHit : [],
661
+ at: new Date().toISOString(),
662
+ };
663
+ await writeJsonAtomicLocal(statePath, state);
664
+
665
+ return decision;
666
+ } catch {
667
+ return null; // advisory lane: never wedge the coordinator
668
+ }
669
+ }
670
+
671
+
423
672
  // ─── Handoff-cursor evaluator (in-process, advisory-on-failure) ───────────
424
673
  /**
425
674
  * Read `.ukit/storage/config.json` → `handoff.fullstack.stopGate*` merged over
@@ -813,7 +1062,6 @@ export async function runStopCoordinator({
813
1062
  failures.watchdog = error?.message || String(error);
814
1063
  process.stderr.write(`[ukit-stop-coordinator] task-watchdog evaluator failed: ${failures.watchdog}\n`);
815
1064
  }
816
-
817
1065
  let handoff = null;
818
1066
  try {
819
1067
  handoff = await evaluateHandoffCursor({ projectRoot, now, lockBudgetMs });
@@ -822,7 +1070,76 @@ export async function runStopCoordinator({
822
1070
  process.stderr.write(`[ukit-stop-coordinator] handoff-cursor evaluator failed: ${failures.handoff}\n`);
823
1071
  }
824
1072
 
825
- return { skip: null, merged: coordinateStopDecisions({ completion, watchdog, handoff, failures }) };
1073
+ const merged = coordinateStopDecisions({ completion, watchdog, handoff, failures });
1074
+
1075
+ // C85 TASK-015 (BL-015): advisory review-policy verdict. The evaluator runs
1076
+ // only when a route record exists; its action + namedCheck are recorded on
1077
+ // the output and a review-policy receipt lands in decisions.tsv. The gate's
1078
+ // own block logic is unchanged — a non-finish verdict only surfaces as a
1079
+ // systemMessage note, never a block.
1080
+ try {
1081
+ const reviewPolicy = await evaluateStopReviewPolicy({
1082
+ projectRoot,
1083
+ completionKind: completion?.kind ?? null,
1084
+ });
1085
+ if (reviewPolicy) {
1086
+ merged.reviewPolicy = reviewPolicy;
1087
+ if (reviewPolicy.action !== 'finish') {
1088
+ const note = `[ukit-stop-coordinator] review policy (advisory): ${reviewPolicy.action}`
1089
+ + (reviewPolicy.namedCheck ? ` — ${reviewPolicy.namedCheck}` : '')
1090
+ + (Array.isArray(reviewPolicy.signalsHit) && reviewPolicy.signalsHit.length
1091
+ ? ` (signals: ${reviewPolicy.signalsHit.join(', ')})` : '');
1092
+ merged.systemMessage = merged.systemMessage
1093
+ ? `${merged.systemMessage}\n${note}`
1094
+ : note;
1095
+ }
1096
+ }
1097
+ } catch { /* advisory verdicts never affect the stop decision */ }
1098
+
1099
+ // TASK-005 (FR-005): the coordinator's terminal verdict lands ONE
1100
+ // `outcome.observed` record per coordinated stop — the merged evaluator
1101
+ // decision is the session's honest outcome at this stop. Fire-and-forget:
1102
+ // awaited inside the emit module's own 250ms bound; any failure returns
1103
+ // null and never changes the decision the caller is about to emit.
1104
+ {
1105
+ const stopOutcomeVerdict = merged.decision === 'block' ? 'BLOCKED'
1106
+ : typeof merged.systemMessage === 'string' && merged.systemMessage ? 'INCONCLUSIVE'
1107
+ : 'VERIFIED';
1108
+ // runId joins ledger-emitted outcomes on the same session identity the
1109
+ // ledger derives (run-<session>); transcript-only payloads hash the path
1110
+ // instead of leaking it; 'no-session' stays null (missing-data contract).
1111
+ const outcomePayload = sessionKey === 'no-session'
1112
+ ? payload
1113
+ : (typeof payload?.session_id === 'string' && payload.session_id
1114
+ ? { session_id: sessionKey }
1115
+ : { transcript_path: sessionKey });
1116
+ try {
1117
+ const emitModule = await loadOutcomeEmitModule();
1118
+ if (!emitModule || typeof emitModule.buildOutcomeObservedRecord !== 'function'
1119
+ || typeof emitModule.emitOutcomeObserved !== 'function') {
1120
+ return { skip: null, merged };
1121
+ }
1122
+ const outcomeRecord = emitModule.buildOutcomeObservedRecord({
1123
+ outcome: {
1124
+ verdict: stopOutcomeVerdict,
1125
+ source: 'stop-coordinator',
1126
+ routeId: null,
1127
+ routeFingerprint: null,
1128
+ ledgerKey: null,
1129
+ runId: emitModule.outcomeRunId(outcomePayload),
1130
+ evidenceIds: [],
1131
+ cost: { latencyMs: null, rounds: null },
1132
+ reviewActions: [],
1133
+ },
1134
+ projectRoot,
1135
+ payload: typeof payload?.session_id === 'string' && payload.session_id ? payload : outcomePayload,
1136
+ env: process.env,
1137
+ });
1138
+ await emitModule.emitOutcomeObserved({ projectRoot, record: outcomeRecord, deadlineMs: 200 });
1139
+ } catch { /* telemetry never blocks or delays the stop decision */ }
1140
+ }
1141
+
1142
+ return { skip: null, merged };
826
1143
  }
827
1144
 
828
1145
  // ─── CLI ──────────────────────────────────────────────────────────────────
@@ -80,6 +80,14 @@ For clearly non-code specialist lanes (docs-only, status, task queue), skip the
80
80
  - If verification is needed, prefer `node .codex/ukit/index/verify-context.mjs ...`.
81
81
  - **Treat helper commands as internal orchestration.** Run them yourself when needed; never ask end users to run them.
82
82
 
83
+ ## Enforcement Ceiling + Playbooks (Codex contract)
84
+
85
+ - Codex exposes **no hook surface** — routing, gates, and playbooks reach you only as instructions in this file. Every playbook-covered ask resolves to either a measured live path or this advisory fallback; nothing here is hard-blocked.
86
+ - Natural prompts route through `node .codex/ukit/index/route-task.mjs "<prompt>"` — the same route record as the other engines, invoked by instruction, not by hook.
87
+ - When a route returns a `playbookId`, read the playbook file `~/.ukit/playbooks/<id>.md` before implementing, and follow its steps.
88
+ - Outcome records for codex-engine runs carry `enforcement: advisory` — the honest marker that nothing was enforced.
89
+ - Honest ceiling: the codex lane caps at **SHADOW/advisory**. Report advisory, never a live-enforcement claim.
90
+
83
91
  ## UKit Shared Runtime
84
92
 
85
93
  - Shared runtime state lives in `.ukit/storage/`.
@@ -63,7 +63,11 @@ export const HOOK_EVENT_MAP = {
63
63
  },
64
64
  before_agent_start: ['sensitive-data-guard.mjs', 'skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
65
65
  'session.compacting': ['reinject-context.sh'],
66
- session_start: ['project-important.sh', 'auto-prune-bash.sh', 'reset-compact-pressure.sh', 'handoff-resume.sh'],
66
+ // TASK-003 (FR-003): the in-proc emit step runs LAST — project-important.sh
67
+ // owns the mandate-first lead (TASK-042 contract); the emit is advisory,
68
+ // <250ms, and writes the session's `execution.started` record after the
69
+ // context-producing scripts.
70
+ session_start: ['project-important.sh', 'auto-prune-bash.sh', 'reset-compact-pressure.sh', 'handoff-resume.sh', 'observability-emit.mjs'],
67
71
  // TASK-002 (cycle 45): omp emits `session_stop` (no `session_end`); the chain
68
72
  // runs only on RELEASED stops inside runSessionStop — never on blocked or
69
73
  // dedupe-skipped stops, since `ukit memory episode` dedupes on meta.ledgerKey
@@ -135,6 +139,9 @@ export const ADVISORY_SCRIPTS = new Set([
135
139
  'reset-compact-pressure.sh',
136
140
  'handoff-resume.sh',
137
141
  'session-episode.sh',
142
+ // TASK-003 (FR-003): session-start emit step, advisory — a telemetry fault
143
+ // can never block or delay session start.
144
+ 'observability-emit.mjs',
138
145
  ]);
139
146
 
140
147
  function classifyFailure(scriptName) {
@@ -23,7 +23,7 @@ UKit state lives at two levels:
23
23
  `ukit install` and **never overwritten** afterwards — every user item is
24
24
  `mergeStrategy: skip`. Per `manifests/platform.user.yaml` the seeds are:
25
25
  `README.md`, `storage/config.json`, `playbooks/bug-fix.md`,
26
- `playbooks/issue-implementation.md`.
26
+ `playbooks/feature-implementation.md`.
27
27
 
28
28
  Precedence (per key / per id):
29
29
 
@@ -97,6 +97,11 @@
97
97
  "stage": "default"
98
98
  }
99
99
  },
100
+ "crossCheck": {
101
+ "enabled": true,
102
+ "byRole": {},
103
+ "byModel": {}
104
+ },
100
105
  "decisionPlane": {
101
106
  "enabled": true,
102
107
  "stage": "default",
@@ -229,6 +234,7 @@
229
234
  "escalation": {
230
235
  "enabled": true,
231
236
  "debugLoopThreshold": 2,
237
+ "laneDeepening": true,
232
238
  "tierOrder": [
233
239
  "lite",
234
240
  "code",
@@ -422,6 +428,20 @@
422
428
  "stage": "default"
423
429
  }
424
430
  },
431
+ "subagentOrchestrator": {
432
+ "telemetry": {
433
+ "stage": "off"
434
+ },
435
+ "referenceHandoff": {
436
+ "stage": "off"
437
+ },
438
+ "roleAdvice": {
439
+ "stage": "off"
440
+ },
441
+ "contextSelection": {
442
+ "stage": "off"
443
+ }
444
+ },
425
445
  "safePatch": {
426
446
  "enabled": true,
427
447
  "strictSharedRisk": true,