@ngockhoale/ukit 3.3.3 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/CHANGELOG.md +40 -0
  2. package/manifests/engineConformance.yaml +17 -1
  3. package/manifests/hostCapabilities.yaml +68 -1
  4. package/manifests/platform.full.yaml +138 -0
  5. package/manifests/platform.user.yaml +255 -3
  6. package/package.json +1 -1
  7. package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
  8. package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
  9. package/scripts/probe/codex-capability-probe.mjs +169 -0
  10. package/src/cli/commands/doctor.js +168 -0
  11. package/src/cli/commands/indexTools.js +7 -0
  12. package/src/cli/commands/metrics.js +66 -2
  13. package/src/cli/commands/playbook.js +4 -4
  14. package/src/cli/commands/vm.js +49 -8
  15. package/src/core/agentRuntime/adapters.js +328 -27
  16. package/src/core/agentRuntime/artifacts.js +89 -0
  17. package/src/core/agentRuntime/context.js +345 -1
  18. package/src/core/agentRuntime/contract.js +296 -0
  19. package/src/core/agentRuntime/eventStore.js +176 -0
  20. package/src/core/agentRuntime/shadowRun.js +481 -5
  21. package/src/core/agentRuntime/telemetry.js +121 -0
  22. package/src/core/observability/emit/lifecycle.js +68 -1
  23. package/src/core/observability/emit/sessionBoot.js +393 -0
  24. package/src/core/observability/privacy/allowlist.js +10 -1
  25. package/src/core/observability/schema/registry.js +10 -0
  26. package/src/core/runtimeConfig.js +133 -0
  27. package/src/core/userPlaybooks.js +18 -3
  28. package/src/decision/registry.js +19 -0
  29. package/src/diagnostics/feedbackEvents.js +7 -4
  30. package/src/diagnostics/routeOutcomes.js +51 -6
  31. package/src/diagnostics/skillAccuracy.js +43 -3
  32. package/src/index/crossCheckMatrix.js +412 -0
  33. package/src/index/fixLoopEscalation.js +453 -0
  34. package/src/index/playbookRegistry.js +691 -0
  35. package/src/index/reviewPolicy.js +368 -0
  36. package/src/index/routeResolver.js +915 -0
  37. package/src/index/sessionHistoryExtractor.js +359 -0
  38. package/src/index/taskRouting.js +764 -581
  39. package/src/index/tierSelection.js +308 -0
  40. package/src/index/verificationMap.js +404 -0
  41. package/template_project/.claude/hooks/observability-emit.mjs +14 -0
  42. package/template_project/.claude/hooks/record-execution.mjs +19 -1
  43. package/template_project/.claude/hooks/skill-router.sh +691 -25
  44. package/template_project/.claude/hooks/verification-guard.sh +230 -1
  45. package/template_project/.claude/settings.json +2 -2
  46. package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
  47. package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
  48. package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
  49. package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
  50. package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
  51. package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
  52. package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
  53. package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
  54. package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
  55. package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
  56. package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
  57. package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
  58. package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
  59. package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
  60. package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
  61. package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
  62. package/template_project/.codex/README.md +8 -0
  63. package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
  64. package/template_project/ukit/README.md +1 -1
  65. package/template_project/ukit/storage/config.json +20 -0
  66. package/template_user/playbooks/architecture-decision.md +28 -0
  67. package/template_user/playbooks/autonomous-run.md +43 -0
  68. package/template_user/playbooks/autopilot-full.md +59 -0
  69. package/template_user/playbooks/autopilot-stack.md +54 -0
  70. package/template_user/playbooks/babysit.md +39 -0
  71. package/template_user/playbooks/bug-fix.md +3 -1
  72. package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
  73. package/template_user/playbooks/hillclimb.md +44 -0
  74. package/template_user/playbooks/investigation.md +21 -0
  75. package/template_user/playbooks/migration.md +21 -0
  76. package/template_user/playbooks/open-pr.md +48 -0
  77. package/template_user/playbooks/orchestrate.md +45 -0
  78. package/template_user/playbooks/performance.md +33 -0
  79. package/template_user/playbooks/prototype.md +28 -0
  80. package/template_user/playbooks/refactor.md +19 -0
  81. package/template_user/playbooks/release.md +28 -0
  82. package/template_user/playbooks/runtime-forensics.md +23 -0
  83. package/template_user/playbooks/session-pickup.md +31 -0
  84. package/template_user/playbooks/shipping.md +53 -0
  85. package/template_user/playbooks/skill-evaluation.md +48 -0
  86. package/template_user/playbooks/small-feature.md +20 -0
  87. package/template_user/playbooks/verification-map.json +153 -0
  88. package/template_user/playbooks/verification.md +22 -0
  89. package/template_user/playbooks/worktree-cleanup.md +37 -0
@@ -22,16 +22,35 @@
22
22
  * unroutable node degrades to the deterministic escalation lane
23
23
  * (recovery_required) exactly as the V-05 contract requires.
24
24
  *
25
+ * Live lane (BL-031): `runLivePlan` drives the same engine through the
26
+ * REAL startFn — a `createSupervisor` launch per RUN node rooted at
27
+ * `.ukit/storage/agent-runtime/live/<planId>/`. Only plan specs carrying
28
+ * the supervisor contract (`spec.command` or `spec.argv`) launch;
29
+ * descriptive `spec.cmd` prose is never shelled out — an unlaunchable
30
+ * spec resolves its node 'failed' with the launch code recorded.
31
+ * Supervisor terminal outcomes are relayed back into the engine as
32
+ * selector-matched `operation.completed` events. The lane is gated at
33
+ * `decisionRuntime.vm` canary/default: 'off' → `stage_off`, 'shadow' →
34
+ * `live_stage_required`, both zero-write typed refuses.
35
+ *
36
+ * Replay comparison (BL-031): `buildReplayComparison` pairs VM replay
37
+ * verdicts with archived doc-pipeline outcomes and applies the
38
+ * pre-registered promotion bar verbatim — promote ONLY if the VM catches
39
+ * ≥1 defect the doc pipeline missed OR cuts cycle time ≥25% at equal
40
+ * verdicts. `recordReplayComparison` appends the artifact to the evidence
41
+ * store where `ukit vm evidence` reports it.
42
+ *
25
43
  * Never throws: every failure resolves to a typed
26
44
  * `{ ok:false, code, reason }` — `stage_off` (decisionRuntime.vm 'off' →
27
- * zero filesystem writes, G-D1), plan-load error codes verbatim from
28
- * loadPlan, `malformed_plan`, `unknown_plan_id`, `engine_error`,
29
- * `unterminated` (bounded-loop guard tripped).
45
+ * zero filesystem writes, G-D1), `live_stage_required`, plan-load error
46
+ * codes verbatim from loadPlan, `malformed_plan`, `unknown_plan_id`,
47
+ * `engine_error`, `unterminated` (bounded-loop guard tripped).
30
48
  *
31
49
  * Evidence (D-03): each run appends one JSONL record to
32
50
  * `.ukit/storage/agent-runtime/evidence/<key>.jsonl`
33
- * `{planId, irHash, planVersion, planInstanceId, status, transitions,
51
+ * `{planId, irHash, planVersion, planInstanceId, status, lane?, transitions,
34
52
  * wallMs, qualityDelta, wallP95Ratio, costRatio, forbidden, at}` —
53
+ * `lane:'live'` marks records produced by runLivePlan (absent = shadow).
35
54
  * qualityDelta/wallP95Ratio/costRatio stay `null` until a measured
36
55
  * baseline exists (promotion.js treats missing metrics as non-passing,
37
56
  * so shadow evidence can never self-promote). The file rotates to
@@ -43,8 +62,9 @@ import { promises as fs } from 'node:fs';
43
62
  import path from 'node:path';
44
63
 
45
64
  import { resolveDecisionRuntimeStage } from '../runtimeConfig.js';
46
- import { isTerminal, CONTRACT_VERSION } from './contract.js';
65
+ import { isTerminal, CONTRACT_VERSION, SIDE_EFFECT_CLASSES } from './contract.js';
47
66
  import { createVmEngine } from './vmEngine.js';
67
+ import { createSupervisor } from './supervisor.js';
48
68
  import { loadPlan } from './planLibrary.js';
49
69
 
50
70
  const RUNTIME_REL = path.join('.ukit', 'storage', 'agent-runtime');
@@ -382,3 +402,459 @@ async function engineStateSafe(engine, pi) {
382
402
  return null;
383
403
  }
384
404
  }
405
+
406
+ // --- BL-031: live lane (real startFn) --------------------------------------
407
+
408
+ // Live runs are the evidence-gated lane: only canary/default stages execute
409
+ // real operation launches. 'shadow' is evidence collection via replay only.
410
+ const LIVE_STAGES = new Set(['canary', 'default']);
411
+
412
+ // Poll cadence and a hard wall budget for the live driver loop. A launched
413
+ // op can legitimately run for minutes; nodes whose spec cannot produce an
414
+ // outcome (parked WAIT_EVENT/BRANCH, lost launches) surface 'unterminated'
415
+ // the moment nothing is pending rather than spinning to the deadline.
416
+ const LIVE_POLL_MS = 50;
417
+ const LIVE_MAX_WALL_MS = 10 * 60 * 1000;
418
+
419
+ const SIDE_EFFECT_SET = new Set(SIDE_EFFECT_CLASSES);
420
+
421
+ function sleep(ms) {
422
+ return new Promise((resolve) => setTimeout(resolve, ms));
423
+ }
424
+
425
+ /**
426
+ * Translate a plan-node spec into a supervisor launch spec, or null when
427
+ * the spec is not launchable. `spec.cmd` in the plan DSL is descriptive
428
+ * prose — it is never shelled out; only the supervisor contract fields
429
+ * (`spec.command` or `spec.argv`) produce a real child process.
430
+ */
431
+ function liveLaunchSpec(pi, nodeId, node, attempt) {
432
+ const spec = node?.spec && typeof node.spec === 'object' ? node.spec : {};
433
+ const argv = Array.isArray(spec.argv) && spec.argv.length > 0
434
+ && spec.argv.every((a) => typeof a === 'string' && a !== '')
435
+ ? spec.argv : null;
436
+ const command = typeof spec.command === 'string' && spec.command !== ''
437
+ ? spec.command : null;
438
+ if (argv == null && command == null) return null;
439
+ const firstClass = Array.isArray(spec.sideEffects) ? spec.sideEffects[0] : null;
440
+ const retryClass = node?.retry?.sideEffectClass;
441
+ const sideEffectClass = SIDE_EFFECT_SET.has(firstClass)
442
+ ? firstClass
443
+ : (SIDE_EFFECT_SET.has(retryClass) ? retryClass : 'write');
444
+ const launch = {
445
+ // one supervisor operation per attempt — a retry must land on a fresh
446
+ // journal, not collide with the previous attempt's seq
447
+ operationId: `${pi}:${nodeId}:a${attempt}`,
448
+ attempt,
449
+ sideEffectClass,
450
+ };
451
+ if (argv != null) launch.argv = argv;
452
+ else launch.command = command;
453
+ if (typeof spec.cwd === 'string' && spec.cwd !== '') launch.cwd = spec.cwd;
454
+ if (spec.env !== null && typeof spec.env === 'object' && !Array.isArray(spec.env)) {
455
+ launch.env = spec.env;
456
+ }
457
+ if (typeof spec.host === 'string' && spec.host !== '') launch.host = spec.host;
458
+ if (Number.isInteger(node?.timeoutMs) && node.timeoutMs > 0) {
459
+ launch.wallTimeoutMs = node.timeoutMs;
460
+ }
461
+ return launch;
462
+ }
463
+
464
+ /** SemanticEvent relaying a supervised terminal outcome into the engine. */
465
+ function liveOutcomeEvent(pi, nodeId, node, seq, outcome, observedAt, code) {
466
+ return {
467
+ eventId: `live-${pi}-${nodeId}-${seq}`,
468
+ operationId: `${pi}:${nodeId}`,
469
+ seq,
470
+ eventType: node?.selector?.eventType ?? 'operation.completed',
471
+ observedAt,
472
+ producerVersion: 'ukit-vm-live',
473
+ contractVersion: CONTRACT_VERSION,
474
+ privacyClass: 'internal',
475
+ artifactRefs: [],
476
+ safePayload: {
477
+ outcome,
478
+ ...(typeof code === 'string' && code !== '' ? { code } : {}),
479
+ },
480
+ };
481
+ }
482
+
483
+ /**
484
+ * Run one plan through the REAL live path: a real `createVmEngine` (journals
485
+ * under `.ukit/storage/agent-runtime/live/<planId>/`) whose `startFn`
486
+ * launches RUN-node specs through `createSupervisor`, and a driver that
487
+ * relays each supervised terminal outcome back into the engine as a
488
+ * selector-matched outcome event.
489
+ *
490
+ * Stage gate (BL-031): 'off' → `stage_off`; 'shadow' → `live_stage_required`
491
+ * — both typed refuses with zero writes. Execution requires canary/default.
492
+ *
493
+ * A run that reaches a terminal status (incl. 'unterminated' — a real
494
+ * bounded outcome, e.g. a plan parked on a producer-free WAIT_EVENT)
495
+ * appends one evidence record with `lane:'live'`; 'failed' and
496
+ * 'recovery_required' mark the run forbidden. Never throws.
497
+ *
498
+ * @param {string} planId
499
+ * @param {{projectRoot?: string, config?: object, plansDir?: string,
500
+ * engineDir?: string, supervisorDir?: string, supervisor?: object,
501
+ * classifyFn?: Function, decisionClient?: object, decider?: object,
502
+ * evidence?: boolean, maxWallMs?: number, pollMs?: number}} [opts]
503
+ * `supervisor` passthrough injects createSupervisor opts (spawnImpl etc.)
504
+ * for tests; `plansDir`/`engineDir`/`supervisorDir` are fixture escapes —
505
+ * the CLI never passes them.
506
+ * @returns {Promise<object>} frozen result or typed refuse — same shape as
507
+ * runShadowPlan's contract.
508
+ */
509
+ export async function runLivePlan(planId, opts = {}) {
510
+ const { projectRoot, config = null } = opts;
511
+ try {
512
+ const stage = resolveDecisionRuntimeStage(config, 'vm');
513
+ if (stage === 'off') {
514
+ return refuse('stage_off', 'decisionRuntime.vm.stage is off — zero writes');
515
+ }
516
+ if (!LIVE_STAGES.has(stage)) {
517
+ return refuse(
518
+ 'live_stage_required',
519
+ `decisionRuntime.vm.stage is ${stage} — live execution requires canary or default`,
520
+ );
521
+ }
522
+
523
+ const loaded = loadPlan(planId, {
524
+ ...(opts.plansDir ? { plansDir: opts.plansDir } : {}),
525
+ config,
526
+ });
527
+ if (!loaded.ok) {
528
+ const first = loaded.errors?.[0] ?? {};
529
+ return refuse(
530
+ typeof first.code === 'string' ? first.code : 'plan_not_found',
531
+ `${planId}: ${JSON.stringify(first)}`,
532
+ );
533
+ }
534
+ const plan = loaded.plan;
535
+ if (!plan || !(plan.nodes instanceof Map) || plan.nodes.size === 0) {
536
+ return refuse('malformed_plan', `${planId}: compiled plan has no nodes`);
537
+ }
538
+
539
+ const dir = typeof opts.engineDir === 'string' && opts.engineDir !== ''
540
+ ? opts.engineDir
541
+ : (typeof projectRoot === 'string' && projectRoot !== ''
542
+ ? path.join(runtimeDir(projectRoot), 'live', planId)
543
+ : null);
544
+ if (dir == null) return refuse('invalid_root', 'projectRoot or engineDir required');
545
+
546
+ const supDir = typeof opts.supervisorDir === 'string' && opts.supervisorDir !== ''
547
+ ? opts.supervisorDir
548
+ : path.join(dir, 'supervisor');
549
+
550
+ // The vm stage is this lane's gate; the supervisor's own
551
+ // decisionRuntime.supervisor.stage kill switch still applies — an
552
+ // operator-set 'off' refuses launches typed and fails the node honestly.
553
+ const supervisor = createSupervisor({
554
+ runtimeDir: supDir,
555
+ config,
556
+ // closeWaitMs is capped low: supervisor.close()'s deadline timer keeps
557
+ // the event loop alive for its full span even when op chains already
558
+ // settled — 250ms still drains terminal journals without parking the
559
+ // CLI for the library default (5s per run).
560
+ limits: { closeWaitMs: 250 },
561
+ ...(opts.supervisor && typeof opts.supervisor === 'object' ? opts.supervisor : {}),
562
+ });
563
+
564
+ const inFlight = new Map(); // nodeId -> supervisor start() handle
565
+ const launchErrors = new Map(); // nodeId -> typed code (preflight/launch refuse)
566
+ const attempts = new Map(); // nodeId -> attempt counter (journal-safe opIds)
567
+
568
+ // REAL startFn: every activated RUN node launches a supervised child.
569
+ // A spec outside the supervisor contract (descriptive `spec.cmd`) never
570
+ // reaches spawn — its failure code is delivered as a 'failed' outcome.
571
+ const startFn = async (spec, { planInstanceId, nodeId, node }) => {
572
+ const attempt = (attempts.get(nodeId) ?? 0) + 1;
573
+ attempts.set(nodeId, attempt);
574
+ const launchSpec = liveLaunchSpec(planInstanceId, nodeId, node ?? { spec }, attempt);
575
+ if (launchSpec == null) {
576
+ launchErrors.set(nodeId, 'invalid_spec');
577
+ return;
578
+ }
579
+ try {
580
+ const res = await supervisor.start(launchSpec, {});
581
+ if (res?.unsupported) {
582
+ launchErrors.set(nodeId, res.code ?? 'launch_refused');
583
+ } else if (res?.ok === false) {
584
+ launchErrors.set(nodeId, res.code ?? 'launch_failed');
585
+ } else {
586
+ inFlight.set(nodeId, res);
587
+ }
588
+ } catch (err) {
589
+ launchErrors.set(nodeId, err?.code ?? 'launch_error');
590
+ }
591
+ };
592
+
593
+ const engine = createVmEngine({
594
+ dir,
595
+ ...(typeof opts.classifyFn === 'function' ? { classifyFn: opts.classifyFn } : {}),
596
+ ...(opts.decisionClient != null ? { decisionClient: opts.decisionClient } : {}),
597
+ ...(opts.decider != null ? { decider: opts.decider } : {}),
598
+ startFn,
599
+ config,
600
+ });
601
+
602
+ const started = Date.now();
603
+ const maxWallMs = Number.isInteger(opts.maxWallMs) && opts.maxWallMs > 0
604
+ ? opts.maxWallMs : LIVE_MAX_WALL_MS;
605
+ const pollMs = Number.isInteger(opts.pollMs) && opts.pollMs > 0
606
+ ? opts.pollMs : LIVE_POLL_MS;
607
+ const deadline = started + maxWallMs;
608
+
609
+ let pi = null;
610
+ const transitions = [];
611
+ let status = 'running';
612
+ try {
613
+ const startRes = await engine.start(plan);
614
+ if (!startRes || startRes.unsupported || typeof startRes.planInstanceId !== 'string') {
615
+ return refuse('malformed_plan', `${planId}: engine refused start (${startRes?.code ?? 'no instance'})`);
616
+ }
617
+ pi = startRes.planInstanceId;
618
+
619
+ while (Date.now() <= deadline) {
620
+ // eslint-disable-next-line no-await-in-loop
621
+ const snapshot = await engine.state(pi);
622
+ status = snapshot.status;
623
+ if (isTerminal(status) || status === 'recovery_required') break;
624
+
625
+ let acted = false;
626
+ const seqFor = (nodeId) => (snapshot.cursor?.[nodeId] ?? 0) + 1;
627
+ const observed = () => new Date().toISOString();
628
+
629
+ // Deliver preflight/launch failures for nodes still 'running'.
630
+ for (const [nodeId, code] of launchErrors) {
631
+ if (snapshot.nodes[nodeId]?.state === 'running' && !inFlight.has(nodeId)) {
632
+ // eslint-disable-next-line no-await-in-loop
633
+ const res = await engine.deliver(
634
+ liveOutcomeEvent(pi, nodeId, plan.nodes.get(nodeId), seqFor(nodeId), 'failed', observed(), code),
635
+ );
636
+ if (!res.consumed) {
637
+ return refuse('event_not_consumed',
638
+ `${planId}/${nodeId}: deliver rejected (${res.code ?? 'unknown'})`);
639
+ }
640
+ for (const t of res.transitions ?? []) {
641
+ transitions.push({ node: t.node, to: t.to, ...(t.branch !== undefined ? { branch: t.branch } : {}) });
642
+ }
643
+ launchErrors.delete(nodeId);
644
+ acted = true;
645
+ break; // re-snapshot after each delivered event (shadow parity)
646
+ }
647
+ }
648
+ if (acted) continue;
649
+
650
+ // Relay terminal supervisor outcomes back into the engine.
651
+ for (const [nodeId, handle] of inFlight) {
652
+ const opState = handle?.currentState;
653
+ if (opState === 'completed' || opState === 'failed' || opState === 'cancelled') {
654
+ // eslint-disable-next-line no-await-in-loop
655
+ const res = await engine.deliver(
656
+ liveOutcomeEvent(pi, nodeId, plan.nodes.get(nodeId), seqFor(nodeId), opState, observed()),
657
+ );
658
+ if (!res.consumed) {
659
+ return refuse('event_not_consumed',
660
+ `${planId}/${nodeId}: deliver rejected (${res.code ?? 'unknown'})`);
661
+ }
662
+ for (const t of res.transitions ?? []) {
663
+ transitions.push({ node: t.node, to: t.to, ...(t.branch !== undefined ? { branch: t.branch } : {}) });
664
+ }
665
+ inFlight.delete(nodeId);
666
+ acted = true;
667
+ break;
668
+ }
669
+ }
670
+ if (acted) continue;
671
+
672
+ // Terminate honestly as soon as nothing can still produce an
673
+ // outcome: a running RUN node with neither a live op nor a pending
674
+ // launch failure is a driver fault — never spin it to the wall cap.
675
+ let pending = false;
676
+ let launchlessRun = false;
677
+ for (const [id, rec] of Object.entries(snapshot.nodes)) {
678
+ if (rec.state !== 'running') continue;
679
+ const op = plan.nodes.get(id)?.op;
680
+ if (op === 'RUN') {
681
+ if (inFlight.has(id) || launchErrors.has(id)) pending = true;
682
+ else launchlessRun = true;
683
+ } else if (op === 'WAIT_EVENT' || op === 'BRANCH') {
684
+ pending = true;
685
+ }
686
+ }
687
+ if (launchlessRun) break;
688
+ if (!pending) break;
689
+
690
+ // eslint-disable-next-line no-await-in-loop
691
+ await sleep(pollMs);
692
+ }
693
+ } catch (err) {
694
+ return refuse('engine_error', err?.code ?? String(err));
695
+ } finally {
696
+ try { await engine.close(); } catch { /* persist-best-effort; run result stands */ }
697
+ try { await supervisor.close(); } catch { /* kill+drain best-effort */ }
698
+ }
699
+
700
+ const final = pi ? await engineStateSafe(engine, pi) : null;
701
+ if (final) status = final.status;
702
+ const wallMs = Date.now() - started;
703
+
704
+ const nodeStates = {};
705
+ for (const [id, n] of Object.entries(final?.nodes ?? {})) {
706
+ nodeStates[id] = { state: n.state, attempt: n.attempt };
707
+ }
708
+
709
+ const resolved = isTerminal(status) || status === 'recovery_required';
710
+ const result = Object.freeze({
711
+ ok: true,
712
+ code: resolved ? status : 'unterminated',
713
+ planId,
714
+ planInstanceId: pi,
715
+ status: resolved ? status : 'unterminated',
716
+ transitions: Object.freeze(transitions),
717
+ nodeStates: Object.freeze(nodeStates),
718
+ wallMs,
719
+ irHash: loaded.irHash,
720
+ planVersion: loaded.planVersion,
721
+ });
722
+
723
+ // D-03 evidence append (caller may opt out with evidence:false). Live
724
+ // rows carry lane:'live'; 'unterminated' is a recorded real outcome,
725
+ // not a refuse — the evidence bar needs it counted honestly.
726
+ if (opts.evidence !== false && typeof projectRoot === 'string' && projectRoot !== '') {
727
+ const append = await appendEvidence(projectRoot, {
728
+ planId,
729
+ irHash: loaded.irHash,
730
+ planVersion: loaded.planVersion,
731
+ planInstanceId: pi,
732
+ status: result.status,
733
+ lane: 'live',
734
+ transitions: transitions.length,
735
+ wallMs,
736
+ qualityDelta: null,
737
+ wallP95Ratio: null,
738
+ costRatio: null,
739
+ // Live failures and recovery ambiguities are forbidden lanes;
740
+ // 'unterminated'/'cancelled' are bounded honest outcomes.
741
+ forbidden: result.status === 'failed' || result.status === 'recovery_required',
742
+ at: new Date().toISOString(),
743
+ }, { key: 'vm' });
744
+ if (!append.ok) {
745
+ return refuse('evidence_write_error', `${planId}: ${append.reason}`);
746
+ }
747
+ }
748
+ return result;
749
+ } catch (err) {
750
+ // Outer never-throws guard — same contract as runShadowPlan.
751
+ return refuse('engine_error', err?.code ?? String(err?.message ?? err));
752
+ }
753
+ }
754
+
755
+ // --- BL-031: replay comparison artifact ------------------------------------
756
+
757
+ /**
758
+ * The pre-registered BL-031 promotion bar, verbatim (frozen before any
759
+ * comparison exists): the live lane promotes ONLY if the VM catches ≥1
760
+ * defect the doc pipeline missed OR cuts cycle time ≥25% at equal verdicts
761
+ * — otherwise it stays shadow. `promote()` remains the only mechanism that
762
+ * can move a stage; this artifact is recorded evidence, never a write.
763
+ */
764
+ export const REPLAY_PROMOTION_BAR = Object.freeze({
765
+ minDefectsCaught: 1,
766
+ minCycleTimeCutRatio: 0.25,
767
+ });
768
+
769
+ const COMPARISON_KIND = 'replay-comparison';
770
+ const isNum = (v) => typeof v === 'number' && Number.isFinite(v);
771
+
772
+ function replayRow(input) {
773
+ const r = input && typeof input === 'object' ? input : {};
774
+ const vm = typeof r.vmVerdict === 'string' ? r.vmVerdict : 'unknown';
775
+ const doc = typeof r.docVerdict === 'string' ? r.docVerdict : 'unknown';
776
+ return {
777
+ planId: typeof r.planId === 'string' ? r.planId : 'unknown',
778
+ vmVerdict: vm,
779
+ docVerdict: doc,
780
+ vmWallMs: isNum(r.vmWallMs) ? r.vmWallMs : null,
781
+ docWallMs: isNum(r.docWallMs) ? r.docWallMs : null,
782
+ equalVerdicts: vm === doc,
783
+ // Asymmetric on purpose: the VM catching what the doc lane called good
784
+ // is evidence FOR promotion; the doc lane catching what the VM missed
785
+ // is evidence AGAINST — never counted as a VM save.
786
+ defectCaught: vm !== doc && doc === 'completed',
787
+ };
788
+ }
789
+
790
+ /**
791
+ * Pair VM replay verdicts with archived doc-pipeline outcomes and apply the
792
+ * pre-registered bar. Pure: no I/O, never throws — missing/malformed fields
793
+ * degrade to 'unknown' verdicts and null timings.
794
+ *
795
+ * Verdict order: defects caught → verdict divergence → ≥25% cycle-time cut
796
+ * (only when EVERY row carries both timings — archived runs with thin
797
+ * telemetry resolve 'insufficient_data:cycle_time', never a fabricated pass)
798
+ * → 'bar_not_met'.
799
+ *
800
+ * @param {object[]} replays [{planId, vmVerdict, docVerdict, vmWallMs?, docWallMs?}]
801
+ * @returns {object} frozen { kind, bar, rows, defectsCaught, divergedRows,
802
+ * timedRows, timedRatio, verdict, reason }
803
+ */
804
+ export function buildReplayComparison(replays = []) {
805
+ const rows = (Array.isArray(replays) ? replays : []).map(replayRow);
806
+ const defectsCaught = rows.filter((r) => r.defectCaught).length;
807
+ const divergedRows = rows.filter((r) => !r.equalVerdicts && !r.defectCaught).length;
808
+ const timedRows = rows.filter((r) => isNum(r.vmWallMs) && isNum(r.docWallMs)).length;
809
+ let timedRatio = null;
810
+ if (rows.length > 0 && timedRows === rows.length) {
811
+ const vmTotal = rows.reduce((s, r) => s + r.vmWallMs, 0);
812
+ const docTotal = rows.reduce((s, r) => s + r.docWallMs, 0);
813
+ if (docTotal > 0) timedRatio = (docTotal - vmTotal) / docTotal;
814
+ }
815
+ const bar = REPLAY_PROMOTION_BAR;
816
+ let verdict = 'no-promote';
817
+ let reason;
818
+ if (defectsCaught >= bar.minDefectsCaught) {
819
+ verdict = 'promote';
820
+ reason = `defects_caught:${defectsCaught}`;
821
+ } else if (divergedRows > 0) {
822
+ reason = `verdicts_diverged:${divergedRows}:doc_better`;
823
+ } else if (timedRatio == null) {
824
+ reason = 'insufficient_data:cycle_time';
825
+ } else if (timedRatio >= bar.minCycleTimeCutRatio) {
826
+ verdict = 'promote';
827
+ reason = `cycle_time_cut:${timedRatio.toFixed(2)}>=${bar.minCycleTimeCutRatio}`;
828
+ } else {
829
+ reason = 'bar_not_met';
830
+ }
831
+ return Object.freeze({
832
+ kind: COMPARISON_KIND,
833
+ bar,
834
+ rows: Object.freeze(rows),
835
+ defectsCaught,
836
+ divergedRows,
837
+ timedRows,
838
+ timedRatio,
839
+ verdict,
840
+ reason,
841
+ });
842
+ }
843
+
844
+ /**
845
+ * Build the comparison artifact and append it to the evidence store as one
846
+ * `kind:'replay-comparison'` record — the surface `ukit vm evidence` reads
847
+ * to report the recorded promote/no-promote verdict.
848
+ *
849
+ * @param {string} projectRoot
850
+ * @param {object[]} replays
851
+ * @param {{key?: string}} [opts]
852
+ * @returns {Promise<{ok:true, record:object, bytes:number}
853
+ * |{ok:false, code:string, reason:string}>}
854
+ */
855
+ export async function recordReplayComparison(projectRoot, replays, { key = 'vm' } = {}) {
856
+ const record = { ...buildReplayComparison(replays), at: new Date().toISOString() };
857
+ const append = await appendEvidence(projectRoot, record, { key });
858
+ if (!append.ok) return append;
859
+ return { ok: true, record: Object.freeze(record), bytes: append.bytes };
860
+ }
@@ -37,6 +37,7 @@ import { sanitizeForSupport } from '../observability/privacy/sanitizeForSupport.
37
37
  import { MAX_SUPPORT_STRING_CHARS } from '../observability/privacy/allowlist.js';
38
38
  import { redactString } from '../observability/privacy/redaction.js';
39
39
  import { scanText } from '../sensitiveValueScanner.js';
40
+ import { resolveConfigStage } from '../runtimeConfig.js';
40
41
 
41
42
  /** Monotonic nanoseconds — BigInt where available (process.hrtime). */
42
43
  export function nowNs() {
@@ -108,6 +109,10 @@ export function resolveTelemetry({ recorder, projectRoot, config, env, clock } =
108
109
  recorder: rec,
109
110
  agent_id,
110
111
  nowNs: clock && typeof clock.nowNs === 'function' ? clock.nowNs.bind(clock) : nowNs,
112
+ // TASK-C89-002 (SPEC §11): subagentOrchestrator.telemetry is opt-in —
113
+ // 'off' or absent means recordSubagentUsage is a typed no-op and the
114
+ // emit path is never touched, preserving route/prompt/Stop parity.
115
+ subagentTelemetryStage: resolveConfigStage(config, 'subagentOrchestrator.telemetry.stage'),
111
116
  };
112
117
  }
113
118
 
@@ -146,6 +151,9 @@ export function makeSpanContext(parent) {
146
151
  * floor. Never throws — a throwing recorder returns a typed drop.
147
152
  */
148
153
  export function emitSpanRecord(telemetry, ctx, semanticName, payload) {
154
+ if (telemetry == null || ctx == null || typeof telemetry.recorder?.emit !== 'function') {
155
+ return { status: 'dropped', reason: 'recorder-disabled' };
156
+ }
149
157
  try {
150
158
  return telemetry.recorder.emit({
151
159
  semantic_name: semanticName,
@@ -213,6 +221,119 @@ export function typedResource(resource) {
213
221
  return { resource: { value, source: 'ESTIMATED' }, complete: true };
214
222
  }
215
223
 
224
+ // --- TASK-C89-002 subagent usage provenance (SPEC §3 FR-01, §11) -----------
225
+ // `recordSubagentUsage` is observation-only: it emits one 'subagent.usage'
226
+ // span record describing a child run's prompt composition and usage
227
+ // provenance. It never fabricates: bytes are locally measured facts,
228
+ // tokens are deterministic ESTIMATED unless a caller types them, provider
229
+ // usage survives only via typedResource's PROVIDER rule, and unknown
230
+ // lineage/usage stays UNKNOWN/null — never 0, never a guessed id.
231
+ // The emit is gated on subagentOrchestrator.telemetry.stage (off = the
232
+ // lane never touches the recorder — zero-cost typed drop).
233
+
234
+ /** Deterministic token estimate: ceil(bytes / 4), the repo's TOKEN_CHAR_RATIO
235
+ * convention (codeintel/compiler.js). Always labeled ESTIMATED. */
236
+ export function estimateSubagentTokens(bytes) {
237
+ return Number.isFinite(bytes) && bytes >= 0 ? Math.ceil(bytes / 4) : null;
238
+ }
239
+
240
+ function sha256Hex(input) {
241
+ return crypto.createHash('sha256').update(input).digest('hex');
242
+ }
243
+
244
+ // Normalize one prompt component into a content-free metadata row:
245
+ // {ref, fingerprint, bytes, tokens{value,source}, count}. `text` is hashed
246
+ // and measured, then discarded — raw prompt bodies never enter the payload.
247
+ function componentRow(component) {
248
+ if (!isPlainObject(component)) return null;
249
+ const ref = isNonEmptyString(component.ref)
250
+ ? component.ref
251
+ : `sha256:component-${sha256Hex(JSON.stringify(component)).slice(0, 16)}`;
252
+ const text = typeof component.text === 'string' ? component.text : null;
253
+ const bytes = Number.isFinite(component.bytes) && component.bytes >= 0
254
+ ? component.bytes
255
+ : (text !== null ? Buffer.byteLength(text, 'utf8') : null);
256
+ // The fingerprint binds ref + content so a re-read of the same bytes
257
+ // dedupes while an edited same-path file cannot pretend to be the same
258
+ // component. Content hash covers text; absent text it covers bytes only.
259
+ const contentKey = text !== null ? `text:${sha256Hex(text)}` : `bytes:${bytes}`;
260
+ const fingerprint = `sha256:${sha256Hex(`${ref}\n${contentKey}`).slice(0, 24)}`;
261
+ let tokens;
262
+ if (component.tokens !== undefined) {
263
+ tokens = typedResource(component.tokens).resource;
264
+ } else if (bytes !== null) {
265
+ tokens = { value: estimateSubagentTokens(bytes), source: 'ESTIMATED' };
266
+ } else {
267
+ tokens = { value: null, source: 'UNKNOWN' };
268
+ }
269
+ return { ref, fingerprint, bytes, tokens, count: 1 };
270
+ }
271
+
272
+ // Exact dedupe by fingerprint (SPEC §3: exact dedupe only; semantic pruning
273
+ // is deferred). A repeated identical read increments `count` once per extra
274
+ // occurrence and keeps a single row.
275
+ function dedupeComponents(components) {
276
+ const rows = [];
277
+ const byFingerprint = new Map();
278
+ for (const raw of Array.isArray(components) ? components : []) {
279
+ const row = componentRow(raw);
280
+ if (row === null) continue;
281
+ const seen = byFingerprint.get(row.fingerprint);
282
+ if (seen) {
283
+ seen.count += 1;
284
+ continue;
285
+ }
286
+ byFingerprint.set(row.fingerprint, row);
287
+ rows.push(row);
288
+ }
289
+ return rows;
290
+ }
291
+
292
+ /**
293
+ * Emit one 'subagent.usage' provenance record for a child run.
294
+ *
295
+ * recordSubagentUsage(telemetry, ctx, {components, parentRunId, usage})
296
+ * → recorder emit result ({status:'accepted'|...}) or a typed drop.
297
+ *
298
+ * `ctx` should come from makeSpanContext(parent) when the parent span was
299
+ * observed — that is what links child lineage; a fabricated parent id is
300
+ * never synthesized. `parentRunId` is the caller's observed parent run
301
+ * identifier (execution_ref family); absent → parent_run_ref: null.
302
+ *
303
+ * Stage gate: subagentOrchestrator.telemetry.stage must resolve to
304
+ * shadow/canary/default; 'off'/absent/garbage returns
305
+ * {status:'dropped',reason:'stage_off'} without touching the recorder.
306
+ */
307
+ export function recordSubagentUsage(telemetry, ctx, { components, parentRunId, usage } = {}) {
308
+ if (telemetry == null) {
309
+ return { status: 'dropped', reason: 'recorder-disabled' };
310
+ }
311
+ if ((telemetry.subagentTelemetryStage ?? 'off') === 'off') {
312
+ return { status: 'dropped', reason: 'stage_off' };
313
+ }
314
+ const rows = dedupeComponents(components);
315
+ const instanceCount = rows.reduce((n, row) => n + row.count, 0);
316
+ // Aggregate token estimate is ESTIMATED only when every component's
317
+ // estimate is numeric; a single unknown component keeps the aggregate
318
+ // honestly UNKNOWN rather than silently summing a partial.
319
+ const tokens = rows.length > 0 && rows.every((row) => typeof row.tokens.value === 'number')
320
+ ? {
321
+ value: rows.reduce((n, row) => n + row.tokens.value * row.count, 0),
322
+ source: 'ESTIMATED',
323
+ }
324
+ : { value: null, source: 'UNKNOWN' };
325
+ const { resource, complete } = typedResource(usage);
326
+ return emitSpanRecord(telemetry, ctx, 'subagent.usage', {
327
+ kind: 'child-run',
328
+ parent_run_ref: isNonEmptyString(parentRunId) ? parentRunId : null,
329
+ components: rows,
330
+ count: instanceCount,
331
+ tokens,
332
+ resource,
333
+ telemetry_complete: complete,
334
+ });
335
+ }
336
+
216
337
  // --- V-06 trace excerpt for support bundles --------------------------------
217
338
  // `buildTraceExcerpt` is the read-side counterpart of emitSpanRecord: it
218
339
  // turns recorder-fed SemanticRecords into bounded, sanitized trace lines