@ngockhoale/ukit 3.3.3 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/CHANGELOG.md +40 -0
  2. package/manifests/engineConformance.yaml +17 -1
  3. package/manifests/hostCapabilities.yaml +68 -1
  4. package/manifests/platform.full.yaml +138 -0
  5. package/manifests/platform.user.yaml +255 -3
  6. package/package.json +1 -1
  7. package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
  8. package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
  9. package/scripts/probe/codex-capability-probe.mjs +169 -0
  10. package/src/cli/commands/doctor.js +168 -0
  11. package/src/cli/commands/indexTools.js +7 -0
  12. package/src/cli/commands/metrics.js +66 -2
  13. package/src/cli/commands/playbook.js +4 -4
  14. package/src/cli/commands/vm.js +49 -8
  15. package/src/core/agentRuntime/adapters.js +328 -27
  16. package/src/core/agentRuntime/artifacts.js +89 -0
  17. package/src/core/agentRuntime/context.js +345 -1
  18. package/src/core/agentRuntime/contract.js +296 -0
  19. package/src/core/agentRuntime/eventStore.js +176 -0
  20. package/src/core/agentRuntime/shadowRun.js +481 -5
  21. package/src/core/agentRuntime/telemetry.js +121 -0
  22. package/src/core/observability/emit/lifecycle.js +68 -1
  23. package/src/core/observability/emit/sessionBoot.js +393 -0
  24. package/src/core/observability/privacy/allowlist.js +10 -1
  25. package/src/core/observability/schema/registry.js +10 -0
  26. package/src/core/runtimeConfig.js +133 -0
  27. package/src/core/userPlaybooks.js +18 -3
  28. package/src/decision/registry.js +19 -0
  29. package/src/diagnostics/feedbackEvents.js +7 -4
  30. package/src/diagnostics/routeOutcomes.js +51 -6
  31. package/src/diagnostics/skillAccuracy.js +43 -3
  32. package/src/index/crossCheckMatrix.js +412 -0
  33. package/src/index/fixLoopEscalation.js +453 -0
  34. package/src/index/playbookRegistry.js +691 -0
  35. package/src/index/reviewPolicy.js +368 -0
  36. package/src/index/routeResolver.js +915 -0
  37. package/src/index/sessionHistoryExtractor.js +359 -0
  38. package/src/index/taskRouting.js +764 -581
  39. package/src/index/tierSelection.js +308 -0
  40. package/src/index/verificationMap.js +404 -0
  41. package/template_project/.claude/hooks/observability-emit.mjs +14 -0
  42. package/template_project/.claude/hooks/record-execution.mjs +19 -1
  43. package/template_project/.claude/hooks/skill-router.sh +691 -25
  44. package/template_project/.claude/hooks/verification-guard.sh +230 -1
  45. package/template_project/.claude/settings.json +2 -2
  46. package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
  47. package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
  48. package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
  49. package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
  50. package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
  51. package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
  52. package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
  53. package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
  54. package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
  55. package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
  56. package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
  57. package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
  58. package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
  59. package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
  60. package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
  61. package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
  62. package/template_project/.codex/README.md +8 -0
  63. package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
  64. package/template_project/ukit/README.md +1 -1
  65. package/template_project/ukit/storage/config.json +20 -0
  66. package/template_user/playbooks/architecture-decision.md +28 -0
  67. package/template_user/playbooks/autonomous-run.md +43 -0
  68. package/template_user/playbooks/autopilot-full.md +59 -0
  69. package/template_user/playbooks/autopilot-stack.md +54 -0
  70. package/template_user/playbooks/babysit.md +39 -0
  71. package/template_user/playbooks/bug-fix.md +3 -1
  72. package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
  73. package/template_user/playbooks/hillclimb.md +44 -0
  74. package/template_user/playbooks/investigation.md +21 -0
  75. package/template_user/playbooks/migration.md +21 -0
  76. package/template_user/playbooks/open-pr.md +48 -0
  77. package/template_user/playbooks/orchestrate.md +45 -0
  78. package/template_user/playbooks/performance.md +33 -0
  79. package/template_user/playbooks/prototype.md +28 -0
  80. package/template_user/playbooks/refactor.md +19 -0
  81. package/template_user/playbooks/release.md +28 -0
  82. package/template_user/playbooks/runtime-forensics.md +23 -0
  83. package/template_user/playbooks/session-pickup.md +31 -0
  84. package/template_user/playbooks/shipping.md +53 -0
  85. package/template_user/playbooks/skill-evaluation.md +48 -0
  86. package/template_user/playbooks/small-feature.md +20 -0
  87. package/template_user/playbooks/verification-map.json +153 -0
  88. package/template_user/playbooks/verification.md +22 -0
  89. package/template_user/playbooks/worktree-cleanup.md +37 -0
@@ -0,0 +1,565 @@
1
+ // TASK-C89-007 — subagent-orchestrator golden eval + rollout verdict
2
+ // (SPEC §9 FR-07, §12 truth table honesty, §14 rollout, §15 stop/replan).
3
+ //
4
+ // Pure evaluator + thin CLI. The evaluator replays frozen-corpus fixture runs
5
+ // for two arms — baseline (A: current handoff / B: single-strong) vs candidate
6
+ // (C: orchestrator variant) — and applies the pre-registered gates:
7
+ //
8
+ // 1. Per-class quality gate — computeGateComparison (evaluation.js §5
9
+ // order: forbidden → unknown → score floor → wall p95) over one
10
+ // FR-006 report aggregated per task class. Class/critical gating is
11
+ // expressed by aggregation, so evaluation.js needs no extension.
12
+ // 2. Savings provenance — computeSavings (TASK-C89-001): a percentage
13
+ // exists only when BOTH arms carry provider-measured usage; unknown,
14
+ // estimated, host-measured and local-tokenizer sources all resolve
15
+ // reportable:false — never 0%, never a fabricated number.
16
+ // 3. Independent reviewer — every review-class candidate run must carry
17
+ // reviewer evidence; detected:false is a critical miss (NO_GO), absent
18
+ // reviewer evidence is missing data (INCOMPARABLE, not a free pass).
19
+ // 4. VM-lane promotion — ukit vm evidence records (C87 live lane) are
20
+ // consumed verbatim through promote(); the printed verdict is advisory,
21
+ // never a write. Stage holds shadow until PROMOTION_CRITERIA and the
22
+ // BL-031 net-value bar are both satisfied; the operator chooses rollout.
23
+ //
24
+ // Verdicts: 'GO' | 'NO_GO' | 'INCOMPARABLE'. NO_GO beats INCOMPARABLE beats
25
+ // GO. A critical defect is an independent blocker — savings never overrides
26
+ // quality. Missing data is fail-closed: unscored runs, absent comparability
27
+ // coordinates and unknown usage all withhold comparison, they never guess.
28
+ //
29
+ // RESOURCE_SOURCES ↔ usageSource reconciliation (T-002→T-001 casing gap):
30
+ // telemetry/types.ts provenance is UPPER_SNAKE (PROVIDER, LOCAL_TOKENIZER,
31
+ // ESTIMATED, UNKNOWN); corpus computeSavings expects lower-kebab
32
+ // ('provider', 'host-measured', 'estimated', 'unknown').
33
+ // normalizeUsageSource() is the single mapping point — every other value,
34
+ // including anything unrecognized, degrades to 'unknown'.
35
+
36
+ import fs from 'node:fs/promises';
37
+ import os from 'node:os';
38
+ import path from 'node:path';
39
+ import { fileURLToPath } from 'node:url';
40
+
41
+ import {
42
+ loadSubagentCorpus,
43
+ checkComparability,
44
+ computeSavings,
45
+ CORPUS_REVISION,
46
+ BASELINE_SHA,
47
+ RUBRIC_VERSION,
48
+ ARMS,
49
+ } from './subagent-orchestrator-corpus.mjs';
50
+ import {
51
+ computeGateComparison,
52
+ COMPARABLE_RUNS,
53
+ QUALITY_SCORE_FLOOR,
54
+ } from '../../src/core/agentRuntime/evaluation.js';
55
+ import { UNKNOWN } from './decision-runtime-metrics.mjs';
56
+ import { promote, PROMOTION_STAGE_ORDER } from '../../src/core/agentRuntime/promotion.js';
57
+ import { readEvidence } from '../../src/core/agentRuntime/shadowRun.js';
58
+ import { inspectRuntimeConfig } from '../../src/core/runtimeConfig.js';
59
+
60
+ const MODULE_DIR = path.dirname(fileURLToPath(import.meta.url));
61
+ const REPO_ROOT = path.resolve(MODULE_DIR, '..', '..');
62
+ const VERDICT_DOC_REL = path.join('docs', 'AI_HANDOFF', 'benchmark', 'subagent-orchestrator-verdict.md');
63
+
64
+ export const ROLLBACK_ROUTE = 'promote(config,{rollback:true}) resolves stage "off" instantly — deterministic owners authoritative, zero VM calls';
65
+ function isObj(v) {
66
+ return v != null && typeof v === 'object' && !Array.isArray(v);
67
+ }
68
+
69
+ function isNum(v) {
70
+ return typeof v === 'number' && Number.isFinite(v);
71
+ }
72
+
73
+ /**
74
+ * Map any usage-source spelling to the corpus provenance domain
75
+ * ('provider' | 'host-measured' | 'estimated' | 'local-tokenizer' |
76
+ * 'unknown'). Unrecognized spellings degrade to 'unknown' — a usage claim
77
+ * that cannot name its measurement cannot report savings.
78
+ */
79
+ export function normalizeUsageSource(source) {
80
+ if (typeof source !== 'string' || source.trim() === '') return 'unknown';
81
+ const s = source.trim().toLowerCase().replace(/_/g, '-');
82
+ if (s === 'provider') return 'provider';
83
+ if (s === 'host-measured' || s === 'host') return 'host-measured';
84
+ if (s === 'estimated' || s === 'estimate') return 'estimated';
85
+ if (s === 'local-tokenizer' || s === 'tokenizer') return 'local-tokenizer';
86
+ if (s === 'unknown') return 'unknown';
87
+ return 'unknown';
88
+ }
89
+
90
+ /** Usage pair for one arm: {usageSource, tokens|null}. */
91
+ function usageOf(run) {
92
+ const u = isObj(run.usage) ? run.usage : {};
93
+ return {
94
+ usageSource: normalizeUsageSource(u.source ?? run.usageSource),
95
+ tokens: isNum(u.tokens) ? u.tokens : (isNum(run.tokens) ? run.tokens : null),
96
+ };
97
+ }
98
+
99
+ /**
100
+ * Aggregate an arm's usage over its scored runs. Any non-provider run
101
+ * forfeits provider provenance for the whole arm — the most conservative
102
+ * source wins, never the most flattering.
103
+ */
104
+ function aggregateUsage(runs) {
105
+ const ranked = { 'provider': 4, 'host-measured': 3, 'local-tokenizer': 2, 'estimated': 1, 'unknown': 0 };
106
+ let source = 'provider';
107
+ let tokens = 0;
108
+ let allTokens = true;
109
+ for (const run of runs) {
110
+ const u = usageOf(run);
111
+ if ((ranked[u.usageSource] ?? 0) < (ranked[source] ?? 0)) source = u.usageSource;
112
+ if (isNum(u.tokens)) tokens += u.tokens; else allTokens = false;
113
+ }
114
+ return { usageSource: source, tokens: allTokens && runs.length > 0 ? tokens : null };
115
+ }
116
+
117
+ /** One run's class score: mean of finite rubric dimension scores. */
118
+ function runScore(run, dimensions) {
119
+ if (!isObj(run.scores)) return null;
120
+ let sum = 0;
121
+ let measured = false;
122
+ for (const dim of dimensions) {
123
+ const v = run.scores[dim];
124
+ if (isNum(v)) {
125
+ measured = true;
126
+ sum += Math.min(1, Math.max(0, v));
127
+ }
128
+ // absent/unscored dimension contributes 0 — fail-closed, never guessed
129
+ }
130
+ return measured ? sum / dimensions.length : 0;
131
+ }
132
+
133
+ /**
134
+ * Aggregate scored runs into the FR-006 report shape that
135
+ * computeGateComparison consumes ({summary, runs, env}).
136
+ */
137
+ function toReport(runs, dimensions) {
138
+ let satisfied = 0;
139
+ let eligible = 0;
140
+ let forbiddenFailures = 0;
141
+ let unscored = 0;
142
+ const walls = [];
143
+ for (const run of runs) {
144
+ const score = runScore(run, dimensions);
145
+ if (score === null) {
146
+ unscored += 1;
147
+ continue;
148
+ }
149
+ satisfied += score * dimensions.length;
150
+ eligible += dimensions.length;
151
+ if (run.criticalFailure === true) forbiddenFailures += 1;
152
+ if (isNum(run.wallMs)) walls.push(run.wallMs);
153
+ }
154
+ walls.sort((a, b) => a - b);
155
+ const rank = (q) => walls[Math.max(0, Math.ceil((q / 100) * walls.length) - 1)];
156
+ return {
157
+ summary: {
158
+ forbiddenFailures,
159
+ // any unscored run poisons the class score — fail closed to UNKNOWN
160
+ qualityScore: unscored === 0 && eligible > 0 ? Math.min(1, satisfied / eligible) : UNKNOWN,
161
+ p50: walls.length > 0 ? rank(50) : UNKNOWN,
162
+ p95: walls.length > 0 ? rank(95) : UNKNOWN,
163
+ },
164
+ runs: runs.length,
165
+ unscored,
166
+ };
167
+ }
168
+
169
+ /**
170
+ * Golden eval + rollout verdict over frozen-corpus fixture runs.
171
+ *
172
+ * @param {object} input
173
+ * @param {object[]} input.baselineRuns scored runs for arm A/B (baseline)
174
+ * @param {object[]} input.candidateRuns scored runs for the candidate arm
175
+ * @param {object[]} [input.fixtures] frozen corpus fixtures (defaults to
176
+ * loadSubagentCorpus().fixtures)
177
+ * @param {object} [input.config] runtime config (promotion lane stage)
178
+ * @param {object} [input.vmEvidence] {key, runs} — ukit vm evidence rows;
179
+ * replay-comparison artifacts are
180
+ * reported but never sampled
181
+ * @param {string} [input.baselineLabel] default 'A-current-handoff'
182
+ * @param {string} [input.candidateLabel] default 'C-no-pruning'
183
+ * @returns {object} frozen {verdict, perClass, reasons, savings, reviewer,
184
+ * promotion, replay, stage, dropped}
185
+ */
186
+ export function evaluateSubagentOrchestrator(input = {}) {
187
+ const corpus = loadSubagentCorpus();
188
+ const fixtures = Array.isArray(input.fixtures) && input.fixtures.length > 0
189
+ ? input.fixtures
190
+ : corpus.fixtures;
191
+ const rubrics = isObj(corpus.rubrics) ? corpus.rubrics : {};
192
+ const baselineRuns = Array.isArray(input.baselineRuns) ? input.baselineRuns : [];
193
+ const candidateRuns = Array.isArray(input.candidateRuns) ? input.candidateRuns : [];
194
+
195
+ const reasons = [];
196
+ let dropped = 0;
197
+
198
+ // --- stage 1: comparability — drop runs whose frozen coordinates differ --
199
+ function comparableOnly(runs, armLabel) {
200
+ const kept = [];
201
+ for (const run of runs) {
202
+ if (!isObj(run) || typeof run.fixtureId !== 'string') {
203
+ dropped += 1;
204
+ reasons.push(`dropped:${armLabel}:malformed_run`);
205
+ continue;
206
+ }
207
+ const check = checkComparability(run);
208
+ if (!check.comparable) {
209
+ dropped += 1;
210
+ reasons.push(`dropped:${armLabel}:${check.reason}`);
211
+ continue;
212
+ }
213
+ kept.push(run);
214
+ }
215
+ return kept;
216
+ }
217
+
218
+ const keptBaseline = comparableOnly(baselineRuns, input.baselineLabel ?? 'A-current-handoff');
219
+ const keptCandidate = comparableOnly(candidateRuns, input.candidateLabel ?? 'C-no-pruning');
220
+
221
+ // --- stage 2: per-class gate via computeGateComparison -------------------
222
+ const byClass = new Map();
223
+ for (const f of fixtures) {
224
+ const cls = typeof f.taskClass === 'string' ? f.taskClass : null;
225
+ if (cls && !byClass.has(cls)) byClass.set(cls, { fixtures: [], rubric: null });
226
+ if (cls) byClass.get(cls).fixtures.push(f);
227
+ }
228
+ for (const [cls, slot] of byClass) {
229
+ slot.rubric = rubrics[cls] ?? rubrics[slot.fixtures[0]?.rubric] ?? null;
230
+ }
231
+
232
+ const perClass = {};
233
+ let anyNonInferior = false;
234
+ let anyUnknown = false;
235
+ let anyCoverageMissing = false;
236
+ const candidateCritical = [];
237
+
238
+ for (const [cls, slot] of byClass) {
239
+ const ids = new Set(slot.fixtures.map((f) => f.id));
240
+ const bRuns = keptBaseline.filter((r) => ids.has(r.fixtureId));
241
+ const cRuns = keptCandidate.filter((r) => ids.has(r.fixtureId));
242
+ const dimensions = slot.rubric && Array.isArray(slot.rubric.dimensions)
243
+ ? slot.rubric.dimensions
244
+ : ['correctness', 'evidence', 'completeness'];
245
+
246
+ if (bRuns.length === 0 || cRuns.length === 0) {
247
+ anyCoverageMissing = true;
248
+ reasons.push(`class:${cls}:coverage_missing`);
249
+ }
250
+
251
+ const gate = computeGateComparison({
252
+ baseline: toReport(bRuns, dimensions),
253
+ variant: toReport(cRuns, dimensions),
254
+ scoreFloor: QUALITY_SCORE_FLOOR,
255
+ });
256
+ if (gate.verdict === 'non-inferior') anyNonInferior = true;
257
+ if (gate.verdict === 'unknown' || gate.comparable !== true) anyUnknown = true;
258
+
259
+ const critical = cRuns.filter((r) => r.criticalFailure === true);
260
+ for (const r of critical) candidateCritical.push(r.fixtureId);
261
+
262
+ perClass[cls] = {
263
+ fixtureCount: slot.fixtures.length,
264
+ baseline: { runs: bRuns.length, usage: aggregateUsage(bRuns) },
265
+ candidate: {
266
+ runs: cRuns.length,
267
+ unscored: toReport(cRuns, dimensions).unscored,
268
+ criticalFailures: critical.length,
269
+ usage: aggregateUsage(cRuns),
270
+ },
271
+ gate,
272
+ savings: computeSavings({
273
+ baseline: aggregateUsage(bRuns),
274
+ candidate: aggregateUsage(cRuns),
275
+ }),
276
+ };
277
+ }
278
+
279
+ // --- stage 3: independent reviewer (seeded-defect check) ------------------
280
+ const reviewFixtures = fixtures.filter((f) => f.taskClass === 'review');
281
+ const reviewIds = new Set(reviewFixtures.map((f) => f.id));
282
+ const candidateReviewRuns = keptCandidate.filter((r) => reviewIds.has(r.fixtureId));
283
+ const reviewerEvidence = candidateReviewRuns.filter((r) => isObj(r.reviewer));
284
+ const reviewerCaught = reviewerEvidence.filter((r) => r.reviewer.detected === true);
285
+ const reviewerMissed = reviewerEvidence.filter((r) => r.reviewer.detected === false);
286
+
287
+ const reviewer = {
288
+ reviewFixtures: reviewIds.size,
289
+ runsWithEvidence: reviewerEvidence.length,
290
+ detected: reviewerCaught.length,
291
+ missed: reviewerMissed.length,
292
+ caught: reviewerMissed.length === 0 && reviewerCaught.length > 0,
293
+ missing: reviewIds.size > 0 && reviewerEvidence.length === 0,
294
+ };
295
+ if (reviewer.missed > 0) {
296
+ reasons.push(`reviewer:missed_seeded_defect:${reviewer.missed}`);
297
+ }
298
+ if (reviewer.missing && candidateReviewRuns.length > 0) {
299
+ reasons.push('reviewer:evidence_missing');
300
+ }
301
+
302
+ // --- stage 4: savings (provenance-gated, verbatim from corpus module) -----
303
+ const savings = computeSavings({
304
+ baseline: aggregateUsage(keptBaseline),
305
+ candidate: aggregateUsage(keptCandidate),
306
+ });
307
+ if (!savings.reportable) reasons.push(`savings:${savings.reason}`);
308
+
309
+ // --- stage 5: VM-lane promotion verdict (advisory, verbatim) --------------
310
+ const vmEvidence = isObj(input.vmEvidence) ? input.vmEvidence : { key: 'vm', runs: [] };
311
+ const vmRuns = Array.isArray(vmEvidence.runs) ? vmEvidence.runs : [];
312
+ const replay = [...vmRuns].reverse().find((r) => r?.kind === 'replay-comparison') ?? null;
313
+ const sampled = vmRuns.filter((r) => r?.kind !== 'replay-comparison');
314
+ const config = isObj(input.config) ? input.config : {};
315
+ const promotion = promote(config, {
316
+ key: typeof vmEvidence.key === 'string' ? vmEvidence.key : 'vm',
317
+ runs: sampled,
318
+ });
319
+
320
+ if (candidateCritical.length > 0 || reviewer.missed > 0) {
321
+ reasons.push(`critical_failure:blocker:${[...new Set(candidateCritical)].join(',') || 'review'}`);
322
+ }
323
+
324
+ // --- verdict: NO_GO > INCOMPARABLE > GO -----------------------------------
325
+ let verdict = 'GO';
326
+ if (anyNonInferior || candidateCritical.length > 0 || reviewer.missed > 0) {
327
+ verdict = 'NO_GO';
328
+ } else if (
329
+ anyUnknown || anyCoverageMissing || !savings.reportable
330
+ || reviewer.missing
331
+ ) {
332
+ verdict = 'INCOMPARABLE';
333
+ }
334
+
335
+ return Object.freeze({
336
+ verdict,
337
+ corpus: { revision: CORPUS_REVISION, baselineSha: BASELINE_SHA, rubricVersion: RUBRIC_VERSION, arms: ARMS },
338
+ baselineLabel: input.baselineLabel ?? 'A-current-handoff',
339
+ candidateLabel: input.candidateLabel ?? 'C-no-pruning',
340
+ perClass,
341
+ savings,
342
+ reviewer,
343
+ promotion,
344
+ replay,
345
+ sampledRuns: sampled.length,
346
+ comparableRunsFloor: COMPARABLE_RUNS,
347
+ dropped,
348
+ stage: {
349
+ unchanged: true,
350
+ grammar: PROMOTION_STAGE_ORDER.join(' → '),
351
+ rollbackRoute: ROLLBACK_ROUTE,
352
+ operatorChooses: true,
353
+ },
354
+ reasons: Object.freeze(reasons),
355
+ });
356
+ }
357
+
358
+ /**
359
+ * Render the human-readable verdict document. Percentages appear ONLY when
360
+ * savings.reportable — the doc can never print an unearned cache or cost
361
+ * number (SPEC §3 FR-01, §12).
362
+ */
363
+ export function formatVerdictDoc(report, { generatedAt = new Date().toISOString().slice(0, 10) } = {}) {
364
+ const r = report;
365
+ const lines = [];
366
+ lines.push('# Subagent Orchestrator — Golden Eval Verdict (TASK-C89-007)');
367
+ lines.push('');
368
+ lines.push(`> ${generatedAt} · corpus \`${r.corpus.revision}\` · baseline sha \`${r.corpus.baselineSha.slice(0, 8)}\` · rubric \`${r.corpus.rubricVersion}\``);
369
+ lines.push(`> arms: ${r.corpus.arms.map((a) => `\`${a}\``).join(', ')} — compared baseline \`${r.baselineLabel}\` vs candidate \`${r.candidateLabel}\``);
370
+ lines.push('');
371
+ lines.push(`## Verdict: **${r.verdict}**`);
372
+ lines.push('');
373
+ lines.push('Evidence-backed, not automatic promotion: the evaluator publishes a');
374
+ lines.push('verdict and reasons only — it never writes config or flips a stage.');
375
+ lines.push('');
376
+
377
+ // Per-class table
378
+ lines.push('## Per-class quality gate');
379
+ lines.push('');
380
+ lines.push('| class | fixtures | baseline runs | candidate runs | gate | comparable | candidate critical | savings |');
381
+ lines.push('|---|---|---|---|---|---|---|---|');
382
+ for (const [cls, c] of Object.entries(r.perClass)) {
383
+ const sav = c.savings.reportable ? `${c.savings.savingsPct.toFixed(1)}%` : `— (${c.savings.reason})`;
384
+ lines.push(`| ${cls} | ${c.fixtureCount} | ${c.baseline.runs} | ${c.candidate.runs} | ${c.gate.verdict} | ${c.gate.comparable} | ${c.candidate.criticalFailures} | ${sav} |`);
385
+ }
386
+ lines.push('');
387
+
388
+ // Savings
389
+ lines.push('## Savings (provenance-gated)');
390
+ lines.push('');
391
+ if (r.savings.reportable) {
392
+ lines.push(`Reportable savings: **${r.savings.savingsPct.toFixed(2)}%** (both arms provider-measured).`);
393
+ } else {
394
+ lines.push(`No reportable savings — \`${r.savings.reason}\`. Unknown usage is never reported as zero, never a fabricated number.`);
395
+ }
396
+ lines.push('');
397
+
398
+ // Reviewer
399
+ lines.push('## Independent reviewer (seeded defect)');
400
+ lines.push('');
401
+ if (r.reviewer.runsWithEvidence === 0) {
402
+ lines.push(`No reviewer evidence recorded (${r.reviewer.reviewFixtures} review fixture(s)) — evidence missing, not assumed.`);
403
+ } else {
404
+ lines.push(`Review runs with evidence: ${r.reviewer.runsWithEvidence}; seeded defect detected: ${r.reviewer.detected}, missed: ${r.reviewer.missed}; caught=${r.reviewer.caught}.`);
405
+ }
406
+ lines.push('');
407
+
408
+ // Promotion lane
409
+ lines.push('## VM-lane promotion (advisory)');
410
+ lines.push('');
411
+ lines.push(`Sampled runs: ${r.sampledRuns} (floor ${r.comparableRunsFloor}).`);
412
+ lines.push(`promote() verdict verbatim: stage=\`${r.promotion.stage}\` promoted=\`${r.promotion.promoted}\` reason=\`${r.promotion.reason}\`.`);
413
+ if (r.replay) {
414
+ lines.push(`Replay comparison artifact: verdict=\`${r.replay.verdict}\` reason=\`${r.replay.reason}\` defectsCaught=${r.replay.defectsCaught}.`);
415
+ } else {
416
+ lines.push('Replay comparison artifact: none recorded.');
417
+ }
418
+ lines.push('');
419
+ lines.push('## Stage & rollout');
420
+ lines.push('');
421
+ lines.push(`- Stage grammar: \`${r.stage.grammar}\` — unchanged by this evaluation (${r.stage.unchanged}).`);
422
+ lines.push(`- Rollback route: ${r.stage.rollbackRoute}.`);
423
+ lines.push('- The operator, not the evaluator, chooses any rollout. Live `ukit vm run` evidence at canary/default is input, not license.');
424
+ lines.push('');
425
+ lines.push('## Context & deferred work');
426
+ lines.push('');
427
+ if (r.verdict === 'INCOMPARABLE') {
428
+ lines.push('Verdict is INCOMPARABLE: no scored fixture runs were recorded on this host');
429
+ lines.push('(all lanes remain `unsupported-probe` — live in-host E2E not yet proven),');
430
+ lines.push('and the VM evidence store holds fewer than the COMPARABLE_RUNS floor. This is');
431
+ lines.push('an evidence verdict, not a failure: quality claims, savings and rollout are all');
432
+ lines.push('held pending real runs.');
433
+ } else if (r.verdict === 'NO_GO') {
434
+ lines.push('Verdict is NO_GO: quality/critical gates failed — see reasons above. Savings,');
435
+ lines.push('if any, cannot override a quality blocker.');
436
+ } else {
437
+ lines.push('Verdict is GO: all pre-registered gates passed on recorded evidence.');
438
+ }
439
+ lines.push('');
440
+ lines.push('SO-008/009 (budget checkpoints, small-model retrieval/extraction, adaptive');
441
+ lines.push('policies) remain deferred: SPEC §10 FR-08 requires good evidence from this');
442
+ lines.push('eval before any follow-on task set — INCOMPARABLE provides none, so no');
443
+ lines.push('follow-on promotion, rollback or adaptive policy work is licensed by this run.');
444
+ lines.push('');
445
+
446
+ // Reasons
447
+ lines.push('## Reasons');
448
+ lines.push('');
449
+ if (r.reasons.length === 0) {
450
+ lines.push('- none');
451
+ } else {
452
+ for (const reason of r.reasons) lines.push(`- ${reason}`);
453
+ }
454
+ lines.push('');
455
+ lines.push(`Dropped runs (non-comparable coordinates): ${r.dropped}.`);
456
+ lines.push('');
457
+ return lines.join('\n');
458
+ }
459
+
460
+ /**
461
+ * Compose the full evaluation: verdict + rendered document. Pure given its
462
+ * inputs — the caller supplies runs, config and VM evidence.
463
+ */
464
+ export function buildEvaluation(input = {}) {
465
+ const report = evaluateSubagentOrchestrator(input);
466
+ const markdown = formatVerdictDoc(report, { generatedAt: input.generatedAt });
467
+ return { report, markdown };
468
+ }
469
+
470
+ // --- CLI ------------------------------------------------------------------
471
+
472
+ const USAGE = 'Usage: node scripts/bench/subagent-orchestrator-eval.mjs [options]\n'
473
+ + ' --runs <file> JSON {baselineRuns:[], candidateRuns:[], config?, vmEvidence?}\n'
474
+ + ' --vm-evidence <key> read .ukit/storage/agent-runtime/evidence/<key>.jsonl (default: vm)\n'
475
+ + ' --write-verdict <path> write the markdown verdict doc (default: docs/AI_HANDOFF/benchmark/subagent-orchestrator-verdict.md)\n'
476
+ + ' --json print the evaluation report JSON to stdout\n'
477
+ + 'Exit codes: 0 = ok, 1 = usage error, 2 = crash.';
478
+
479
+ async function loadJsonFile(file) {
480
+ try {
481
+ return JSON.parse(await fs.readFile(file, 'utf8'));
482
+ } catch (err) {
483
+ return { __error: err?.message ?? String(err) };
484
+ }
485
+ }
486
+
487
+ async function main(argv) {
488
+ const args = argv.slice(2);
489
+ const flag = (name) => args.find((a) => a === name);
490
+ const value = (name) => {
491
+ const i = args.indexOf(name);
492
+ return i >= 0 && i + 1 < args.length ? args[i + 1] : null;
493
+ };
494
+ if (flag('--help') || flag('-h')) {
495
+ console.log(USAGE);
496
+ return 0;
497
+ }
498
+
499
+ const runsFile = value('--runs');
500
+ const vmKey = value('--vm-evidence') ?? 'vm';
501
+ const writeVerdict = value('--write-verdict');
502
+ const wantJson = flag('--json') != null;
503
+
504
+ let input = {};
505
+ if (runsFile) {
506
+ const parsed = await loadJsonFile(path.resolve(runsFile));
507
+ if (parsed.__error) {
508
+ console.error(`[UKit] eval: cannot parse runs file — ${parsed.__error}`);
509
+ return 1;
510
+ }
511
+ input = {
512
+ baselineRuns: parsed.baselineRuns ?? [],
513
+ candidateRuns: parsed.candidateRuns ?? [],
514
+ fixtures: parsed.fixtures,
515
+ config: parsed.config,
516
+ vmEvidence: parsed.vmEvidence,
517
+ baselineLabel: parsed.baselineLabel,
518
+ candidateLabel: parsed.candidateLabel,
519
+ };
520
+ }
521
+
522
+ // Consumed evidence sources (live lane where available; honest absence
523
+ // otherwise). readEvidence returns zero runs for a missing file — never a
524
+ // fabricated sample.
525
+ const evidence = await readEvidence(REPO_ROOT, { key: vmKey });
526
+ if (evidence.ok) input.vmEvidence = { key: vmKey, runs: evidence.runs };
527
+
528
+ if (input.config == null) {
529
+ try {
530
+ const inspection = await inspectRuntimeConfig(REPO_ROOT, { homeDir: os.homedir() });
531
+ input.config = inspection && typeof inspection.config === 'object' ? inspection.config : {};
532
+ } catch {
533
+ input.config = {};
534
+ }
535
+ }
536
+
537
+ const { report, markdown } = buildEvaluation(input);
538
+
539
+ const out = writeVerdict ? path.resolve(writeVerdict) : path.join(REPO_ROOT, VERDICT_DOC_REL);
540
+ await fs.mkdir(path.dirname(out), { recursive: true });
541
+ await fs.writeFile(out, markdown, 'utf8');
542
+
543
+ if (wantJson) {
544
+ console.log(JSON.stringify(report, null, 2));
545
+ } else {
546
+ console.log(`[UKit] subagent-orchestrator eval: verdict=${report.verdict}`);
547
+ console.log(` per-class gates: ${Object.values(report.perClass).map((c) => c.gate.verdict).join(', ')}`);
548
+ console.log(` savings: ${report.savings.reportable ? `${report.savings.savingsPct.toFixed(2)}%` : `not reportable (${report.savings.reason})`}`);
549
+ console.log(` promote ${vmKey}: stage=${report.promotion.stage} promoted=${report.promotion.promoted} reason=${report.promotion.reason}`);
550
+ console.log(` verdict doc: ${path.relative(REPO_ROOT, out)}`);
551
+ }
552
+ return 0;
553
+ }
554
+
555
+ const isMain = process.argv[1]
556
+ && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
557
+
558
+ if (isMain) {
559
+ main(process.argv).then((code) => {
560
+ process.exitCode = code;
561
+ }).catch((err) => {
562
+ console.error(`[UKit] subagent-orchestrator eval crashed: ${err?.message ?? err}`);
563
+ process.exitCode = 2;
564
+ });
565
+ }