@ngockhoale/ukit 3.3.3 → 3.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/manifests/engineConformance.yaml +17 -1
- package/manifests/hostCapabilities.yaml +68 -1
- package/manifests/platform.full.yaml +138 -0
- package/manifests/platform.user.yaml +255 -3
- package/package.json +1 -1
- package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
- package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
- package/scripts/probe/codex-capability-probe.mjs +169 -0
- package/src/cli/commands/doctor.js +168 -0
- package/src/cli/commands/indexTools.js +7 -0
- package/src/cli/commands/metrics.js +66 -2
- package/src/cli/commands/playbook.js +4 -4
- package/src/cli/commands/vm.js +49 -8
- package/src/core/agentRuntime/adapters.js +328 -27
- package/src/core/agentRuntime/artifacts.js +89 -0
- package/src/core/agentRuntime/context.js +345 -1
- package/src/core/agentRuntime/contract.js +296 -0
- package/src/core/agentRuntime/eventStore.js +176 -0
- package/src/core/agentRuntime/shadowRun.js +481 -5
- package/src/core/agentRuntime/telemetry.js +121 -0
- package/src/core/observability/emit/lifecycle.js +68 -1
- package/src/core/observability/emit/sessionBoot.js +393 -0
- package/src/core/observability/privacy/allowlist.js +10 -1
- package/src/core/observability/schema/registry.js +10 -0
- package/src/core/runtimeConfig.js +133 -0
- package/src/core/userPlaybooks.js +18 -3
- package/src/decision/registry.js +19 -0
- package/src/diagnostics/feedbackEvents.js +7 -4
- package/src/diagnostics/routeOutcomes.js +51 -6
- package/src/diagnostics/skillAccuracy.js +43 -3
- package/src/index/crossCheckMatrix.js +412 -0
- package/src/index/fixLoopEscalation.js +453 -0
- package/src/index/playbookRegistry.js +691 -0
- package/src/index/reviewPolicy.js +368 -0
- package/src/index/routeResolver.js +915 -0
- package/src/index/sessionHistoryExtractor.js +359 -0
- package/src/index/taskRouting.js +764 -581
- package/src/index/tierSelection.js +308 -0
- package/src/index/verificationMap.js +404 -0
- package/template_project/.claude/hooks/observability-emit.mjs +14 -0
- package/template_project/.claude/hooks/record-execution.mjs +19 -1
- package/template_project/.claude/hooks/skill-router.sh +691 -25
- package/template_project/.claude/hooks/verification-guard.sh +230 -1
- package/template_project/.claude/settings.json +2 -2
- package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
- package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
- package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
- package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
- package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
- package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
- package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
- package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
- package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
- package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
- package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
- package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
- package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
- package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
- package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
- package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
- package/template_project/.codex/README.md +8 -0
- package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
- package/template_project/ukit/README.md +1 -1
- package/template_project/ukit/storage/config.json +20 -0
- package/template_user/playbooks/architecture-decision.md +28 -0
- package/template_user/playbooks/autonomous-run.md +43 -0
- package/template_user/playbooks/autopilot-full.md +59 -0
- package/template_user/playbooks/autopilot-stack.md +54 -0
- package/template_user/playbooks/babysit.md +39 -0
- package/template_user/playbooks/bug-fix.md +3 -1
- package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
- package/template_user/playbooks/hillclimb.md +44 -0
- package/template_user/playbooks/investigation.md +21 -0
- package/template_user/playbooks/migration.md +21 -0
- package/template_user/playbooks/open-pr.md +48 -0
- package/template_user/playbooks/orchestrate.md +45 -0
- package/template_user/playbooks/performance.md +33 -0
- package/template_user/playbooks/prototype.md +28 -0
- package/template_user/playbooks/refactor.md +19 -0
- package/template_user/playbooks/release.md +28 -0
- package/template_user/playbooks/runtime-forensics.md +23 -0
- package/template_user/playbooks/session-pickup.md +31 -0
- package/template_user/playbooks/shipping.md +53 -0
- package/template_user/playbooks/skill-evaluation.md +48 -0
- package/template_user/playbooks/small-feature.md +20 -0
- package/template_user/playbooks/verification-map.json +153 -0
- package/template_user/playbooks/verification.md +22 -0
- package/template_user/playbooks/worktree-cleanup.md +37 -0
|
@@ -0,0 +1,565 @@
|
|
|
1
|
+
// TASK-C89-007 — subagent-orchestrator golden eval + rollout verdict
|
|
2
|
+
// (SPEC §9 FR-07, §12 truth table honesty, §14 rollout, §15 stop/replan).
|
|
3
|
+
//
|
|
4
|
+
// Pure evaluator + thin CLI. The evaluator replays frozen-corpus fixture runs
|
|
5
|
+
// for two arms — baseline (A: current handoff / B: single-strong) vs candidate
|
|
6
|
+
// (C: orchestrator variant) — and applies the pre-registered gates:
|
|
7
|
+
//
|
|
8
|
+
// 1. Per-class quality gate — computeGateComparison (evaluation.js §5
|
|
9
|
+
// order: forbidden → unknown → score floor → wall p95) over one
|
|
10
|
+
// FR-006 report aggregated per task class. Class/critical gating is
|
|
11
|
+
// expressed by aggregation, so evaluation.js needs no extension.
|
|
12
|
+
// 2. Savings provenance — computeSavings (TASK-C89-001): a percentage
|
|
13
|
+
// exists only when BOTH arms carry provider-measured usage; unknown,
|
|
14
|
+
// estimated, host-measured and local-tokenizer sources all resolve
|
|
15
|
+
// reportable:false — never 0%, never a fabricated number.
|
|
16
|
+
// 3. Independent reviewer — every review-class candidate run must carry
|
|
17
|
+
// reviewer evidence; detected:false is a critical miss (NO_GO), absent
|
|
18
|
+
// reviewer evidence is missing data (INCOMPARABLE, not a free pass).
|
|
19
|
+
// 4. VM-lane promotion — ukit vm evidence records (C87 live lane) are
|
|
20
|
+
// consumed verbatim through promote(); the printed verdict is advisory,
|
|
21
|
+
// never a write. Stage holds shadow until PROMOTION_CRITERIA and the
|
|
22
|
+
// BL-031 net-value bar are both satisfied; the operator chooses rollout.
|
|
23
|
+
//
|
|
24
|
+
// Verdicts: 'GO' | 'NO_GO' | 'INCOMPARABLE'. NO_GO beats INCOMPARABLE beats
|
|
25
|
+
// GO. A critical defect is an independent blocker — savings never overrides
|
|
26
|
+
// quality. Missing data is fail-closed: unscored runs, absent comparability
|
|
27
|
+
// coordinates and unknown usage all withhold comparison, they never guess.
|
|
28
|
+
//
|
|
29
|
+
// RESOURCE_SOURCES ↔ usageSource reconciliation (T-002→T-001 casing gap):
|
|
30
|
+
// telemetry/types.ts provenance is UPPER_SNAKE (PROVIDER, LOCAL_TOKENIZER,
|
|
31
|
+
// ESTIMATED, UNKNOWN); corpus computeSavings expects lower-kebab
|
|
32
|
+
// ('provider', 'host-measured', 'estimated', 'unknown').
|
|
33
|
+
// normalizeUsageSource() is the single mapping point — every other value,
|
|
34
|
+
// including anything unrecognized, degrades to 'unknown'.
|
|
35
|
+
|
|
36
|
+
import fs from 'node:fs/promises';
|
|
37
|
+
import os from 'node:os';
|
|
38
|
+
import path from 'node:path';
|
|
39
|
+
import { fileURLToPath } from 'node:url';
|
|
40
|
+
|
|
41
|
+
import {
|
|
42
|
+
loadSubagentCorpus,
|
|
43
|
+
checkComparability,
|
|
44
|
+
computeSavings,
|
|
45
|
+
CORPUS_REVISION,
|
|
46
|
+
BASELINE_SHA,
|
|
47
|
+
RUBRIC_VERSION,
|
|
48
|
+
ARMS,
|
|
49
|
+
} from './subagent-orchestrator-corpus.mjs';
|
|
50
|
+
import {
|
|
51
|
+
computeGateComparison,
|
|
52
|
+
COMPARABLE_RUNS,
|
|
53
|
+
QUALITY_SCORE_FLOOR,
|
|
54
|
+
} from '../../src/core/agentRuntime/evaluation.js';
|
|
55
|
+
import { UNKNOWN } from './decision-runtime-metrics.mjs';
|
|
56
|
+
import { promote, PROMOTION_STAGE_ORDER } from '../../src/core/agentRuntime/promotion.js';
|
|
57
|
+
import { readEvidence } from '../../src/core/agentRuntime/shadowRun.js';
|
|
58
|
+
import { inspectRuntimeConfig } from '../../src/core/runtimeConfig.js';
|
|
59
|
+
|
|
60
|
+
const MODULE_DIR = path.dirname(fileURLToPath(import.meta.url));
|
|
61
|
+
const REPO_ROOT = path.resolve(MODULE_DIR, '..', '..');
|
|
62
|
+
const VERDICT_DOC_REL = path.join('docs', 'AI_HANDOFF', 'benchmark', 'subagent-orchestrator-verdict.md');
|
|
63
|
+
|
|
64
|
+
export const ROLLBACK_ROUTE = 'promote(config,{rollback:true}) resolves stage "off" instantly — deterministic owners authoritative, zero VM calls';
|
|
65
|
+
function isObj(v) {
|
|
66
|
+
return v != null && typeof v === 'object' && !Array.isArray(v);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function isNum(v) {
|
|
70
|
+
return typeof v === 'number' && Number.isFinite(v);
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Map any usage-source spelling to the corpus provenance domain
|
|
75
|
+
* ('provider' | 'host-measured' | 'estimated' | 'local-tokenizer' |
|
|
76
|
+
* 'unknown'). Unrecognized spellings degrade to 'unknown' — a usage claim
|
|
77
|
+
* that cannot name its measurement cannot report savings.
|
|
78
|
+
*/
|
|
79
|
+
export function normalizeUsageSource(source) {
|
|
80
|
+
if (typeof source !== 'string' || source.trim() === '') return 'unknown';
|
|
81
|
+
const s = source.trim().toLowerCase().replace(/_/g, '-');
|
|
82
|
+
if (s === 'provider') return 'provider';
|
|
83
|
+
if (s === 'host-measured' || s === 'host') return 'host-measured';
|
|
84
|
+
if (s === 'estimated' || s === 'estimate') return 'estimated';
|
|
85
|
+
if (s === 'local-tokenizer' || s === 'tokenizer') return 'local-tokenizer';
|
|
86
|
+
if (s === 'unknown') return 'unknown';
|
|
87
|
+
return 'unknown';
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/** Usage pair for one arm: {usageSource, tokens|null}. */
|
|
91
|
+
function usageOf(run) {
|
|
92
|
+
const u = isObj(run.usage) ? run.usage : {};
|
|
93
|
+
return {
|
|
94
|
+
usageSource: normalizeUsageSource(u.source ?? run.usageSource),
|
|
95
|
+
tokens: isNum(u.tokens) ? u.tokens : (isNum(run.tokens) ? run.tokens : null),
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Aggregate an arm's usage over its scored runs. Any non-provider run
|
|
101
|
+
* forfeits provider provenance for the whole arm — the most conservative
|
|
102
|
+
* source wins, never the most flattering.
|
|
103
|
+
*/
|
|
104
|
+
function aggregateUsage(runs) {
|
|
105
|
+
const ranked = { 'provider': 4, 'host-measured': 3, 'local-tokenizer': 2, 'estimated': 1, 'unknown': 0 };
|
|
106
|
+
let source = 'provider';
|
|
107
|
+
let tokens = 0;
|
|
108
|
+
let allTokens = true;
|
|
109
|
+
for (const run of runs) {
|
|
110
|
+
const u = usageOf(run);
|
|
111
|
+
if ((ranked[u.usageSource] ?? 0) < (ranked[source] ?? 0)) source = u.usageSource;
|
|
112
|
+
if (isNum(u.tokens)) tokens += u.tokens; else allTokens = false;
|
|
113
|
+
}
|
|
114
|
+
return { usageSource: source, tokens: allTokens && runs.length > 0 ? tokens : null };
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** One run's class score: mean of finite rubric dimension scores. */
|
|
118
|
+
function runScore(run, dimensions) {
|
|
119
|
+
if (!isObj(run.scores)) return null;
|
|
120
|
+
let sum = 0;
|
|
121
|
+
let measured = false;
|
|
122
|
+
for (const dim of dimensions) {
|
|
123
|
+
const v = run.scores[dim];
|
|
124
|
+
if (isNum(v)) {
|
|
125
|
+
measured = true;
|
|
126
|
+
sum += Math.min(1, Math.max(0, v));
|
|
127
|
+
}
|
|
128
|
+
// absent/unscored dimension contributes 0 — fail-closed, never guessed
|
|
129
|
+
}
|
|
130
|
+
return measured ? sum / dimensions.length : 0;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Aggregate scored runs into the FR-006 report shape that
|
|
135
|
+
* computeGateComparison consumes ({summary, runs, env}).
|
|
136
|
+
*/
|
|
137
|
+
function toReport(runs, dimensions) {
|
|
138
|
+
let satisfied = 0;
|
|
139
|
+
let eligible = 0;
|
|
140
|
+
let forbiddenFailures = 0;
|
|
141
|
+
let unscored = 0;
|
|
142
|
+
const walls = [];
|
|
143
|
+
for (const run of runs) {
|
|
144
|
+
const score = runScore(run, dimensions);
|
|
145
|
+
if (score === null) {
|
|
146
|
+
unscored += 1;
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
satisfied += score * dimensions.length;
|
|
150
|
+
eligible += dimensions.length;
|
|
151
|
+
if (run.criticalFailure === true) forbiddenFailures += 1;
|
|
152
|
+
if (isNum(run.wallMs)) walls.push(run.wallMs);
|
|
153
|
+
}
|
|
154
|
+
walls.sort((a, b) => a - b);
|
|
155
|
+
const rank = (q) => walls[Math.max(0, Math.ceil((q / 100) * walls.length) - 1)];
|
|
156
|
+
return {
|
|
157
|
+
summary: {
|
|
158
|
+
forbiddenFailures,
|
|
159
|
+
// any unscored run poisons the class score — fail closed to UNKNOWN
|
|
160
|
+
qualityScore: unscored === 0 && eligible > 0 ? Math.min(1, satisfied / eligible) : UNKNOWN,
|
|
161
|
+
p50: walls.length > 0 ? rank(50) : UNKNOWN,
|
|
162
|
+
p95: walls.length > 0 ? rank(95) : UNKNOWN,
|
|
163
|
+
},
|
|
164
|
+
runs: runs.length,
|
|
165
|
+
unscored,
|
|
166
|
+
};
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Golden eval + rollout verdict over frozen-corpus fixture runs.
|
|
171
|
+
*
|
|
172
|
+
* @param {object} input
|
|
173
|
+
* @param {object[]} input.baselineRuns scored runs for arm A/B (baseline)
|
|
174
|
+
* @param {object[]} input.candidateRuns scored runs for the candidate arm
|
|
175
|
+
* @param {object[]} [input.fixtures] frozen corpus fixtures (defaults to
|
|
176
|
+
* loadSubagentCorpus().fixtures)
|
|
177
|
+
* @param {object} [input.config] runtime config (promotion lane stage)
|
|
178
|
+
* @param {object} [input.vmEvidence] {key, runs} — ukit vm evidence rows;
|
|
179
|
+
* replay-comparison artifacts are
|
|
180
|
+
* reported but never sampled
|
|
181
|
+
* @param {string} [input.baselineLabel] default 'A-current-handoff'
|
|
182
|
+
* @param {string} [input.candidateLabel] default 'C-no-pruning'
|
|
183
|
+
* @returns {object} frozen {verdict, perClass, reasons, savings, reviewer,
|
|
184
|
+
* promotion, replay, stage, dropped}
|
|
185
|
+
*/
|
|
186
|
+
export function evaluateSubagentOrchestrator(input = {}) {
|
|
187
|
+
const corpus = loadSubagentCorpus();
|
|
188
|
+
const fixtures = Array.isArray(input.fixtures) && input.fixtures.length > 0
|
|
189
|
+
? input.fixtures
|
|
190
|
+
: corpus.fixtures;
|
|
191
|
+
const rubrics = isObj(corpus.rubrics) ? corpus.rubrics : {};
|
|
192
|
+
const baselineRuns = Array.isArray(input.baselineRuns) ? input.baselineRuns : [];
|
|
193
|
+
const candidateRuns = Array.isArray(input.candidateRuns) ? input.candidateRuns : [];
|
|
194
|
+
|
|
195
|
+
const reasons = [];
|
|
196
|
+
let dropped = 0;
|
|
197
|
+
|
|
198
|
+
// --- stage 1: comparability — drop runs whose frozen coordinates differ --
|
|
199
|
+
function comparableOnly(runs, armLabel) {
|
|
200
|
+
const kept = [];
|
|
201
|
+
for (const run of runs) {
|
|
202
|
+
if (!isObj(run) || typeof run.fixtureId !== 'string') {
|
|
203
|
+
dropped += 1;
|
|
204
|
+
reasons.push(`dropped:${armLabel}:malformed_run`);
|
|
205
|
+
continue;
|
|
206
|
+
}
|
|
207
|
+
const check = checkComparability(run);
|
|
208
|
+
if (!check.comparable) {
|
|
209
|
+
dropped += 1;
|
|
210
|
+
reasons.push(`dropped:${armLabel}:${check.reason}`);
|
|
211
|
+
continue;
|
|
212
|
+
}
|
|
213
|
+
kept.push(run);
|
|
214
|
+
}
|
|
215
|
+
return kept;
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
const keptBaseline = comparableOnly(baselineRuns, input.baselineLabel ?? 'A-current-handoff');
|
|
219
|
+
const keptCandidate = comparableOnly(candidateRuns, input.candidateLabel ?? 'C-no-pruning');
|
|
220
|
+
|
|
221
|
+
// --- stage 2: per-class gate via computeGateComparison -------------------
|
|
222
|
+
const byClass = new Map();
|
|
223
|
+
for (const f of fixtures) {
|
|
224
|
+
const cls = typeof f.taskClass === 'string' ? f.taskClass : null;
|
|
225
|
+
if (cls && !byClass.has(cls)) byClass.set(cls, { fixtures: [], rubric: null });
|
|
226
|
+
if (cls) byClass.get(cls).fixtures.push(f);
|
|
227
|
+
}
|
|
228
|
+
for (const [cls, slot] of byClass) {
|
|
229
|
+
slot.rubric = rubrics[cls] ?? rubrics[slot.fixtures[0]?.rubric] ?? null;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
const perClass = {};
|
|
233
|
+
let anyNonInferior = false;
|
|
234
|
+
let anyUnknown = false;
|
|
235
|
+
let anyCoverageMissing = false;
|
|
236
|
+
const candidateCritical = [];
|
|
237
|
+
|
|
238
|
+
for (const [cls, slot] of byClass) {
|
|
239
|
+
const ids = new Set(slot.fixtures.map((f) => f.id));
|
|
240
|
+
const bRuns = keptBaseline.filter((r) => ids.has(r.fixtureId));
|
|
241
|
+
const cRuns = keptCandidate.filter((r) => ids.has(r.fixtureId));
|
|
242
|
+
const dimensions = slot.rubric && Array.isArray(slot.rubric.dimensions)
|
|
243
|
+
? slot.rubric.dimensions
|
|
244
|
+
: ['correctness', 'evidence', 'completeness'];
|
|
245
|
+
|
|
246
|
+
if (bRuns.length === 0 || cRuns.length === 0) {
|
|
247
|
+
anyCoverageMissing = true;
|
|
248
|
+
reasons.push(`class:${cls}:coverage_missing`);
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
const gate = computeGateComparison({
|
|
252
|
+
baseline: toReport(bRuns, dimensions),
|
|
253
|
+
variant: toReport(cRuns, dimensions),
|
|
254
|
+
scoreFloor: QUALITY_SCORE_FLOOR,
|
|
255
|
+
});
|
|
256
|
+
if (gate.verdict === 'non-inferior') anyNonInferior = true;
|
|
257
|
+
if (gate.verdict === 'unknown' || gate.comparable !== true) anyUnknown = true;
|
|
258
|
+
|
|
259
|
+
const critical = cRuns.filter((r) => r.criticalFailure === true);
|
|
260
|
+
for (const r of critical) candidateCritical.push(r.fixtureId);
|
|
261
|
+
|
|
262
|
+
perClass[cls] = {
|
|
263
|
+
fixtureCount: slot.fixtures.length,
|
|
264
|
+
baseline: { runs: bRuns.length, usage: aggregateUsage(bRuns) },
|
|
265
|
+
candidate: {
|
|
266
|
+
runs: cRuns.length,
|
|
267
|
+
unscored: toReport(cRuns, dimensions).unscored,
|
|
268
|
+
criticalFailures: critical.length,
|
|
269
|
+
usage: aggregateUsage(cRuns),
|
|
270
|
+
},
|
|
271
|
+
gate,
|
|
272
|
+
savings: computeSavings({
|
|
273
|
+
baseline: aggregateUsage(bRuns),
|
|
274
|
+
candidate: aggregateUsage(cRuns),
|
|
275
|
+
}),
|
|
276
|
+
};
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
// --- stage 3: independent reviewer (seeded-defect check) ------------------
|
|
280
|
+
const reviewFixtures = fixtures.filter((f) => f.taskClass === 'review');
|
|
281
|
+
const reviewIds = new Set(reviewFixtures.map((f) => f.id));
|
|
282
|
+
const candidateReviewRuns = keptCandidate.filter((r) => reviewIds.has(r.fixtureId));
|
|
283
|
+
const reviewerEvidence = candidateReviewRuns.filter((r) => isObj(r.reviewer));
|
|
284
|
+
const reviewerCaught = reviewerEvidence.filter((r) => r.reviewer.detected === true);
|
|
285
|
+
const reviewerMissed = reviewerEvidence.filter((r) => r.reviewer.detected === false);
|
|
286
|
+
|
|
287
|
+
const reviewer = {
|
|
288
|
+
reviewFixtures: reviewIds.size,
|
|
289
|
+
runsWithEvidence: reviewerEvidence.length,
|
|
290
|
+
detected: reviewerCaught.length,
|
|
291
|
+
missed: reviewerMissed.length,
|
|
292
|
+
caught: reviewerMissed.length === 0 && reviewerCaught.length > 0,
|
|
293
|
+
missing: reviewIds.size > 0 && reviewerEvidence.length === 0,
|
|
294
|
+
};
|
|
295
|
+
if (reviewer.missed > 0) {
|
|
296
|
+
reasons.push(`reviewer:missed_seeded_defect:${reviewer.missed}`);
|
|
297
|
+
}
|
|
298
|
+
if (reviewer.missing && candidateReviewRuns.length > 0) {
|
|
299
|
+
reasons.push('reviewer:evidence_missing');
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// --- stage 4: savings (provenance-gated, verbatim from corpus module) -----
|
|
303
|
+
const savings = computeSavings({
|
|
304
|
+
baseline: aggregateUsage(keptBaseline),
|
|
305
|
+
candidate: aggregateUsage(keptCandidate),
|
|
306
|
+
});
|
|
307
|
+
if (!savings.reportable) reasons.push(`savings:${savings.reason}`);
|
|
308
|
+
|
|
309
|
+
// --- stage 5: VM-lane promotion verdict (advisory, verbatim) --------------
|
|
310
|
+
const vmEvidence = isObj(input.vmEvidence) ? input.vmEvidence : { key: 'vm', runs: [] };
|
|
311
|
+
const vmRuns = Array.isArray(vmEvidence.runs) ? vmEvidence.runs : [];
|
|
312
|
+
const replay = [...vmRuns].reverse().find((r) => r?.kind === 'replay-comparison') ?? null;
|
|
313
|
+
const sampled = vmRuns.filter((r) => r?.kind !== 'replay-comparison');
|
|
314
|
+
const config = isObj(input.config) ? input.config : {};
|
|
315
|
+
const promotion = promote(config, {
|
|
316
|
+
key: typeof vmEvidence.key === 'string' ? vmEvidence.key : 'vm',
|
|
317
|
+
runs: sampled,
|
|
318
|
+
});
|
|
319
|
+
|
|
320
|
+
if (candidateCritical.length > 0 || reviewer.missed > 0) {
|
|
321
|
+
reasons.push(`critical_failure:blocker:${[...new Set(candidateCritical)].join(',') || 'review'}`);
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
// --- verdict: NO_GO > INCOMPARABLE > GO -----------------------------------
|
|
325
|
+
let verdict = 'GO';
|
|
326
|
+
if (anyNonInferior || candidateCritical.length > 0 || reviewer.missed > 0) {
|
|
327
|
+
verdict = 'NO_GO';
|
|
328
|
+
} else if (
|
|
329
|
+
anyUnknown || anyCoverageMissing || !savings.reportable
|
|
330
|
+
|| reviewer.missing
|
|
331
|
+
) {
|
|
332
|
+
verdict = 'INCOMPARABLE';
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
return Object.freeze({
|
|
336
|
+
verdict,
|
|
337
|
+
corpus: { revision: CORPUS_REVISION, baselineSha: BASELINE_SHA, rubricVersion: RUBRIC_VERSION, arms: ARMS },
|
|
338
|
+
baselineLabel: input.baselineLabel ?? 'A-current-handoff',
|
|
339
|
+
candidateLabel: input.candidateLabel ?? 'C-no-pruning',
|
|
340
|
+
perClass,
|
|
341
|
+
savings,
|
|
342
|
+
reviewer,
|
|
343
|
+
promotion,
|
|
344
|
+
replay,
|
|
345
|
+
sampledRuns: sampled.length,
|
|
346
|
+
comparableRunsFloor: COMPARABLE_RUNS,
|
|
347
|
+
dropped,
|
|
348
|
+
stage: {
|
|
349
|
+
unchanged: true,
|
|
350
|
+
grammar: PROMOTION_STAGE_ORDER.join(' → '),
|
|
351
|
+
rollbackRoute: ROLLBACK_ROUTE,
|
|
352
|
+
operatorChooses: true,
|
|
353
|
+
},
|
|
354
|
+
reasons: Object.freeze(reasons),
|
|
355
|
+
});
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/**
|
|
359
|
+
* Render the human-readable verdict document. Percentages appear ONLY when
|
|
360
|
+
* savings.reportable — the doc can never print an unearned cache or cost
|
|
361
|
+
* number (SPEC §3 FR-01, §12).
|
|
362
|
+
*/
|
|
363
|
+
export function formatVerdictDoc(report, { generatedAt = new Date().toISOString().slice(0, 10) } = {}) {
|
|
364
|
+
const r = report;
|
|
365
|
+
const lines = [];
|
|
366
|
+
lines.push('# Subagent Orchestrator — Golden Eval Verdict (TASK-C89-007)');
|
|
367
|
+
lines.push('');
|
|
368
|
+
lines.push(`> ${generatedAt} · corpus \`${r.corpus.revision}\` · baseline sha \`${r.corpus.baselineSha.slice(0, 8)}\` · rubric \`${r.corpus.rubricVersion}\``);
|
|
369
|
+
lines.push(`> arms: ${r.corpus.arms.map((a) => `\`${a}\``).join(', ')} — compared baseline \`${r.baselineLabel}\` vs candidate \`${r.candidateLabel}\``);
|
|
370
|
+
lines.push('');
|
|
371
|
+
lines.push(`## Verdict: **${r.verdict}**`);
|
|
372
|
+
lines.push('');
|
|
373
|
+
lines.push('Evidence-backed, not automatic promotion: the evaluator publishes a');
|
|
374
|
+
lines.push('verdict and reasons only — it never writes config or flips a stage.');
|
|
375
|
+
lines.push('');
|
|
376
|
+
|
|
377
|
+
// Per-class table
|
|
378
|
+
lines.push('## Per-class quality gate');
|
|
379
|
+
lines.push('');
|
|
380
|
+
lines.push('| class | fixtures | baseline runs | candidate runs | gate | comparable | candidate critical | savings |');
|
|
381
|
+
lines.push('|---|---|---|---|---|---|---|---|');
|
|
382
|
+
for (const [cls, c] of Object.entries(r.perClass)) {
|
|
383
|
+
const sav = c.savings.reportable ? `${c.savings.savingsPct.toFixed(1)}%` : `— (${c.savings.reason})`;
|
|
384
|
+
lines.push(`| ${cls} | ${c.fixtureCount} | ${c.baseline.runs} | ${c.candidate.runs} | ${c.gate.verdict} | ${c.gate.comparable} | ${c.candidate.criticalFailures} | ${sav} |`);
|
|
385
|
+
}
|
|
386
|
+
lines.push('');
|
|
387
|
+
|
|
388
|
+
// Savings
|
|
389
|
+
lines.push('## Savings (provenance-gated)');
|
|
390
|
+
lines.push('');
|
|
391
|
+
if (r.savings.reportable) {
|
|
392
|
+
lines.push(`Reportable savings: **${r.savings.savingsPct.toFixed(2)}%** (both arms provider-measured).`);
|
|
393
|
+
} else {
|
|
394
|
+
lines.push(`No reportable savings — \`${r.savings.reason}\`. Unknown usage is never reported as zero, never a fabricated number.`);
|
|
395
|
+
}
|
|
396
|
+
lines.push('');
|
|
397
|
+
|
|
398
|
+
// Reviewer
|
|
399
|
+
lines.push('## Independent reviewer (seeded defect)');
|
|
400
|
+
lines.push('');
|
|
401
|
+
if (r.reviewer.runsWithEvidence === 0) {
|
|
402
|
+
lines.push(`No reviewer evidence recorded (${r.reviewer.reviewFixtures} review fixture(s)) — evidence missing, not assumed.`);
|
|
403
|
+
} else {
|
|
404
|
+
lines.push(`Review runs with evidence: ${r.reviewer.runsWithEvidence}; seeded defect detected: ${r.reviewer.detected}, missed: ${r.reviewer.missed}; caught=${r.reviewer.caught}.`);
|
|
405
|
+
}
|
|
406
|
+
lines.push('');
|
|
407
|
+
|
|
408
|
+
// Promotion lane
|
|
409
|
+
lines.push('## VM-lane promotion (advisory)');
|
|
410
|
+
lines.push('');
|
|
411
|
+
lines.push(`Sampled runs: ${r.sampledRuns} (floor ${r.comparableRunsFloor}).`);
|
|
412
|
+
lines.push(`promote() verdict verbatim: stage=\`${r.promotion.stage}\` promoted=\`${r.promotion.promoted}\` reason=\`${r.promotion.reason}\`.`);
|
|
413
|
+
if (r.replay) {
|
|
414
|
+
lines.push(`Replay comparison artifact: verdict=\`${r.replay.verdict}\` reason=\`${r.replay.reason}\` defectsCaught=${r.replay.defectsCaught}.`);
|
|
415
|
+
} else {
|
|
416
|
+
lines.push('Replay comparison artifact: none recorded.');
|
|
417
|
+
}
|
|
418
|
+
lines.push('');
|
|
419
|
+
lines.push('## Stage & rollout');
|
|
420
|
+
lines.push('');
|
|
421
|
+
lines.push(`- Stage grammar: \`${r.stage.grammar}\` — unchanged by this evaluation (${r.stage.unchanged}).`);
|
|
422
|
+
lines.push(`- Rollback route: ${r.stage.rollbackRoute}.`);
|
|
423
|
+
lines.push('- The operator, not the evaluator, chooses any rollout. Live `ukit vm run` evidence at canary/default is input, not license.');
|
|
424
|
+
lines.push('');
|
|
425
|
+
lines.push('## Context & deferred work');
|
|
426
|
+
lines.push('');
|
|
427
|
+
if (r.verdict === 'INCOMPARABLE') {
|
|
428
|
+
lines.push('Verdict is INCOMPARABLE: no scored fixture runs were recorded on this host');
|
|
429
|
+
lines.push('(all lanes remain `unsupported-probe` — live in-host E2E not yet proven),');
|
|
430
|
+
lines.push('and the VM evidence store holds fewer than the COMPARABLE_RUNS floor. This is');
|
|
431
|
+
lines.push('an evidence verdict, not a failure: quality claims, savings and rollout are all');
|
|
432
|
+
lines.push('held pending real runs.');
|
|
433
|
+
} else if (r.verdict === 'NO_GO') {
|
|
434
|
+
lines.push('Verdict is NO_GO: quality/critical gates failed — see reasons above. Savings,');
|
|
435
|
+
lines.push('if any, cannot override a quality blocker.');
|
|
436
|
+
} else {
|
|
437
|
+
lines.push('Verdict is GO: all pre-registered gates passed on recorded evidence.');
|
|
438
|
+
}
|
|
439
|
+
lines.push('');
|
|
440
|
+
lines.push('SO-008/009 (budget checkpoints, small-model retrieval/extraction, adaptive');
|
|
441
|
+
lines.push('policies) remain deferred: SPEC §10 FR-08 requires good evidence from this');
|
|
442
|
+
lines.push('eval before any follow-on task set — INCOMPARABLE provides none, so no');
|
|
443
|
+
lines.push('follow-on promotion, rollback or adaptive policy work is licensed by this run.');
|
|
444
|
+
lines.push('');
|
|
445
|
+
|
|
446
|
+
// Reasons
|
|
447
|
+
lines.push('## Reasons');
|
|
448
|
+
lines.push('');
|
|
449
|
+
if (r.reasons.length === 0) {
|
|
450
|
+
lines.push('- none');
|
|
451
|
+
} else {
|
|
452
|
+
for (const reason of r.reasons) lines.push(`- ${reason}`);
|
|
453
|
+
}
|
|
454
|
+
lines.push('');
|
|
455
|
+
lines.push(`Dropped runs (non-comparable coordinates): ${r.dropped}.`);
|
|
456
|
+
lines.push('');
|
|
457
|
+
return lines.join('\n');
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
/**
|
|
461
|
+
* Compose the full evaluation: verdict + rendered document. Pure given its
|
|
462
|
+
* inputs — the caller supplies runs, config and VM evidence.
|
|
463
|
+
*/
|
|
464
|
+
export function buildEvaluation(input = {}) {
|
|
465
|
+
const report = evaluateSubagentOrchestrator(input);
|
|
466
|
+
const markdown = formatVerdictDoc(report, { generatedAt: input.generatedAt });
|
|
467
|
+
return { report, markdown };
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
// --- CLI ------------------------------------------------------------------
|
|
471
|
+
|
|
472
|
+
const USAGE = 'Usage: node scripts/bench/subagent-orchestrator-eval.mjs [options]\n'
|
|
473
|
+
+ ' --runs <file> JSON {baselineRuns:[], candidateRuns:[], config?, vmEvidence?}\n'
|
|
474
|
+
+ ' --vm-evidence <key> read .ukit/storage/agent-runtime/evidence/<key>.jsonl (default: vm)\n'
|
|
475
|
+
+ ' --write-verdict <path> write the markdown verdict doc (default: docs/AI_HANDOFF/benchmark/subagent-orchestrator-verdict.md)\n'
|
|
476
|
+
+ ' --json print the evaluation report JSON to stdout\n'
|
|
477
|
+
+ 'Exit codes: 0 = ok, 1 = usage error, 2 = crash.';
|
|
478
|
+
|
|
479
|
+
async function loadJsonFile(file) {
|
|
480
|
+
try {
|
|
481
|
+
return JSON.parse(await fs.readFile(file, 'utf8'));
|
|
482
|
+
} catch (err) {
|
|
483
|
+
return { __error: err?.message ?? String(err) };
|
|
484
|
+
}
|
|
485
|
+
}
|
|
486
|
+
|
|
487
|
+
async function main(argv) {
|
|
488
|
+
const args = argv.slice(2);
|
|
489
|
+
const flag = (name) => args.find((a) => a === name);
|
|
490
|
+
const value = (name) => {
|
|
491
|
+
const i = args.indexOf(name);
|
|
492
|
+
return i >= 0 && i + 1 < args.length ? args[i + 1] : null;
|
|
493
|
+
};
|
|
494
|
+
if (flag('--help') || flag('-h')) {
|
|
495
|
+
console.log(USAGE);
|
|
496
|
+
return 0;
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
const runsFile = value('--runs');
|
|
500
|
+
const vmKey = value('--vm-evidence') ?? 'vm';
|
|
501
|
+
const writeVerdict = value('--write-verdict');
|
|
502
|
+
const wantJson = flag('--json') != null;
|
|
503
|
+
|
|
504
|
+
let input = {};
|
|
505
|
+
if (runsFile) {
|
|
506
|
+
const parsed = await loadJsonFile(path.resolve(runsFile));
|
|
507
|
+
if (parsed.__error) {
|
|
508
|
+
console.error(`[UKit] eval: cannot parse runs file — ${parsed.__error}`);
|
|
509
|
+
return 1;
|
|
510
|
+
}
|
|
511
|
+
input = {
|
|
512
|
+
baselineRuns: parsed.baselineRuns ?? [],
|
|
513
|
+
candidateRuns: parsed.candidateRuns ?? [],
|
|
514
|
+
fixtures: parsed.fixtures,
|
|
515
|
+
config: parsed.config,
|
|
516
|
+
vmEvidence: parsed.vmEvidence,
|
|
517
|
+
baselineLabel: parsed.baselineLabel,
|
|
518
|
+
candidateLabel: parsed.candidateLabel,
|
|
519
|
+
};
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
// Consumed evidence sources (live lane where available; honest absence
|
|
523
|
+
// otherwise). readEvidence returns zero runs for a missing file — never a
|
|
524
|
+
// fabricated sample.
|
|
525
|
+
const evidence = await readEvidence(REPO_ROOT, { key: vmKey });
|
|
526
|
+
if (evidence.ok) input.vmEvidence = { key: vmKey, runs: evidence.runs };
|
|
527
|
+
|
|
528
|
+
if (input.config == null) {
|
|
529
|
+
try {
|
|
530
|
+
const inspection = await inspectRuntimeConfig(REPO_ROOT, { homeDir: os.homedir() });
|
|
531
|
+
input.config = inspection && typeof inspection.config === 'object' ? inspection.config : {};
|
|
532
|
+
} catch {
|
|
533
|
+
input.config = {};
|
|
534
|
+
}
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
const { report, markdown } = buildEvaluation(input);
|
|
538
|
+
|
|
539
|
+
const out = writeVerdict ? path.resolve(writeVerdict) : path.join(REPO_ROOT, VERDICT_DOC_REL);
|
|
540
|
+
await fs.mkdir(path.dirname(out), { recursive: true });
|
|
541
|
+
await fs.writeFile(out, markdown, 'utf8');
|
|
542
|
+
|
|
543
|
+
if (wantJson) {
|
|
544
|
+
console.log(JSON.stringify(report, null, 2));
|
|
545
|
+
} else {
|
|
546
|
+
console.log(`[UKit] subagent-orchestrator eval: verdict=${report.verdict}`);
|
|
547
|
+
console.log(` per-class gates: ${Object.values(report.perClass).map((c) => c.gate.verdict).join(', ')}`);
|
|
548
|
+
console.log(` savings: ${report.savings.reportable ? `${report.savings.savingsPct.toFixed(2)}%` : `not reportable (${report.savings.reason})`}`);
|
|
549
|
+
console.log(` promote ${vmKey}: stage=${report.promotion.stage} promoted=${report.promotion.promoted} reason=${report.promotion.reason}`);
|
|
550
|
+
console.log(` verdict doc: ${path.relative(REPO_ROOT, out)}`);
|
|
551
|
+
}
|
|
552
|
+
return 0;
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
const isMain = process.argv[1]
|
|
556
|
+
&& path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
557
|
+
|
|
558
|
+
if (isMain) {
|
|
559
|
+
main(process.argv).then((code) => {
|
|
560
|
+
process.exitCode = code;
|
|
561
|
+
}).catch((err) => {
|
|
562
|
+
console.error(`[UKit] subagent-orchestrator eval crashed: ${err?.message ?? err}`);
|
|
563
|
+
process.exitCode = 2;
|
|
564
|
+
});
|
|
565
|
+
}
|