@ngockhoale/ukit 3.0.8 → 3.0.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -1
- package/manifests/documentation.yaml +11 -0
- package/package.json +1 -1
- package/scripts/audit/decision-coverage.mjs +29 -2
- package/scripts/bench/data-foundation.mjs +52 -3
- package/scripts/bench/decision-runtime-baseline.mjs +427 -0
- package/scripts/bench/decision-runtime-metrics.mjs +67 -0
- package/scripts/bench/decision-runtime-variant.mjs +626 -0
- package/scripts/bench/memory-ablation.mjs +495 -0
- package/scripts/bench/memory-baseline.mjs +596 -0
- package/scripts/bench/memory-bench.mjs +661 -0
- package/scripts/bench/memory-canary.mjs +321 -0
- package/scripts/bench/memory-corpus.mjs +354 -0
- package/scripts/bench/memory-gate.mjs +389 -0
- package/scripts/bench/memory-metrics.mjs +179 -0
- package/scripts/bench/parallel-agents.mjs +33 -11
- package/scripts/bench/recorder-overhead.mjs +204 -0
- package/scripts/bench/sqlite-spike.mjs +451 -0
- package/scripts/measure-decision-gateway.mjs +306 -0
- package/scripts/perf/audit-perf.mjs +35 -17
- package/src/bug/triageBug.js +4 -3
- package/src/cli/commands/memory.js +357 -63
- package/src/context/detectProjectContext.js +11 -1
- package/src/core/agentRuntime/adapters.js +254 -0
- package/src/core/agentRuntime/artifacts.js +192 -0
- package/src/core/agentRuntime/completionGate.js +176 -0
- package/src/core/agentRuntime/context.js +149 -0
- package/src/core/agentRuntime/contract.js +247 -0
- package/src/core/agentRuntime/diagnostics.js +244 -0
- package/src/core/agentRuntime/evaluation.js +163 -0
- package/src/core/agentRuntime/eventStore.js +404 -0
- package/src/core/agentRuntime/liveness.js +60 -0
- package/src/core/agentRuntime/planCompiler.js +322 -0
- package/src/core/agentRuntime/promotion.js +53 -0
- package/src/core/agentRuntime/qualityComparison.js +112 -0
- package/src/core/agentRuntime/recovery.js +266 -0
- package/src/core/agentRuntime/resourcePolicy.js +78 -0
- package/src/core/agentRuntime/runtimeSupport.js +237 -0
- package/src/core/agentRuntime/supervisor.js +565 -0
- package/src/core/agentRuntime/vmEngine.js +621 -0
- package/src/core/codeintel/analogy.js +3 -2
- package/src/core/experiments/dynamicWorkflow.js +17 -2
- package/src/core/fileOps.js +21 -3
- package/src/core/memory/deltaOverlays.js +75 -30
- package/src/core/memory/learningCandidates.js +93 -48
- package/src/core/memory/memoryFlags.js +83 -0
- package/src/core/memory/memoryFreshness.js +190 -0
- package/src/core/memory/memoryHit.js +144 -0
- package/src/core/memory/migrate.js +69 -189
- package/src/core/memory/migrateMapping.js +232 -0
- package/src/core/memory/mutateMemory.js +323 -0
- package/src/core/memory/policy.js +96 -0
- package/src/core/memory/projectIdentity.js +266 -0
- package/src/core/memory/recordIndex.js +178 -0
- package/src/core/memory/recordStore.js +133 -20
- package/src/core/memory/records.js +144 -6
- package/src/core/memory/retrieval.js +259 -125
- package/src/core/memory/store.js +16 -5
- package/src/core/memory/storeBackup.js +226 -0
- package/src/core/memory/storeV2.js +63 -26
- package/src/core/memory/storeV2Loader.js +30 -12
- package/src/core/memory/userMemory.js +38 -20
- package/src/core/memory/writeClassification.js +161 -0
- package/src/core/memory/writeGuard.js +129 -0
- package/src/core/observability/adapters/hookTelemetryAdapter.js +90 -0
- package/src/core/observability/analytics/cohorts.js +148 -0
- package/src/core/observability/analytics/storeDigest.js +163 -0
- package/src/core/observability/evaluation/experimentPlan.js +95 -0
- package/src/core/observability/evaluation/findings.js +99 -0
- package/src/core/observability/evaluation/optimizationKnowledge.js +10 -1
- package/src/core/observability/evaluation/perturbation.js +273 -0
- package/src/core/observability/evaluation/replay.js +7 -1
- package/src/core/observability/evaluation/scorecard.js +23 -3
- package/src/core/observability/rollout.js +11 -7
- package/src/core/observability/schema/compatibility.js +135 -0
- package/src/core/observability/schema/registry.js +99 -0
- package/src/core/observability/schema/validate.js +7 -0
- package/src/core/observability/support/import.js +53 -9
- package/src/core/observability/support/paths.js +13 -3
- package/src/core/observability/support/projector.js +148 -12
- package/src/core/output/index.js +12 -2
- package/src/core/runtimeConfig.js +83 -0
- package/src/core/runtimePaths.js +3 -0
- package/src/core/sensitiveValueScanner.js +40 -0
- package/src/core/token/index.js +40 -3
- package/src/decision/client.js +37 -13
- package/src/decision/protocol.js +1 -1
- package/src/decision/registry.js +5 -3
- package/src/decision/runtimeDecide.js +242 -0
- package/src/decision/runtimeFilter.js +150 -0
- package/src/decision/runtimeScheduler.js +239 -0
- package/src/index/buildIndex.js +13 -12
- package/src/index/queryIndex.js +35 -14
- package/src/index/relatedTests.js +50 -8
- package/src/index/resolveContext.js +9 -4
- package/src/manifest/selectItems.js +7 -3
- package/src/render/instructionRenderer.js +17 -5
- package/template_project/.claude/ukit/index/lib/index-core.mjs +94 -39
- package/template_project/.claude/ukit/index/route-task.mjs +121 -19
- package/template_project/.claude/ukit/index/unic-decision.mjs +28 -13
- package/template_project/.claude/ukit/runtime/memory-flags.mjs +51 -0
- package/template_project/.claude/ukit/runtime/memory-freshness.mjs +155 -0
- package/template_project/.claude/ukit/runtime/memory-policy.mjs +286 -0
- package/template_project/.claude/ukit/runtime/output-compression.mjs +3 -0
- package/template_project/.claude/ukit/runtime/reinject-context.mjs +145 -14
|
@@ -0,0 +1,626 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// TASK-004 (C68 G6) — paired A/B variant runner (SPEC §2 G6-FR04/06, §4, §5).
|
|
3
|
+
//
|
|
4
|
+
// Replays the frozen docs/AI_HANDOFF/benchmark/corpus.yaml through the real
|
|
5
|
+
// G2 supervisor with the G6 policies applied — caller-wait promotion
|
|
6
|
+
// (selectWaitPolicy + DurationProfile) and the resource policy
|
|
7
|
+
// (selectResourcePolicy on every supervisor event) — then compares against a
|
|
8
|
+
// paired baseline report via computeGateComparison and emits a VariantReport:
|
|
9
|
+
// baseline report shape + { variantId, policy:{promotion,resource},
|
|
10
|
+
// comparison: GateComparison }.
|
|
11
|
+
//
|
|
12
|
+
// Both arms execute the corpus under the SAME supervisor on the same host so
|
|
13
|
+
// the comparison is truly paired; the baseline arm applies no policies
|
|
14
|
+
// (foreground wait, emit-all). A pre-existing baseline report may be supplied
|
|
15
|
+
// via --baseline but only counts when env fields pair (SPEC §5 environment).
|
|
16
|
+
//
|
|
17
|
+
// Zero external model calls: every wake/token field is 'UNKNOWN' (metric-spec
|
|
18
|
+
// §5 — no provider data source exists; chars/4 estimates are banned).
|
|
19
|
+
// Smoke runs (--runs < 9) are recorded as non-comparable per metric-spec §3.
|
|
20
|
+
//
|
|
21
|
+
// Exit codes: 0 = ran to completion (the verdict is data — a 'non-inferior'
|
|
22
|
+
// verdict means rollback stands, which is itself a valid gate outcome),
|
|
23
|
+
// 2 = malformed corpus/args or harness crash.
|
|
24
|
+
|
|
25
|
+
import { spawn, execSync } from 'node:child_process';
|
|
26
|
+
import fs from 'node:fs';
|
|
27
|
+
import os from 'node:os';
|
|
28
|
+
import path from 'node:path';
|
|
29
|
+
import { fileURLToPath } from 'node:url';
|
|
30
|
+
import yaml from 'yaml';
|
|
31
|
+
|
|
32
|
+
import { createSupervisor } from '../../src/core/agentRuntime/supervisor.js';
|
|
33
|
+
import { selectWaitPolicy } from '../../src/core/agentRuntime/promotion.js';
|
|
34
|
+
import { selectResourcePolicy } from '../../src/core/agentRuntime/resourcePolicy.js';
|
|
35
|
+
import { computeGateComparison } from '../../src/core/agentRuntime/evaluation.js';
|
|
36
|
+
import {
|
|
37
|
+
UNKNOWN,
|
|
38
|
+
computeExternalWakeRate,
|
|
39
|
+
computeQualityScore,
|
|
40
|
+
summarizeRuns,
|
|
41
|
+
} from './decision-runtime-metrics.mjs';
|
|
42
|
+
|
|
43
|
+
const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
|
|
44
|
+
const DEFAULT_RUNS = 9;
|
|
45
|
+
const DEFAULT_TIMEOUT_MS = 30_000;
|
|
46
|
+
const DEFAULT_CORPUS = path.join(REPO_ROOT, 'docs/AI_HANDOFF/benchmark/corpus.yaml');
|
|
47
|
+
const ORACLE_TYPES = ['exit-code', 'output-match', 'file-exists'];
|
|
48
|
+
const TERMINAL_TO = new Set(['completed', 'failed', 'cancelled']);
|
|
49
|
+
|
|
50
|
+
// G6-FR02 DurationProfile — frozen config, thresholds are literal values
|
|
51
|
+
// traceable to docs/AI_HANDOFF/inventory/baseline-report.json latency.p50
|
|
52
|
+
// (reinject-context p50 = 29 ms is the cheapest measured known-transition
|
|
53
|
+
// on this host; promoteAfterMs=29 ≈ the first real decision point where a
|
|
54
|
+
// caller-visible wait can be promoted to a handle). Classes map 1:1 onto
|
|
55
|
+
// corpus `class` fields: multi-step long-known promotes, everything else —
|
|
56
|
+
// including ambiguous/false-completion classes — stays foreground.
|
|
57
|
+
const DURATION_PROFILE = Object.freeze({
|
|
58
|
+
version: 1,
|
|
59
|
+
classes: Object.freeze({
|
|
60
|
+
'long-known': Object.freeze({ promoteAfterMs: 29 }),
|
|
61
|
+
'short-known': Object.freeze({ promoteAfterMs: null }),
|
|
62
|
+
'ambiguous-branch': Object.freeze({ promoteAfterMs: null }),
|
|
63
|
+
'false-completion': Object.freeze({ promoteAfterMs: null }),
|
|
64
|
+
'unsupported-host': Object.freeze({ promoteAfterMs: null }),
|
|
65
|
+
'crash-mid-operation': Object.freeze({ promoteAfterMs: null }),
|
|
66
|
+
}),
|
|
67
|
+
fallback: 'foreground',
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
const USAGE = `Usage: node scripts/bench/decision-runtime-variant.mjs [options]
|
|
71
|
+
|
|
72
|
+
Replays the DR corpus through the supervisor with the G6 promotion/resource
|
|
73
|
+
policies applied, pairs it against a baseline arm, and writes a VariantReport
|
|
74
|
+
with a pre-registered GateComparison (pass | non-inferior | unknown).
|
|
75
|
+
The verdict is data: a non-pass exit-0 report means the flags stay off.
|
|
76
|
+
|
|
77
|
+
Options:
|
|
78
|
+
--runs <n> Repetitions per scenario (default ${DEFAULT_RUNS}; <9 = smoke, non-comparable)
|
|
79
|
+
--corpus <file.yaml> Corpus path (default ${DEFAULT_CORPUS})
|
|
80
|
+
--out <file.json> Report output path (default docs/AI_HANDOFF/inventory/g6-variant-report.json)
|
|
81
|
+
--baseline <file.json> Pre-existing baseline report to pair against (must env-match, else a fresh baseline arm runs)
|
|
82
|
+
--pressure <level> Injected cpuPressure for the resource policy: normal|elevated|critical (default normal)
|
|
83
|
+
--variant-id <id> Variant identifier recorded in the report (default g6-promotion-resource)
|
|
84
|
+
--timeout <ms> Per-operation terminal wait deadline (default ${DEFAULT_TIMEOUT_MS})
|
|
85
|
+
--help Show this help
|
|
86
|
+
|
|
87
|
+
Exit codes: 0 = completed, 2 = malformed corpus/args or crash.`;
|
|
88
|
+
|
|
89
|
+
// ---------- args ----------
|
|
90
|
+
|
|
91
|
+
function parseArgs(argv) {
|
|
92
|
+
const opts = {
|
|
93
|
+
runs: DEFAULT_RUNS,
|
|
94
|
+
timeoutMs: DEFAULT_TIMEOUT_MS,
|
|
95
|
+
corpus: DEFAULT_CORPUS,
|
|
96
|
+
out: path.join(REPO_ROOT, 'docs/AI_HANDOFF/inventory/g6-variant-report.json'),
|
|
97
|
+
baseline: null,
|
|
98
|
+
pressure: 'normal',
|
|
99
|
+
variantId: 'g6-promotion-resource',
|
|
100
|
+
};
|
|
101
|
+
for (let i = 0; i < argv.length; i += 1) {
|
|
102
|
+
const arg = argv[i];
|
|
103
|
+
if (arg === '--help' || arg === '-h') return { help: true };
|
|
104
|
+
const takeValue = () => {
|
|
105
|
+
i += 1;
|
|
106
|
+
if (i >= argv.length) throw new Error(`Missing value for ${arg}`);
|
|
107
|
+
return argv[i];
|
|
108
|
+
};
|
|
109
|
+
if (arg === '--corpus') opts.corpus = takeValue();
|
|
110
|
+
else if (arg === '--out') opts.out = takeValue();
|
|
111
|
+
else if (arg === '--baseline') opts.baseline = takeValue();
|
|
112
|
+
else if (arg === '--runs') opts.runs = Number(takeValue());
|
|
113
|
+
else if (arg === '--timeout') opts.timeoutMs = Number(takeValue());
|
|
114
|
+
else if (arg === '--pressure') opts.pressure = takeValue();
|
|
115
|
+
else if (arg === '--variant-id') opts.variantId = takeValue();
|
|
116
|
+
else throw new Error(`Unknown argument: ${arg}`);
|
|
117
|
+
}
|
|
118
|
+
return opts;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
// ---------- corpus loading + validation (same contract as the baseline runner) ----------
|
|
122
|
+
|
|
123
|
+
function isNonEmptyString(value) {
|
|
124
|
+
return typeof value === 'string' && value.trim() !== '';
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function isNonNegativeInt(value) {
|
|
128
|
+
return typeof value === 'number' && Number.isInteger(value) && value >= 0;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function validateCorpus(doc) {
|
|
132
|
+
const problems = [];
|
|
133
|
+
if (doc == null || typeof doc !== 'object') return ['document is not an object'];
|
|
134
|
+
if (doc.version !== 1) problems.push('version must be 1');
|
|
135
|
+
if (!Array.isArray(doc.scenarios) || doc.scenarios.length === 0) {
|
|
136
|
+
problems.push('scenarios must be a non-empty array');
|
|
137
|
+
return problems;
|
|
138
|
+
}
|
|
139
|
+
doc.scenarios.forEach((scenario, index) => {
|
|
140
|
+
const label = `scenarios[${index}]`;
|
|
141
|
+
if (!isNonEmptyString(scenario?.id)) problems.push(`${label}.id must be a non-empty string`);
|
|
142
|
+
if (!isNonEmptyString(scenario?.class)) problems.push(`${label}.class must be a non-empty string`);
|
|
143
|
+
if (!Array.isArray(scenario?.steps) || scenario.steps.length === 0) {
|
|
144
|
+
problems.push(`${label}.steps must be a non-empty array`);
|
|
145
|
+
} else {
|
|
146
|
+
scenario.steps.forEach((step, s) => {
|
|
147
|
+
if (!isNonEmptyString(step?.run)) problems.push(`${label}.steps[${s}].run must be a non-empty string`);
|
|
148
|
+
if (step?.expectExit !== undefined && !Number.isInteger(step.expectExit)) {
|
|
149
|
+
problems.push(`${label}.steps[${s}].expectExit must be an integer`);
|
|
150
|
+
}
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
if (!ORACLE_TYPES.includes(scenario?.oracle?.type)) {
|
|
154
|
+
problems.push(`${label}.oracle.type must be one of: ${ORACLE_TYPES.join(', ')}`);
|
|
155
|
+
}
|
|
156
|
+
if (!isNonNegativeInt(scenario?.eligibleTransitions)) {
|
|
157
|
+
problems.push(`${label}.eligibleTransitions must be a non-negative integer`);
|
|
158
|
+
}
|
|
159
|
+
if (!isNonNegativeInt(scenario?.expectedExternalWakes)) {
|
|
160
|
+
problems.push(`${label}.expectedExternalWakes must be a non-negative integer`);
|
|
161
|
+
}
|
|
162
|
+
});
|
|
163
|
+
return problems;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
function loadCorpus(corpusPath) {
|
|
167
|
+
if (!fs.existsSync(corpusPath)) {
|
|
168
|
+
throw new Error(`corpus not found: ${corpusPath}`);
|
|
169
|
+
}
|
|
170
|
+
let doc;
|
|
171
|
+
try {
|
|
172
|
+
doc = yaml.parse(fs.readFileSync(corpusPath, 'utf8'));
|
|
173
|
+
} catch (error) {
|
|
174
|
+
throw new Error(`corpus YAML parse error: ${error.message}`);
|
|
175
|
+
}
|
|
176
|
+
const problems = validateCorpus(doc);
|
|
177
|
+
if (problems.length > 0) {
|
|
178
|
+
throw new Error(`malformed corpus:\n - ${problems.join('\n - ')}`);
|
|
179
|
+
}
|
|
180
|
+
return doc;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
// ---------- oracle (identical rules to the baseline runner) ----------
|
|
184
|
+
|
|
185
|
+
function checkOracle(oracle, { lastExit, output, cwd }) {
|
|
186
|
+
switch (oracle.type) {
|
|
187
|
+
case 'exit-code':
|
|
188
|
+
return lastExit === oracle.expect;
|
|
189
|
+
case 'output-match':
|
|
190
|
+
return output.includes(oracle.pattern);
|
|
191
|
+
case 'file-exists':
|
|
192
|
+
return fs.existsSync(path.resolve(cwd, oracle.path));
|
|
193
|
+
default:
|
|
194
|
+
return false;
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// ---------- supervised step replay ----------
|
|
199
|
+
|
|
200
|
+
const SHELL = process.platform === 'win32'
|
|
201
|
+
? { cmd: 'cmd.exe', args: ['/d', '/s', '/c'] }
|
|
202
|
+
: { cmd: '/bin/sh', args: ['-c'] };
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Run one corpus step as a supervised operation. Returns the process wall
|
|
206
|
+
* time and the caller-observed wait: under a promoted ('handle') wait policy
|
|
207
|
+
* the caller unblocks at promoteAfterMs while the operation finishes in the
|
|
208
|
+
* background — caller wall is the promoted threshold, never fabricated.
|
|
209
|
+
* Progress-class events flow through selectResourcePolicy; terminal/failure
|
|
210
|
+
* class events are always emitted (DR-FR09 — the policy itself enforces it).
|
|
211
|
+
*/
|
|
212
|
+
async function runStepSupervised(step, {
|
|
213
|
+
supervisor,
|
|
214
|
+
operationId,
|
|
215
|
+
cwd,
|
|
216
|
+
waitPolicy,
|
|
217
|
+
cpuPressure,
|
|
218
|
+
queueDepth,
|
|
219
|
+
timeoutMs,
|
|
220
|
+
events,
|
|
221
|
+
}) {
|
|
222
|
+
const started = Date.now();
|
|
223
|
+
const spec = {
|
|
224
|
+
operationId,
|
|
225
|
+
attempt: 1,
|
|
226
|
+
argv: [SHELL.cmd, ...SHELL.args, step.run],
|
|
227
|
+
sideEffectClass: 'read_only',
|
|
228
|
+
cwd,
|
|
229
|
+
env: process.env,
|
|
230
|
+
};
|
|
231
|
+
const launched = await supervisor.start(spec);
|
|
232
|
+
if (launched.unsupported) {
|
|
233
|
+
return { exitCode: -1, durationMs: Date.now() - started, callerWaitMs: Date.now() - started, error: `launch:${launched.code}` };
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
// Poll the journal until the terminal transition lands or the deadline
|
|
237
|
+
// hits. observe() replays the journal as it exists right now — a snapshot,
|
|
238
|
+
// so liveness comes from re-polling, never from a live tail.
|
|
239
|
+
let exitCode = null;
|
|
240
|
+
let coalesced = 0;
|
|
241
|
+
const deadline = Date.now() + timeoutMs;
|
|
242
|
+
const seenSeq = new Set();
|
|
243
|
+
|
|
244
|
+
while (Date.now() <= deadline) {
|
|
245
|
+
for await (const event of supervisor.observe(operationId)) {
|
|
246
|
+
if (event?.seq == null || seenSeq.has(event.seq)) continue;
|
|
247
|
+
seenSeq.add(event.seq);
|
|
248
|
+
const isTransition = event?.eventType === 'operation.transition';
|
|
249
|
+
const eventClass = isTransition && TERMINAL_TO.has(event?.safePayload?.to)
|
|
250
|
+
? 'terminal'
|
|
251
|
+
: 'progress';
|
|
252
|
+
const decision = selectResourcePolicy({
|
|
253
|
+
cpuPressure,
|
|
254
|
+
eventClass,
|
|
255
|
+
queueDepth: queueDepth + coalesced,
|
|
256
|
+
});
|
|
257
|
+
// Coalesced progress collapses into the pending receipt — bounded
|
|
258
|
+
// batching, never a drop. Deferred progress still lands as one event.
|
|
259
|
+
// Terminal/failure classes always emit via the policy itself (DR-FR09).
|
|
260
|
+
if (decision.action === 'emit' || eventClass === 'terminal') {
|
|
261
|
+
events.emitted += 1;
|
|
262
|
+
} else if (decision.action === 'coalesce') {
|
|
263
|
+
coalesced += 1;
|
|
264
|
+
events.coalesced += 1;
|
|
265
|
+
} else {
|
|
266
|
+
events.deferred += 1;
|
|
267
|
+
events.emitted += 1;
|
|
268
|
+
}
|
|
269
|
+
if (isTransition && TERMINAL_TO.has(event.safePayload?.to)) {
|
|
270
|
+
exitCode = event.safePayload?.exitCode ?? (event.safePayload?.to === 'completed' ? 0 : -1);
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
if (exitCode !== null) break;
|
|
274
|
+
await new Promise((resolve) => setTimeout(resolve, 5));
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
const durationMs = Date.now() - started;
|
|
278
|
+
// Caller-wait semantics (DR-FR04): a promoted caller unblocks at
|
|
279
|
+
// promoteAfterMs; the process itself still runs to terminal above.
|
|
280
|
+
const callerWaitMs = waitPolicy.mode === 'handle'
|
|
281
|
+
? Math.min(durationMs, waitPolicy.promoteAfterMs)
|
|
282
|
+
: durationMs;
|
|
283
|
+
return {
|
|
284
|
+
exitCode: exitCode ?? -1,
|
|
285
|
+
durationMs,
|
|
286
|
+
callerWaitMs,
|
|
287
|
+
error: exitCode === null ? 'terminal_timeout' : null,
|
|
288
|
+
};
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
function readOperationOutput(operationId, artifactDir) {
|
|
292
|
+
let output = '';
|
|
293
|
+
for (const stream of ['stdout', 'stderr']) {
|
|
294
|
+
const file = path.join(artifactDir, `${operationId}.${stream}.log`);
|
|
295
|
+
try {
|
|
296
|
+
output += fs.readFileSync(file, 'utf8');
|
|
297
|
+
} catch { /* absent stream — no output captured */ }
|
|
298
|
+
}
|
|
299
|
+
return output;
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// ---------- scenario replay ----------
|
|
303
|
+
|
|
304
|
+
async function replayScenario(scenario, { runs, workDir, arm, cpuPressure, timeoutMs, supervisorFactory }) {
|
|
305
|
+
const durations = [];
|
|
306
|
+
const callerWaits = [];
|
|
307
|
+
let oraclePasses = 0;
|
|
308
|
+
let forbiddenFailures = 0;
|
|
309
|
+
let reachedTransitions = 0;
|
|
310
|
+
let stepMismatches = 0;
|
|
311
|
+
let promotedRuns = 0;
|
|
312
|
+
const events = { emitted: 0, coalesced: 0, deferred: 0 };
|
|
313
|
+
|
|
314
|
+
for (let r = 0; r < runs; r += 1) {
|
|
315
|
+
const runDir = path.join(workDir, `run-${r}`);
|
|
316
|
+
const artifactDir = path.join(runDir, 'artifacts');
|
|
317
|
+
fs.mkdirSync(runDir, { recursive: true });
|
|
318
|
+
const runtimeDir = path.join(runDir, 'runtime');
|
|
319
|
+
fs.mkdirSync(runtimeDir, { recursive: true });
|
|
320
|
+
const supervisor = supervisorFactory(runtimeDir, artifactDir);
|
|
321
|
+
|
|
322
|
+
const started = Date.now();
|
|
323
|
+
let callerElapsed = 0;
|
|
324
|
+
let lastExit = -1;
|
|
325
|
+
let output = '';
|
|
326
|
+
let stepsOk = true;
|
|
327
|
+
let reached = 0;
|
|
328
|
+
let runPromoted = false;
|
|
329
|
+
const operationIds = [];
|
|
330
|
+
|
|
331
|
+
for (let s = 0; s < scenario.steps.length; s += 1) {
|
|
332
|
+
const step = scenario.steps[s];
|
|
333
|
+
const expected = step.expectExit ?? 0;
|
|
334
|
+
const waitPolicy = arm === 'variant'
|
|
335
|
+
? selectWaitPolicy(DURATION_PROFILE, { class: scenario.class }, { pressure: cpuPressure })
|
|
336
|
+
: { mode: 'foreground', reason: 'baseline' };
|
|
337
|
+
const operationId = `${scenario.id}-r${r}-s${s}`;
|
|
338
|
+
operationIds.push(operationId);
|
|
339
|
+
const result = await runStepSupervised(step, {
|
|
340
|
+
supervisor,
|
|
341
|
+
operationId,
|
|
342
|
+
cwd: runDir,
|
|
343
|
+
waitPolicy: {
|
|
344
|
+
mode: waitPolicy.mode,
|
|
345
|
+
promoteAfterMs: waitPolicy.mode === 'handle'
|
|
346
|
+
? DURATION_PROFILE.classes[scenario.class]?.promoteAfterMs ?? Infinity
|
|
347
|
+
: Infinity,
|
|
348
|
+
},
|
|
349
|
+
cpuPressure,
|
|
350
|
+
queueDepth: scenario.steps.length - s - 1,
|
|
351
|
+
timeoutMs,
|
|
352
|
+
events,
|
|
353
|
+
});
|
|
354
|
+
lastExit = result.exitCode;
|
|
355
|
+
callerElapsed += result.callerWaitMs;
|
|
356
|
+
if (waitPolicy.mode === 'handle') runPromoted = true;
|
|
357
|
+
if (result.exitCode !== expected || result.error) {
|
|
358
|
+
stepsOk = false;
|
|
359
|
+
stepMismatches += 1;
|
|
360
|
+
break;
|
|
361
|
+
}
|
|
362
|
+
reached += 1;
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
// Drain journals and close artifact streams before the oracle reads the
|
|
366
|
+
// captured output — artifact bytes flush at close().
|
|
367
|
+
await supervisor.close();
|
|
368
|
+
for (const operationId of operationIds) {
|
|
369
|
+
output += readOperationOutput(operationId, artifactDir);
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
durations.push(Date.now() - started);
|
|
373
|
+
callerWaits.push(callerElapsed);
|
|
374
|
+
if (runPromoted) promotedRuns += 1;
|
|
375
|
+
reachedTransitions += Math.min(reached, scenario.eligibleTransitions);
|
|
376
|
+
|
|
377
|
+
const oraclePass = checkOracle(scenario.oracle, { lastExit, output, cwd: runDir });
|
|
378
|
+
if (oraclePass) oraclePasses += 1;
|
|
379
|
+
if (stepsOk && !oraclePass) forbiddenFailures += 1;
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
const quality = computeQualityScore({
|
|
383
|
+
satisfied: oraclePasses,
|
|
384
|
+
eligible: runs,
|
|
385
|
+
forbiddenFailures,
|
|
386
|
+
});
|
|
387
|
+
const { p50, p95 } = summarizeRuns(callerWaits.length ? callerWaits : durations);
|
|
388
|
+
const wall = summarizeRuns(durations);
|
|
389
|
+
const coverageGap = Math.max(0, scenario.eligibleTransitions * runs - reachedTransitions);
|
|
390
|
+
|
|
391
|
+
return {
|
|
392
|
+
entry: {
|
|
393
|
+
id: scenario.id,
|
|
394
|
+
class: scenario.class,
|
|
395
|
+
verdict: quality.verdict,
|
|
396
|
+
wallMs: { n: runs, p50, p95 },
|
|
397
|
+
processWallMs: { n: wall.n, p50: wall.p50, p95: wall.p95 },
|
|
398
|
+
promotedRuns,
|
|
399
|
+
wakes: 0,
|
|
400
|
+
eligible: reachedTransitions,
|
|
401
|
+
coverageGap,
|
|
402
|
+
events,
|
|
403
|
+
quality: {
|
|
404
|
+
score: quality.score,
|
|
405
|
+
satisfied: oraclePasses,
|
|
406
|
+
eligible: runs,
|
|
407
|
+
forbiddenFailures,
|
|
408
|
+
},
|
|
409
|
+
tokens: UNKNOWN,
|
|
410
|
+
stepMismatches,
|
|
411
|
+
},
|
|
412
|
+
};
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// ---------- env (same fields as the baseline runner, metric-spec §4) ----------
|
|
416
|
+
|
|
417
|
+
function detectFsType(dir) {
|
|
418
|
+
try {
|
|
419
|
+
if (typeof fs.statfsSync === 'function') {
|
|
420
|
+
const stats = fs.statfsSync(dir);
|
|
421
|
+
if (stats && typeof stats.type === 'number') {
|
|
422
|
+
return `0x${stats.type.toString(16)}`;
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
} catch { /* fall through */ }
|
|
426
|
+
return 'local';
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
function collectEnv() {
|
|
430
|
+
const env = {
|
|
431
|
+
sha: 'unknown',
|
|
432
|
+
package: 'unknown',
|
|
433
|
+
node: process.version,
|
|
434
|
+
os: process.platform,
|
|
435
|
+
arch: process.arch,
|
|
436
|
+
fs: detectFsType(REPO_ROOT),
|
|
437
|
+
};
|
|
438
|
+
try {
|
|
439
|
+
env.sha = execSync('git rev-parse HEAD', { cwd: REPO_ROOT, encoding: 'utf8', timeout: 10_000 }).trim();
|
|
440
|
+
} catch { /* leave 'unknown' */ }
|
|
441
|
+
try {
|
|
442
|
+
env.package = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf8')).version ?? 'unknown';
|
|
443
|
+
} catch { /* leave 'unknown' */ }
|
|
444
|
+
return env;
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
// ---------- arm orchestration ----------
|
|
448
|
+
|
|
449
|
+
function buildReport({ corpus, corpusPath, scenarios, runs, extra }) {
|
|
450
|
+
const totals = scenarios.reduce((acc, sc) => ({
|
|
451
|
+
wakes: acc.wakes + sc.wakes,
|
|
452
|
+
eligible: acc.eligible + sc.eligible,
|
|
453
|
+
satisfied: acc.satisfied + sc.quality.satisfied,
|
|
454
|
+
qualityEligible: acc.qualityEligible + sc.quality.eligible,
|
|
455
|
+
forbiddenFailures: acc.forbiddenFailures + sc.quality.forbiddenFailures,
|
|
456
|
+
}), { wakes: 0, eligible: 0, satisfied: 0, qualityEligible: 0, forbiddenFailures: 0 });
|
|
457
|
+
const summaryQuality = computeQualityScore({
|
|
458
|
+
satisfied: totals.satisfied,
|
|
459
|
+
eligible: totals.qualityEligible,
|
|
460
|
+
forbiddenFailures: totals.forbiddenFailures,
|
|
461
|
+
});
|
|
462
|
+
const allP50 = scenarios.map((sc) => sc.wallMs.p50).filter((v) => typeof v === 'number');
|
|
463
|
+
const allP95 = scenarios.map((sc) => sc.wallMs.p95).filter((v) => typeof v === 'number');
|
|
464
|
+
return {
|
|
465
|
+
version: 1,
|
|
466
|
+
env: collectEnv(),
|
|
467
|
+
corpus: { path: path.resolve(corpusPath), version: corpus.version, scenarios: corpus.scenarios.length },
|
|
468
|
+
runs,
|
|
469
|
+
scenarios,
|
|
470
|
+
summary: {
|
|
471
|
+
externalWakeRate: computeExternalWakeRate({
|
|
472
|
+
externalWakes: totals.wakes,
|
|
473
|
+
eligibleTransitions: totals.eligible,
|
|
474
|
+
}),
|
|
475
|
+
qualityScore: summaryQuality.score,
|
|
476
|
+
qualityVerdict: summaryQuality.verdict,
|
|
477
|
+
forbiddenFailures: totals.forbiddenFailures,
|
|
478
|
+
p50: allP50.length ? summarizeRuns(allP50).p50 : UNKNOWN,
|
|
479
|
+
p95: allP95.length ? summarizeRuns(allP95).p95 : UNKNOWN,
|
|
480
|
+
},
|
|
481
|
+
...extra,
|
|
482
|
+
};
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
async function runArm(corpus, { runs, arm, cpuPressure, timeoutMs, tmpRoot }) {
|
|
486
|
+
const supervisorFactory = (runtimeDir, artifactDir) => createSupervisor({
|
|
487
|
+
runtimeDir,
|
|
488
|
+
artifactDir,
|
|
489
|
+
config: { enabled: true },
|
|
490
|
+
});
|
|
491
|
+
const scenarios = [];
|
|
492
|
+
for (const scenario of corpus.scenarios) {
|
|
493
|
+
const workDir = path.join(tmpRoot, arm, scenario.id);
|
|
494
|
+
fs.mkdirSync(workDir, { recursive: true });
|
|
495
|
+
const { entry } = await replayScenario(scenario, {
|
|
496
|
+
runs,
|
|
497
|
+
workDir,
|
|
498
|
+
arm,
|
|
499
|
+
cpuPressure: arm === 'variant' ? cpuPressure : 'normal',
|
|
500
|
+
timeoutMs,
|
|
501
|
+
supervisorFactory,
|
|
502
|
+
});
|
|
503
|
+
scenarios.push(entry);
|
|
504
|
+
}
|
|
505
|
+
return scenarios;
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
function loadBaselineReport(file) {
|
|
509
|
+
try {
|
|
510
|
+
return JSON.parse(fs.readFileSync(path.resolve(file), 'utf8'));
|
|
511
|
+
} catch {
|
|
512
|
+
return null;
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
// ---------- main ----------
|
|
517
|
+
|
|
518
|
+
async function main() {
|
|
519
|
+
let opts;
|
|
520
|
+
try {
|
|
521
|
+
opts = parseArgs(process.argv.slice(2));
|
|
522
|
+
} catch (error) {
|
|
523
|
+
console.error(`[decision-runtime-variant] ${error.message}\n\n${USAGE}`);
|
|
524
|
+
process.exit(2);
|
|
525
|
+
}
|
|
526
|
+
if (opts.help) {
|
|
527
|
+
console.log(USAGE);
|
|
528
|
+
process.exit(0);
|
|
529
|
+
}
|
|
530
|
+
if (!Number.isInteger(opts.runs) || opts.runs < 1) {
|
|
531
|
+
console.error('[decision-runtime-variant] --runs must be a positive integer.');
|
|
532
|
+
process.exit(2);
|
|
533
|
+
}
|
|
534
|
+
if (!['normal', 'elevated', 'critical'].includes(opts.pressure)) {
|
|
535
|
+
console.error('[decision-runtime-variant] --pressure must be normal|elevated|critical.');
|
|
536
|
+
process.exit(2);
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
let corpus;
|
|
540
|
+
try {
|
|
541
|
+
corpus = loadCorpus(path.resolve(opts.corpus));
|
|
542
|
+
} catch (error) {
|
|
543
|
+
console.error(`[decision-runtime-variant] ${error.message}`);
|
|
544
|
+
process.exit(2);
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
const comparable = opts.runs >= 9;
|
|
548
|
+
const tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'ukit-dr-variant-'));
|
|
549
|
+
|
|
550
|
+
// Baseline arm: supplied report if present (env pairing is enforced by the
|
|
551
|
+
// gate), otherwise a fresh same-host arm through the supervisor with no
|
|
552
|
+
// policies — foreground wait, emit-all resource behavior.
|
|
553
|
+
let baseline = opts.baseline ? loadBaselineReport(opts.baseline) : null;
|
|
554
|
+
if (baseline == null) {
|
|
555
|
+
const baselineScenarios = await runArm(corpus, {
|
|
556
|
+
runs: opts.runs,
|
|
557
|
+
arm: 'baseline',
|
|
558
|
+
cpuPressure: 'normal',
|
|
559
|
+
timeoutMs: opts.timeoutMs,
|
|
560
|
+
tmpRoot,
|
|
561
|
+
});
|
|
562
|
+
baseline = buildReport({
|
|
563
|
+
corpus,
|
|
564
|
+
corpusPath: opts.corpus,
|
|
565
|
+
scenarios: baselineScenarios,
|
|
566
|
+
runs: opts.runs,
|
|
567
|
+
extra: { arm: 'baseline', policy: null },
|
|
568
|
+
});
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
// Variant arm: same corpus, same supervisor, G6 policies applied.
|
|
572
|
+
const variantScenarios = await runArm(corpus, {
|
|
573
|
+
runs: opts.runs,
|
|
574
|
+
arm: 'variant',
|
|
575
|
+
cpuPressure: opts.pressure,
|
|
576
|
+
timeoutMs: opts.timeoutMs,
|
|
577
|
+
tmpRoot,
|
|
578
|
+
});
|
|
579
|
+
|
|
580
|
+
const variantReport = buildReport({
|
|
581
|
+
corpus,
|
|
582
|
+
corpusPath: opts.corpus,
|
|
583
|
+
scenarios: variantScenarios,
|
|
584
|
+
runs: opts.runs,
|
|
585
|
+
extra: {
|
|
586
|
+
arm: 'variant',
|
|
587
|
+
variantId: opts.variantId,
|
|
588
|
+
policy: {
|
|
589
|
+
promotion: {
|
|
590
|
+
enabled: true,
|
|
591
|
+
profile: DURATION_PROFILE,
|
|
592
|
+
},
|
|
593
|
+
resource: {
|
|
594
|
+
enabled: true,
|
|
595
|
+
cpuPressure: opts.pressure,
|
|
596
|
+
},
|
|
597
|
+
},
|
|
598
|
+
comparable,
|
|
599
|
+
},
|
|
600
|
+
});
|
|
601
|
+
|
|
602
|
+
variantReport.comparison = computeGateComparison({ baseline, variant: variantReport });
|
|
603
|
+
if (!comparable) {
|
|
604
|
+
variantReport.comparabilityNote = 'smoke run: --runs < 9 is dev-only and non-comparable (metric-spec §3)';
|
|
605
|
+
}
|
|
606
|
+
|
|
607
|
+
const outPath = path.resolve(opts.out);
|
|
608
|
+
fs.mkdirSync(path.dirname(outPath), { recursive: true });
|
|
609
|
+
fs.writeFileSync(outPath, `${JSON.stringify(variantReport, null, 2)}\n`);
|
|
610
|
+
|
|
611
|
+
try {
|
|
612
|
+
fs.rmSync(tmpRoot, { recursive: true, force: true });
|
|
613
|
+
} catch { /* tmp cleanup is best-effort */ }
|
|
614
|
+
|
|
615
|
+
const c = variantReport.comparison;
|
|
616
|
+
console.log(`[decision-runtime-variant] report written: ${outPath}`);
|
|
617
|
+
console.log(`[decision-runtime-variant] scenarios=${variantScenarios.length} runs=${opts.runs} `
|
|
618
|
+
+ `verdict=${c.verdict} comparable=${c.comparable} `
|
|
619
|
+
+ `qualityDelta=${c.qualityDelta} wallP95Delta=${c.wallP95Delta}`);
|
|
620
|
+
process.exit(0);
|
|
621
|
+
}
|
|
622
|
+
|
|
623
|
+
main().catch((error) => {
|
|
624
|
+
console.error(`[decision-runtime-variant] harness crash: ${error?.stack ?? error}`);
|
|
625
|
+
process.exit(2);
|
|
626
|
+
});
|