@ngockhoale/ukit 3.3.3 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/manifests/engineConformance.yaml +17 -1
- package/manifests/hostCapabilities.yaml +68 -1
- package/manifests/platform.full.yaml +138 -0
- package/manifests/platform.user.yaml +255 -3
- package/package.json +1 -1
- package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
- package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
- package/scripts/probe/codex-capability-probe.mjs +169 -0
- package/src/cli/commands/doctor.js +168 -0
- package/src/cli/commands/indexTools.js +7 -0
- package/src/cli/commands/metrics.js +66 -2
- package/src/cli/commands/playbook.js +4 -4
- package/src/cli/commands/vm.js +49 -8
- package/src/core/agentRuntime/adapters.js +328 -27
- package/src/core/agentRuntime/artifacts.js +89 -0
- package/src/core/agentRuntime/context.js +345 -1
- package/src/core/agentRuntime/contract.js +296 -0
- package/src/core/agentRuntime/eventStore.js +176 -0
- package/src/core/agentRuntime/shadowRun.js +481 -5
- package/src/core/agentRuntime/telemetry.js +121 -0
- package/src/core/observability/emit/lifecycle.js +68 -1
- package/src/core/observability/emit/sessionBoot.js +393 -0
- package/src/core/observability/privacy/allowlist.js +10 -1
- package/src/core/observability/schema/registry.js +10 -0
- package/src/core/runtimeConfig.js +133 -0
- package/src/core/userPlaybooks.js +18 -3
- package/src/decision/registry.js +19 -0
- package/src/diagnostics/feedbackEvents.js +7 -4
- package/src/diagnostics/routeOutcomes.js +51 -6
- package/src/diagnostics/skillAccuracy.js +43 -3
- package/src/index/crossCheckMatrix.js +412 -0
- package/src/index/fixLoopEscalation.js +453 -0
- package/src/index/playbookRegistry.js +691 -0
- package/src/index/reviewPolicy.js +368 -0
- package/src/index/routeResolver.js +915 -0
- package/src/index/sessionHistoryExtractor.js +359 -0
- package/src/index/taskRouting.js +764 -581
- package/src/index/tierSelection.js +308 -0
- package/src/index/verificationMap.js +404 -0
- package/template_project/.claude/hooks/observability-emit.mjs +14 -0
- package/template_project/.claude/hooks/record-execution.mjs +19 -1
- package/template_project/.claude/hooks/skill-router.sh +691 -25
- package/template_project/.claude/hooks/verification-guard.sh +230 -1
- package/template_project/.claude/settings.json +2 -2
- package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
- package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
- package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
- package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
- package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
- package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
- package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
- package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
- package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
- package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
- package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
- package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
- package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
- package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
- package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
- package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
- package/template_project/.codex/README.md +8 -0
- package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
- package/template_project/ukit/README.md +1 -1
- package/template_project/ukit/storage/config.json +20 -0
- package/template_user/playbooks/architecture-decision.md +28 -0
- package/template_user/playbooks/autonomous-run.md +43 -0
- package/template_user/playbooks/autopilot-full.md +59 -0
- package/template_user/playbooks/autopilot-stack.md +54 -0
- package/template_user/playbooks/babysit.md +39 -0
- package/template_user/playbooks/bug-fix.md +3 -1
- package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
- package/template_user/playbooks/hillclimb.md +44 -0
- package/template_user/playbooks/investigation.md +21 -0
- package/template_user/playbooks/migration.md +21 -0
- package/template_user/playbooks/open-pr.md +48 -0
- package/template_user/playbooks/orchestrate.md +45 -0
- package/template_user/playbooks/performance.md +33 -0
- package/template_user/playbooks/prototype.md +28 -0
- package/template_user/playbooks/refactor.md +19 -0
- package/template_user/playbooks/release.md +28 -0
- package/template_user/playbooks/runtime-forensics.md +23 -0
- package/template_user/playbooks/session-pickup.md +31 -0
- package/template_user/playbooks/shipping.md +53 -0
- package/template_user/playbooks/skill-evaluation.md +48 -0
- package/template_user/playbooks/small-feature.md +20 -0
- package/template_user/playbooks/verification-map.json +153 -0
- package/template_user/playbooks/verification.md +22 -0
- package/template_user/playbooks/worktree-cleanup.md +37 -0
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// TASK-C89-001 — frozen subagent-orchestrator golden corpus (SPEC §3 FR-01,
|
|
3
|
+
// §9 FR-07, §12 comparability contract).
|
|
4
|
+
//
|
|
5
|
+
// This module is the machine-checkable half of the baseline freeze: a
|
|
6
|
+
// deterministic, immutable fixture manifest consumed by TASK-C89-007's eval
|
|
7
|
+
// harness. Fixture metadata only — no task bodies, no runtime behavior.
|
|
8
|
+
//
|
|
9
|
+
// loadSubagentCorpus() → {revision, baselineSha, rubricVersion, arms,
|
|
10
|
+
// fixtures[], rubrics}
|
|
11
|
+
// checkComparability(run) → {comparable:boolean, reason:string|null}
|
|
12
|
+
// computeSavings({baseline, candidate}) → {reportable:boolean, savingsPct,
|
|
13
|
+
// reason}
|
|
14
|
+
//
|
|
15
|
+
// Frozen constants:
|
|
16
|
+
// CORPUS_REVISION — bump on ANY fixture/rubric/arm change; a run recorded
|
|
17
|
+
// against an older revision is non-comparable (stale_corpus_revision).
|
|
18
|
+
// BASELINE_SHA — git SHA the corpus was frozen against; a run on a
|
|
19
|
+
// different tree is non-comparable (stale_baseline_sha).
|
|
20
|
+
// RUBRIC_VERSION — scoring rubric version; absent/mismatched →
|
|
21
|
+
// missing_rubric_version.
|
|
22
|
+
//
|
|
23
|
+
// Provenance rule (SPEC FR-01): savings may be reported only when BOTH runs
|
|
24
|
+
// carry provider-measured usage. 'estimated'/'host-measured'/'unknown' sources
|
|
25
|
+
// yield savingsPct: null — never 0%, never a fabricated number.
|
|
26
|
+
|
|
27
|
+
import path from 'node:path';
|
|
28
|
+
import { fileURLToPath } from 'node:url';
|
|
29
|
+
|
|
30
|
+
export const CORPUS_REVISION = 'so-corpus-r1';
|
|
31
|
+
export const RUBRIC_VERSION = 'so-rubric-v1';
|
|
32
|
+
// Git SHA this corpus was frozen against (cycle C89 worktree HEAD at freeze).
|
|
33
|
+
export const BASELINE_SHA = 'd86c0be7e9974415c28a212a48ac4247c26d1a0c';
|
|
34
|
+
|
|
35
|
+
// Frozen arms (SPEC §9): the eval compares these — never invented per run.
|
|
36
|
+
export const ARMS = Object.freeze([
|
|
37
|
+
'A-current-handoff', // today's inline handoff/orchestration behavior
|
|
38
|
+
'B-single-strong', // one strong agent, no delegation
|
|
39
|
+
'C-no-pruning', // candidate path with context pruning disabled
|
|
40
|
+
]);
|
|
41
|
+
|
|
42
|
+
// Per-class scoring rubric. Every rubric scores the same three axes
|
|
43
|
+
// (SPEC §9: correctness / evidence / completeness) and a critical defect is
|
|
44
|
+
// an independent blocker — savings never overrides quality.
|
|
45
|
+
const RUBRICS = Object.freeze({
|
|
46
|
+
trivial: Object.freeze({
|
|
47
|
+
dimensions: Object.freeze(['correctness', 'evidence', 'completeness']),
|
|
48
|
+
criticalFailureBlocker: true,
|
|
49
|
+
notes: 'deterministic fast path — delegate=false is the expected route',
|
|
50
|
+
}),
|
|
51
|
+
retrieval: Object.freeze({
|
|
52
|
+
dimensions: Object.freeze(['correctness', 'evidence', 'completeness']),
|
|
53
|
+
criticalFailureBlocker: true,
|
|
54
|
+
notes: 'evidence refs must resolve to the real source span',
|
|
55
|
+
}),
|
|
56
|
+
debug: Object.freeze({
|
|
57
|
+
dimensions: Object.freeze(['correctness', 'evidence', 'completeness']),
|
|
58
|
+
criticalFailureBlocker: true,
|
|
59
|
+
notes: 'root cause must be evidenced, not symptom suppression',
|
|
60
|
+
}),
|
|
61
|
+
review: Object.freeze({
|
|
62
|
+
dimensions: Object.freeze(['correctness', 'evidence', 'completeness']),
|
|
63
|
+
criticalFailureBlocker: true,
|
|
64
|
+
notes: 'seeded defect must be found; missing it is a critical failure',
|
|
65
|
+
}),
|
|
66
|
+
continuation: Object.freeze({
|
|
67
|
+
dimensions: Object.freeze(['correctness', 'evidence', 'completeness']),
|
|
68
|
+
criticalFailureBlocker: true,
|
|
69
|
+
notes: 'long productive continuation — no premature stop, no rabbit hole',
|
|
70
|
+
}),
|
|
71
|
+
constraint: Object.freeze({
|
|
72
|
+
dimensions: Object.freeze(['correctness', 'evidence', 'completeness']),
|
|
73
|
+
criticalFailureBlocker: true,
|
|
74
|
+
notes: 'declared constraints are hard; a violated constraint is critical',
|
|
75
|
+
}),
|
|
76
|
+
failure: Object.freeze({
|
|
77
|
+
dimensions: Object.freeze(['correctness', 'evidence', 'completeness']),
|
|
78
|
+
criticalFailureBlocker: true,
|
|
79
|
+
notes: 'expected behavior is a typed refusal/fallback, never a fabricated result',
|
|
80
|
+
}),
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
const FIXTURES = Object.freeze([
|
|
84
|
+
// --- mandatory classes ---
|
|
85
|
+
Object.freeze({
|
|
86
|
+
id: 'SO-F01-trivial-rename',
|
|
87
|
+
taskClass: 'trivial',
|
|
88
|
+
language: 'en',
|
|
89
|
+
rubric: 'trivial',
|
|
90
|
+
summary: 'rename a local identifier in one file; deterministic fast path',
|
|
91
|
+
}),
|
|
92
|
+
Object.freeze({
|
|
93
|
+
id: 'SO-F02-retrieval-symbol',
|
|
94
|
+
taskClass: 'retrieval',
|
|
95
|
+
language: 'en',
|
|
96
|
+
rubric: 'retrieval',
|
|
97
|
+
summary: 'locate a symbol definition and its consumers; answer carries refs',
|
|
98
|
+
}),
|
|
99
|
+
Object.freeze({
|
|
100
|
+
id: 'SO-F03-debug-failing-test',
|
|
101
|
+
taskClass: 'debug',
|
|
102
|
+
language: 'en',
|
|
103
|
+
rubric: 'debug',
|
|
104
|
+
summary: 'reproduce a failing vitest case and name the root cause',
|
|
105
|
+
}),
|
|
106
|
+
Object.freeze({
|
|
107
|
+
id: 'SO-F04-review-seeded-defect',
|
|
108
|
+
taskClass: 'review',
|
|
109
|
+
language: 'en',
|
|
110
|
+
rubric: 'review',
|
|
111
|
+
summary: 'review a diff containing one seeded defect; defect must surface',
|
|
112
|
+
}),
|
|
113
|
+
Object.freeze({
|
|
114
|
+
id: 'SO-F05-continuation-long',
|
|
115
|
+
taskClass: 'continuation',
|
|
116
|
+
language: 'en',
|
|
117
|
+
rubric: 'continuation',
|
|
118
|
+
summary: 'multi-step productive continuation from a mid-run state',
|
|
119
|
+
}),
|
|
120
|
+
Object.freeze({
|
|
121
|
+
id: 'SO-F06-constraint-no-api-change',
|
|
122
|
+
taskClass: 'constraint',
|
|
123
|
+
language: 'en',
|
|
124
|
+
rubric: 'constraint',
|
|
125
|
+
summary: 'refactor under an explicit "no public API change" constraint',
|
|
126
|
+
}),
|
|
127
|
+
// --- VI language coverage (SPEC §9: EN/VI cases) ---
|
|
128
|
+
Object.freeze({
|
|
129
|
+
id: 'SO-F07-retrieval-vi',
|
|
130
|
+
taskClass: 'retrieval',
|
|
131
|
+
language: 'vi',
|
|
132
|
+
rubric: 'retrieval',
|
|
133
|
+
summary: 'câu hỏi truy xuất: tìm định nghĩa và nơi dùng của hàm',
|
|
134
|
+
}),
|
|
135
|
+
Object.freeze({
|
|
136
|
+
id: 'SO-F08-debug-vi',
|
|
137
|
+
taskClass: 'debug',
|
|
138
|
+
language: 'vi',
|
|
139
|
+
rubric: 'debug',
|
|
140
|
+
summary: 'tái hiện lỗi và chỉ ra nguyên nhân gốc của test hỏng',
|
|
141
|
+
}),
|
|
142
|
+
Object.freeze({
|
|
143
|
+
id: 'SO-F09-continuation-vi',
|
|
144
|
+
taskClass: 'continuation',
|
|
145
|
+
language: 'vi',
|
|
146
|
+
rubric: 'continuation',
|
|
147
|
+
summary: 'tiếp tục công việc dở dang từ trạng thái giữa chừng',
|
|
148
|
+
}),
|
|
149
|
+
// --- failure cases (typed fallbacks, never fabricated results) ---
|
|
150
|
+
Object.freeze({
|
|
151
|
+
id: 'SO-F10-failure-auth-reject',
|
|
152
|
+
taskClass: 'failure',
|
|
153
|
+
language: 'en',
|
|
154
|
+
rubric: 'failure',
|
|
155
|
+
failureKind: 'auth_reject',
|
|
156
|
+
summary: 'decision endpoint rejects auth → deterministic route, unavailable receipt',
|
|
157
|
+
}),
|
|
158
|
+
Object.freeze({
|
|
159
|
+
id: 'SO-F11-failure-stale-ref',
|
|
160
|
+
taskClass: 'failure',
|
|
161
|
+
language: 'en',
|
|
162
|
+
rubric: 'failure',
|
|
163
|
+
failureKind: 'stale_reference',
|
|
164
|
+
summary: 'evidence ref points at a moved/absent source → typed partial, never complete',
|
|
165
|
+
}),
|
|
166
|
+
Object.freeze({
|
|
167
|
+
id: 'SO-F12-failure-host-no-fork',
|
|
168
|
+
taskClass: 'failure',
|
|
169
|
+
language: 'en',
|
|
170
|
+
rubric: 'failure',
|
|
171
|
+
failureKind: 'host_fork_unsupported',
|
|
172
|
+
summary: 'host cannot fork context → fork_unavailable, curated fallback',
|
|
173
|
+
}),
|
|
174
|
+
]);
|
|
175
|
+
|
|
176
|
+
const CORPUS = Object.freeze({
|
|
177
|
+
schema: 1,
|
|
178
|
+
revision: CORPUS_REVISION,
|
|
179
|
+
baselineSha: BASELINE_SHA,
|
|
180
|
+
rubricVersion: RUBRIC_VERSION,
|
|
181
|
+
arms: ARMS,
|
|
182
|
+
rubrics: RUBRICS,
|
|
183
|
+
fixtures: FIXTURES,
|
|
184
|
+
});
|
|
185
|
+
|
|
186
|
+
/** @returns the frozen corpus manifest (immutable — mutation throws). */
|
|
187
|
+
export function loadSubagentCorpus() {
|
|
188
|
+
return CORPUS;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Gate a run's eligibility for baseline comparison.
|
|
193
|
+
* @param {object} run recorded run identity
|
|
194
|
+
* @param {string} [run.corpusRevision]
|
|
195
|
+
* @param {string} [run.baselineSha]
|
|
196
|
+
* @param {string} [run.rubricVersion]
|
|
197
|
+
* @returns {{comparable:boolean, reason:string|null}} typed reason when not
|
|
198
|
+
* comparable — 'stale_corpus_revision' | 'stale_baseline_sha' |
|
|
199
|
+
* 'missing_rubric_version'
|
|
200
|
+
*/
|
|
201
|
+
export function checkComparability(run = {}) {
|
|
202
|
+
if (run.corpusRevision !== CORPUS_REVISION) {
|
|
203
|
+
return { comparable: false, reason: 'stale_corpus_revision' };
|
|
204
|
+
}
|
|
205
|
+
if (run.baselineSha !== BASELINE_SHA) {
|
|
206
|
+
return { comparable: false, reason: 'stale_baseline_sha' };
|
|
207
|
+
}
|
|
208
|
+
if (run.rubricVersion !== RUBRIC_VERSION) {
|
|
209
|
+
return { comparable: false, reason: 'missing_rubric_version' };
|
|
210
|
+
}
|
|
211
|
+
return { comparable: true, reason: null };
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/**
|
|
215
|
+
* Provenance-gated savings between two runs. A percentage is emitted ONLY
|
|
216
|
+
* when both sides carry provider-measured usage ('provider'); every other
|
|
217
|
+
* source — 'host-measured', 'estimated', 'unknown', absent — returns
|
|
218
|
+
* reportable:false with savingsPct:null. An unknown source is never 0%:
|
|
219
|
+
* absence of measurement is not evidence of no cost (SPEC FR-01).
|
|
220
|
+
* @param {{usageSource?:string, tokens?:number|null}} baseline
|
|
221
|
+
* @param {{usageSource?:string, tokens?:number|null}} candidate
|
|
222
|
+
* @returns {{reportable:boolean, savingsPct:number|null, reason:string|null}}
|
|
223
|
+
*/
|
|
224
|
+
export function computeSavings({ baseline = {}, candidate = {} } = {}) {
|
|
225
|
+
for (const [label, run] of [['baseline', baseline], ['candidate', candidate]]) {
|
|
226
|
+
const source = run.usageSource;
|
|
227
|
+
if (source !== 'provider') {
|
|
228
|
+
const reason = source === 'unknown' || source == null
|
|
229
|
+
? 'usage_source_unknown'
|
|
230
|
+
: 'usage_source_not_provider';
|
|
231
|
+
return {
|
|
232
|
+
reportable: false,
|
|
233
|
+
savingsPct: null,
|
|
234
|
+
reason: `${reason}:${label}`,
|
|
235
|
+
};
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
const b = baseline.tokens;
|
|
239
|
+
const c = candidate.tokens;
|
|
240
|
+
if (typeof b !== 'number' || typeof c !== 'number' || b <= 0) {
|
|
241
|
+
return { reportable: false, savingsPct: null, reason: 'usage_tokens_missing' };
|
|
242
|
+
}
|
|
243
|
+
return {
|
|
244
|
+
reportable: true,
|
|
245
|
+
savingsPct: ((b - c) / b) * 100,
|
|
246
|
+
reason: null,
|
|
247
|
+
};
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
// ---------- CLI ----------
|
|
251
|
+
|
|
252
|
+
const USAGE = 'Usage: node scripts/bench/subagent-orchestrator-corpus.mjs --json\n'
|
|
253
|
+
+ ' --json print the frozen corpus manifest to stdout (byte-deterministic)\n'
|
|
254
|
+
+ 'Exit codes: 0 = ok, 1 = usage error, 2 = crash.';
|
|
255
|
+
|
|
256
|
+
function main(argv) {
|
|
257
|
+
const args = argv.slice(2);
|
|
258
|
+
if (args.length !== 1 || args[0] !== '--json') {
|
|
259
|
+
process.stderr.write(USAGE + '\n');
|
|
260
|
+
return 1;
|
|
261
|
+
}
|
|
262
|
+
process.stdout.write(JSON.stringify(CORPUS, null, 2) + '\n');
|
|
263
|
+
return 0;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
const isMain = process.argv[1]
|
|
267
|
+
&& path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
268
|
+
if (isMain) {
|
|
269
|
+
try {
|
|
270
|
+
process.exit(main(process.argv));
|
|
271
|
+
} catch (error) {
|
|
272
|
+
process.stderr.write(`corpus: ${error?.stack ?? error}\n`);
|
|
273
|
+
process.exit(2);
|
|
274
|
+
}
|
|
275
|
+
}
|