@ngockhoale/ukit 2.6.7 → 2.6.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/manifests/documentation.yaml +77 -6
  2. package/manifests/hostCapabilities.yaml +49 -0
  3. package/manifests/instructionRules.yaml +7 -0
  4. package/package.json +1 -1
  5. package/scripts/bench/goldTasks.json +38 -0
  6. package/scripts/bench/runGold.mjs +220 -0
  7. package/scripts/release/verify-release.mjs +6 -0
  8. package/src/cli/commands/code.js +182 -0
  9. package/src/cli/commands/doctor.js +35 -3
  10. package/src/cli/commands/indexTools.js +102 -1
  11. package/src/cli/commands/memory.js +137 -0
  12. package/src/cli/index.js +7 -0
  13. package/src/core/codeintel/compiler.js +316 -0
  14. package/src/core/codeintel/diagnostics.js +114 -0
  15. package/src/core/codeintel/freshness.js +295 -0
  16. package/src/core/codeintel/impact.js +251 -0
  17. package/src/core/codeintel/invalidation.js +150 -0
  18. package/src/core/codeintel/manifest.js +176 -0
  19. package/src/core/codeintel/packet.js +146 -0
  20. package/src/core/codeintel/providers.js +201 -0
  21. package/src/core/codeintel/retriever.js +372 -0
  22. package/src/core/codeintel/router.js +149 -0
  23. package/src/core/codeintel/semanticProvider.js +235 -0
  24. package/src/core/docContracts.js +723 -0
  25. package/src/core/memory/migrate.js +324 -0
  26. package/src/core/memory/records.js +172 -0
  27. package/src/core/memory/retrieval.js +161 -11
  28. package/src/core/memory/store.js +398 -0
  29. package/src/core/memory/storeV2.js +171 -0
  30. package/src/core/memory/storeV2Loader.js +22 -0
  31. package/src/core/runtimeConfig.js +125 -0
  32. package/src/core/runtimePaths.js +3 -0
  33. package/src/index/taskRouting.js +39 -0
  34. package/templates/.claude/ukit/index/route-task.mjs +40 -0
  35. package/templates/AGENTS.md +46 -99
  36. package/templates/CLAUDE.md +46 -99
  37. package/templates/docs/AI_HANDOFF/tasks/_TEMPLATE.md +5 -0
  38. package/templates/docs/BUGFIX.md +2 -19
  39. package/templates/docs/BUG_INDEX.md +43 -0
  40. package/templates/docs/BUG_METRICS.md +1 -5
  41. package/templates/docs/BUG_TEMPLATE.md +1 -11
  42. package/templates/docs/UKIT_INTERNALS.md +4 -0
  43. package/templates/instructions/core.md +46 -99
  44. package/templates/ukit/storage/config.json +30 -0
@@ -134,7 +134,8 @@ entries:
134
134
  load_policy: task-routed
135
135
  validation: [manual]
136
136
  archive_policy: never
137
- budget: { max_lines: 250, enforcement: warning }
137
+ budget: { max_lines: 120, enforcement: warning }
138
+ notes: 'recalibrated TASK-217: actual 92 lines; 250 was >2.7× actual — 120 leaves ~30% headroom'
138
139
  - id: docs-improvement-roadmap
139
140
  path: docs/DOCUMENTATION_IMPROVEMENT_ROADMAP.md
140
141
  class: canonical
@@ -175,7 +176,8 @@ entries:
175
176
  load_policy: task-routed
176
177
  validation: [manual]
177
178
  archive_policy: never
178
- budget: { max_lines: 200, enforcement: warning }
179
+ budget: { max_lines: 110, enforcement: warning }
180
+ notes: 'recalibrated TASK-217: actual 82 lines; 200 was ~2.4× actual — 110 leaves ~34% headroom'
179
181
  - id: docs-project
180
182
  path: docs/PROJECT.md
181
183
  class: canonical
@@ -186,6 +188,16 @@ entries:
186
188
  load_policy: on-demand
187
189
  validation: [manual]
188
190
  archive_policy: never
191
+ - id: docs-project-description
192
+ path: docs/PROJECT_DESCRIPTION.md
193
+ class: canonical
194
+ audience: [agent, maintainer]
195
+ owner: product
196
+ source_of_truth: docs/PROJECT_DESCRIPTION.md
197
+ merge_strategy: none
198
+ load_policy: on-demand
199
+ validation: [manual]
200
+ archive_policy: never
189
201
  - id: docs-project-important-spec
190
202
  path: docs/PROJECT_IMPORTANT_SPEC.md
191
203
  class: canonical
@@ -196,7 +208,8 @@ entries:
196
208
  load_policy: on-demand
197
209
  validation: [manual]
198
210
  archive_policy: replace-with-snapshot
199
- notes: archived DOC-106 → docs/archive/specs/; live file is compact decision record
211
+ budget: { max_lines: 60, enforcement: warning }
212
+ notes: 'archived DOC-106 → docs/archive/specs/; live file is compact decision record. TASK-217: budget added (audit listed 44/150) — actual 44 lines, 60 leaves ~36% headroom'
200
213
  - id: docs-context-budget
201
214
  path: docs/CONTEXT_BUDGET.md
202
215
  class: canonical
@@ -271,7 +284,7 @@ entries:
271
284
  validation: [manual]
272
285
  archive_policy: date-rotate
273
286
  budget: { max_lines: 150, enforcement: error }
274
- notes: enforcement flipped to error by TASK-005 (2026-09-19)
287
+ notes: enforcement flipped to error by TASK-005 (2026-09-19); TASK-217 audit — actual 83 lines (~55% headroom), value kept
275
288
  - id: docs-worklog
276
289
  path: docs/WORKLOG.md
277
290
  class: runtime
@@ -283,7 +296,7 @@ entries:
283
296
  validation: [manual]
284
297
  archive_policy: never
285
298
  budget: { max_lines: 600, enforcement: warning }
286
- notes: rotation = DOC-105
299
+ notes: 'rotation = DOC-105; TASK-217 audit — actual 528 lines (~12% headroom), value kept pending rotation'
287
300
  - id: docs-ai-handoff
288
301
  path: docs/AI_HANDOFF/
289
302
  class: runtime
@@ -295,6 +308,17 @@ entries:
295
308
  validation: [manual]
296
309
  archive_policy: never
297
310
  notes: dir entry — covers all files beneath
311
+ - id: docs-ai-handoff-index
312
+ path: docs/AI_HANDOFF/INDEX.md
313
+ class: runtime
314
+ audience: [agent, maintainer]
315
+ owner: runtime
316
+ source_of_truth: docs/AI_HANDOFF/INDEX.md
317
+ merge_strategy: none
318
+ load_policy: task-routed
319
+ validation: [manual]
320
+ archive_policy: never
321
+ notes: handoff run cursor index; declared in context_layers.intents.handoff (DOC-201)
298
322
  - id: docs-archive
299
323
  path: docs/archive/
300
324
  class: archive
@@ -382,7 +406,7 @@ entries:
382
406
  validation: [manual]
383
407
  archive_policy: never
384
408
  budget: { max_lines: 160, enforcement: warning }
385
- notes: canonical shared core; pre-shrink >160 is warning-only (DOC-104 shrinks)
409
+ notes: 'canonical shared core; TASK-217 shrink 210→157 lines — budget 160 kept per SPEC §14 (now obeyed, warning-only)'
386
410
  - id: tpl-omp-rules
387
411
  path: templates/.omp/RULES.md
388
412
  class: generated
@@ -660,3 +684,50 @@ entries:
660
684
  load_policy: on-demand
661
685
  validation: [manual]
662
686
  archive_policy: never
687
+
688
+ # ── Same-wave additions (DOC-201 batch; files authored by TASK-208/212) ────
689
+ - id: docs-host-capability-matrix
690
+ path: docs/HOST_CAPABILITY_MATRIX.md
691
+ class: canonical
692
+ audience: [maintainer, contributor]
693
+ owner: adapter
694
+ source_of_truth: docs/HOST_CAPABILITY_MATRIX.md
695
+ merge_strategy: none
696
+ load_policy: on-demand
697
+ validation: [manual]
698
+ archive_policy: never
699
+ notes: hand-authored from manifests/hostCapabilities.yaml (TASK-208 adds hostCapabilityMatrix.test.js enforcement)
700
+ - id: manifest-host-capabilities
701
+ path: manifests/hostCapabilities.yaml
702
+ class: canonical
703
+ audience: [maintainer]
704
+ owner: adapter
705
+ source_of_truth: manifests/hostCapabilities.yaml
706
+ merge_strategy: none
707
+ load_policy: on-demand
708
+ validation: [manual]
709
+ archive_policy: never
710
+ notes: machine-readable host capability states (DOC-202); matrix doc renders from it
711
+ - id: docs-prompt-caching-vendor-evidence
712
+ path: docs/research/PROMPT_CACHING_VENDOR_EVIDENCE.md
713
+ class: archive
714
+ audience: [maintainer]
715
+ owner: product
716
+ source_of_truth: docs/research/PROMPT_CACHING_VENDOR_EVIDENCE.md
717
+ merge_strategy: none
718
+ load_policy: never
719
+ validation: [manual]
720
+ archive_policy: immutable
721
+ notes: vendor evidence extracted from docs/PROMPT_CACHING.md (DOC-206, TASK-212)
722
+
723
+ # Declared complexity→docs mapping consumed by the router's `docs=` summary segment
724
+ # (src/index/taskRouting.js CONTEXT_LAYER_DOCS mirrors this block; DOC-201 FR-001/002).
725
+ # queued-task intent omitted v1: docs/TASKS.md is optional per-repo (SPEC §14).
726
+ context_layers:
727
+ task_types:
728
+ trivial: []
729
+ simple: [docs/MEMORY.md]
730
+ non-trivial: [docs/MEMORY.md, docs/PROJECT.md, docs/CODE_MAP.md]
731
+ intents:
732
+ open-ended: [docs/STATUS.md]
733
+ handoff: [docs/AI_HANDOFF/INDEX.md]
@@ -0,0 +1,49 @@
1
+ # Host capability matrix (DOC-202 / FR-003).
2
+ # Every host × capability cell carries a state:
3
+ # tested — an automated test under tests/** exercises the shipped artifact (evidence required)
4
+ # observed — shipped artifact exists and is reviewed by hand; no dedicated automated test (evidence/note required)
5
+ # unknown — not verified in this repo
6
+ # Human table mirror: docs/HOST_CAPABILITY_MATRIX.md
7
+ # Consistency enforced by: tests/consistency/hostCapabilityMatrix.test.js
8
+ version: 1
9
+ hosts:
10
+ claude-code:
11
+ skills: { state: tested, evidence: tests/integration/installPipeline.test.js }
12
+ subagents: { state: tested, evidence: tests/consistency/mappingFingerprint.test.js }
13
+ hooks: { state: tested, evidence: tests/hooks/skillRouterHook.test.js }
14
+ session-start-injection: { state: tested, evidence: tests/hooks/reinjectContextHook.test.js }
15
+ stop-gate: { state: tested, evidence: tests/hooks/stopCoordinator.test.js }
16
+ slash-commands: { state: tested, evidence: tests/integration/artifactReferenceIntegrity.test.js }
17
+ owner-file-injection: { state: tested, evidence: tests/consistency/instructionRender.test.js }
18
+ settings-env: { state: tested, evidence: tests/core/repairBrokenHooks.test.js }
19
+ context-files: { state: tested, evidence: tests/integration/installPipeline.test.js }
20
+ codex:
21
+ skills: { state: observed, note: '.codex/skills mirror ships via install; no dedicated codex-skill test' }
22
+ subagents: { state: unknown }
23
+ hooks: { state: unknown }
24
+ session-start-injection: { state: unknown }
25
+ stop-gate: { state: unknown }
26
+ slash-commands: { state: unknown }
27
+ owner-file-injection: { state: observed, note: 'AGENTS.md rendered for codex; parity asserted only at artifact level' }
28
+ settings-env: { state: tested, evidence: tests/integration/installPipeline.test.js }
29
+ context-files: { state: tested, evidence: tests/integration/installPipeline.test.js }
30
+ opencode:
31
+ skills: { state: observed, note: 'reads .claude/skills via discovery; asserted indirectly by artifact tests' }
32
+ subagents: { state: unknown }
33
+ hooks: { state: unknown }
34
+ session-start-injection: { state: unknown }
35
+ stop-gate: { state: unknown }
36
+ slash-commands: { state: observed, note: 'ukit-* helper entrypoints declared in opencode.json' }
37
+ owner-file-injection: { state: observed, note: 'AGENTS.md loaded at session start; no injection test' }
38
+ settings-env: { state: tested, evidence: tests/integration/installPipeline.test.js }
39
+ context-files: { state: observed, note: 'AGENTS.md asserted by install pipeline; no opencode-specific parse test' }
40
+ omp:
41
+ skills: { state: observed, note: 'omp reads .claude/skills via claude discovery provider' }
42
+ subagents: { state: tested, evidence: tests/consistency/ompAgentParity.test.js }
43
+ hooks: { state: tested, evidence: tests/hooks/ompHookBridge.test.js }
44
+ session-start-injection: { state: tested, evidence: tests/hooks/ompHookBridge.test.js }
45
+ stop-gate: { state: unknown }
46
+ slash-commands: { state: observed, note: 'delegation parentheticals checked by tests/consistency/ompCommandParity.test.js' }
47
+ owner-file-injection: { state: unknown }
48
+ settings-env: { state: observed, note: 'templates/.omp/config.yml modelRoles reviewed by ompAgentParity' }
49
+ context-files: { state: observed, note: 'omp consumes AGENTS.md/CLAUDE.md context layer; no dedicated test' }
@@ -11,6 +11,11 @@ description: >-
11
11
  # SHARED) — its anchor `PROJECT_IMPORTANT.md` does not exist in the CLAUDE
12
12
  # files, and markers are comment-only (no prose edits allowed). AGENTS-set
13
13
  # coverage matches the HOST-OC-01/HOST-OWN-01 rows that tag the same overlay.
14
+ # paths (optional, DOC-302): list of repo-relative glob strings naming the
15
+ # work-tree scope a rule applies to. `**` = any depth, `*` = one segment,
16
+ # `?` = one char; `/`-separated, no leading `/`, no `..`, no backslashes.
17
+ # Absent = file-scoped only (marker_in governs rendered coverage; paths is
18
+ # metadata for future router/doc tooling, consumed by rulesForPath).
14
19
  rules:
15
20
  - id: CORE-01
16
21
  title: One remembered command — ukit install
@@ -268,6 +273,7 @@ rules:
268
273
  The handoff quality gate activates only when a task goes through
269
274
  docs/AI_HANDOFF; daily prompts keep the old flow.
270
275
  marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
276
+ paths: [docs/AI_HANDOFF/**]
271
277
  - id: HAND-02
272
278
  title: RUN.md is the authoritative handoff run cursor
273
279
  kind: critical
@@ -277,6 +283,7 @@ rules:
277
283
  a run.
278
284
  marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
279
285
  anchor: 'RUN.md'
286
+ paths: [docs/AI_HANDOFF/**]
280
287
  - id: BUDGET-01
281
288
  title: Context budget per task class
282
289
  kind: critical
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ngockhoale/ukit",
3
- "version": "2.6.7",
3
+ "version": "2.6.8",
4
4
  "description": "Install/update an index-first AI workspace for Claude Code, OpenAI Codex, OpenCode, and omp (Oh My Pi).",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -0,0 +1,38 @@
1
+ [
2
+ {
3
+ "id": "gold-001-trivial-none",
4
+ "prompt": "fix typo in the header label",
5
+ "expectedMode": "none",
6
+ "expectedAnchors": [],
7
+ "maxTokens": 500
8
+ },
9
+ {
10
+ "id": "gold-002-peek",
11
+ "prompt": "peek src/cli/index.js",
12
+ "flags": { "mode": "peek" },
13
+ "expectedMode": "peek",
14
+ "expectedAnchors": ["src/cli/index.js"],
15
+ "maxTokens": 500
16
+ },
17
+ {
18
+ "id": "gold-003-targeted",
19
+ "prompt": "update src/cli/index.js to add a code command entry",
20
+ "expectedMode": "targeted",
21
+ "expectedAnchors": ["src/cli/index.js"],
22
+ "maxTokens": 2000
23
+ },
24
+ {
25
+ "id": "gold-004-impact",
26
+ "prompt": "build fails with TypeError in src/index/buildIndex.js",
27
+ "expectedMode": "impact",
28
+ "expectedAnchors": ["src/index/buildIndex.js"],
29
+ "maxTokens": 3000
30
+ },
31
+ {
32
+ "id": "gold-005-explore",
33
+ "prompt": "where is index freshness handled",
34
+ "expectedMode": "explore",
35
+ "expectedAnchors": [],
36
+ "maxTokens": 4000
37
+ }
38
+ ]
@@ -0,0 +1,220 @@
1
+ #!/usr/bin/env node
2
+ // Gold-task benchmark harness skeleton (SPEC §10).
3
+ // Loads a corpus of {id, prompt, expectedMode, expectedAnchors[], maxTokens}
4
+ // cases, runs routeTask + compileContext per case, prints a scorecard with a
5
+ // totals block, optionally writes an aggregated JSON scorecard (--json <path>),
6
+ // and exits 0 only when every case passes. CI-3xx gates on the JSON summary.
7
+
8
+ import fs from 'node:fs/promises';
9
+ import path from 'node:path';
10
+ import { fileURLToPath } from 'node:url';
11
+
12
+ const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..');
13
+ const DEFAULT_CORPUS = path.join(REPO_ROOT, 'scripts', 'bench', 'goldTasks.json');
14
+
15
+ function parseArgs(argv) {
16
+ let corpusPath = DEFAULT_CORPUS;
17
+ let projectRoot = REPO_ROOT;
18
+ let jsonPath = null;
19
+ for (let i = 0; i < argv.length; i += 1) {
20
+ if (argv[i] === '--corpus') {
21
+ corpusPath = path.resolve(argv[i + 1] ?? corpusPath);
22
+ i += 1;
23
+ } else if (argv[i] === '--project') {
24
+ projectRoot = path.resolve(argv[i + 1] ?? projectRoot);
25
+ i += 1;
26
+ } else if (argv[i] === '--json') {
27
+ jsonPath = argv[i + 1] ? path.resolve(argv[i + 1]) : null;
28
+ i += 1;
29
+ }
30
+ }
31
+ return { corpusPath, projectRoot, jsonPath };
32
+ }
33
+
34
+ async function loadCorpus(corpusPath) {
35
+ const raw = await fs.readFile(corpusPath, 'utf8');
36
+ const cases = JSON.parse(raw);
37
+ if (!Array.isArray(cases)) {
38
+ throw new Error(`bench corpus must be a JSON array: ${corpusPath}`);
39
+ }
40
+ return cases;
41
+ }
42
+
43
+ function anchorPathsOf(packet) {
44
+ const fromAnchors = (packet.anchors ?? []).map((a) => a.path).filter(Boolean);
45
+ const fromEvidence = (packet.evidence ?? []).map((e) => e.path).filter(Boolean);
46
+ return new Set([...fromAnchors, ...fromEvidence]);
47
+ }
48
+
49
+ function normalizeMode(mode) {
50
+ return typeof mode === 'string' && mode.trim() !== '' ? mode.trim() : null;
51
+ }
52
+
53
+ export async function runGoldCase(projectRoot, testCase, { compileContext, routeTask }) {
54
+ const failures = [];
55
+ const expectedMode = normalizeMode(testCase.expectedMode);
56
+ const expectedAnchors = Array.isArray(testCase.expectedAnchors) ? testCase.expectedAnchors : [];
57
+ const maxTokens = typeof testCase.maxTokens === 'number' ? testCase.maxTokens : Infinity;
58
+
59
+ const routed = routeTask({
60
+ prompt: testCase.prompt ?? '',
61
+ hasError: testCase.hasError === true,
62
+ filesHinted: Array.isArray(testCase.filesHinted) ? testCase.filesHinted : undefined,
63
+ flags: testCase.flags && typeof testCase.flags === 'object' ? testCase.flags : {},
64
+ });
65
+
66
+ const packet = await compileContext(projectRoot, {
67
+ prompt: testCase.prompt ?? '',
68
+ hasError: testCase.hasError === true,
69
+ filesHinted: Array.isArray(testCase.filesHinted) ? testCase.filesHinted : undefined,
70
+ flags: testCase.flags && typeof testCase.flags === 'object' ? testCase.flags : {},
71
+ });
72
+
73
+ if (expectedMode && packet.task_type !== expectedMode) {
74
+ failures.push(`mode: expected ${expectedMode}, got ${packet.task_type || '(empty)'}`);
75
+ }
76
+
77
+ const anchorPaths = anchorPathsOf(packet);
78
+ const anchorHits = [];
79
+ const anchorMisses = [];
80
+ for (const expected of expectedAnchors) {
81
+ if (anchorPaths.has(expected)) {
82
+ anchorHits.push(expected);
83
+ } else {
84
+ anchorMisses.push(expected);
85
+ failures.push(`anchor missing: ${expected}`);
86
+ }
87
+ }
88
+
89
+ if (packet.budget?.used > maxTokens) {
90
+ failures.push(`budget: used ${packet.budget.used} > maxTokens ${maxTokens}`);
91
+ }
92
+
93
+ return {
94
+ id: testCase.id ?? '(unnamed)',
95
+ mode: packet.task_type,
96
+ expectedMode,
97
+ anchors: [...anchorPaths],
98
+ expectedAnchors,
99
+ anchorHits,
100
+ anchorMisses,
101
+ budgetUsed: packet.budget?.used ?? 0,
102
+ maxTokens,
103
+ routedMode: routed.mode,
104
+ pass: failures.length === 0,
105
+ failures,
106
+ };
107
+ }
108
+
109
+ export async function runGold({ projectRoot, corpusPath } = {}) {
110
+ const rootDir = projectRoot ?? REPO_ROOT;
111
+ const { compileContext } = await import(path.join(REPO_ROOT, 'src/core/codeintel/compiler.js'));
112
+ const { routeTask } = await import(path.join(REPO_ROOT, 'src/core/codeintel/router.js'));
113
+ const { buildCodeIndex } = await import(path.join(REPO_ROOT, 'src/index/buildIndex.js'));
114
+ const { INDEX_ARTIFACTS, getArtifactPath } = await import(path.join(REPO_ROOT, 'src/index/paths.js'));
115
+
116
+ // The bench needs an index to score anchors; build once when missing.
117
+ try {
118
+ await fs.stat(getArtifactPath(rootDir, INDEX_ARTIFACTS.files));
119
+ } catch {
120
+ console.log('[bench] index missing — building code index first');
121
+ await buildCodeIndex({ rootDir });
122
+ }
123
+
124
+ const cases = await loadCorpus(corpusPath ?? DEFAULT_CORPUS);
125
+ const results = [];
126
+ for (const testCase of cases) {
127
+ results.push(await runGoldCase(rootDir, testCase, { compileContext, routeTask }));
128
+ }
129
+ return { results, pass: results.every((r) => r.pass) };
130
+ }
131
+
132
+ /**
133
+ * Aggregate per-case results into a machine-readable summary for CI gating.
134
+ * Pooled anchor precision = hits/found, recall = hits/expected; cases with
135
+ * empty expectedAnchors contribute 0 to both denominators (excluded, not 1).
136
+ * @param {Array<object>} results per-case runGoldCase outputs
137
+ * @returns {{total:number, passed:number, passRate:number, modeAccuracy:number|null, anchorPrecision:number|null, anchorRecall:number|null, meanBudgetUsedPct:number|null}}
138
+ */
139
+ export function summarizeResults(results) {
140
+ const total = results.length;
141
+ const passed = results.filter((r) => r.pass).length;
142
+ const passRate = total === 0 ? 0 : passed / total;
143
+
144
+ const withMode = results.filter((r) => r.expectedMode);
145
+ const modeCorrect = withMode.filter((r) => r.mode === r.expectedMode).length;
146
+ const modeAccuracy = withMode.length === 0 ? null : modeCorrect / withMode.length;
147
+
148
+ let hits = 0;
149
+ let found = 0;
150
+ let expected = 0;
151
+ for (const r of results) {
152
+ hits += Array.isArray(r.anchorHits) ? r.anchorHits.length : 0;
153
+ found += Array.isArray(r.anchors) ? r.anchors.length : 0;
154
+ expected += Array.isArray(r.expectedAnchors) ? r.expectedAnchors.length : 0;
155
+ }
156
+ const anchorPrecision = found === 0 ? null : hits / found;
157
+ const anchorRecall = expected === 0 ? null : hits / expected;
158
+
159
+ const budgeted = results.filter((r) => typeof r.maxTokens === 'number' && r.maxTokens > 0);
160
+ const meanBudgetUsedPct = budgeted.length === 0
161
+ ? null
162
+ : budgeted.reduce((acc, r) => acc + (r.budgetUsed ?? 0) / r.maxTokens, 0) / budgeted.length;
163
+
164
+ return { total, passed, passRate, modeAccuracy, anchorPrecision, anchorRecall, meanBudgetUsedPct };
165
+ }
166
+
167
+ /**
168
+ * Write { generatedAt, results, summary } scorecard JSON to `jsonPath` (pretty-printed).
169
+ * @param {string} jsonPath output file path
170
+ * @param {Array<object>} results per-case runGoldCase outputs
171
+ */
172
+ export async function writeScorecardJson(jsonPath, results) {
173
+ const payload = {
174
+ generatedAt: new Date().toISOString(),
175
+ results,
176
+ summary: summarizeResults(results),
177
+ };
178
+ await fs.writeFile(jsonPath, `${JSON.stringify(payload, null, 2)}\n`, 'utf8');
179
+ }
180
+
181
+ function fmtPct(value) {
182
+ return value === null ? 'n/a' : `${(value * 100).toFixed(1)}%`;
183
+ }
184
+
185
+ export function printScorecard(results) {
186
+ console.log('id | mode | expected | anchors | budget | verdict');
187
+ console.log('---|------|----------|---------|--------|--------');
188
+ for (const r of results) {
189
+ console.log(
190
+ `${r.id} | ${r.mode || '-'} | ${r.expectedMode || '-'} | ${r.anchors.length} | ${r.budgetUsed}/${r.maxTokens === Infinity ? '∞' : r.maxTokens} | ${r.pass ? 'PASS' : 'FAIL'}`,
191
+ );
192
+ for (const failure of r.failures) {
193
+ console.log(` ! ${failure}`);
194
+ }
195
+ }
196
+ const s = summarizeResults(results);
197
+ console.log(`scorecard: ${s.passed}/${s.total} passed`);
198
+ console.log(
199
+ `totals: passRate: ${fmtPct(s.passRate)} | modeAccuracy: ${fmtPct(s.modeAccuracy)} | anchorPrecision: ${fmtPct(s.anchorPrecision)} | anchorRecall: ${fmtPct(s.anchorRecall)} | meanBudgetUsedPct: ${fmtPct(s.meanBudgetUsedPct)}`,
200
+ );
201
+ }
202
+
203
+ const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
204
+ if (isMain) {
205
+ const { corpusPath, projectRoot, jsonPath } = parseArgs(process.argv.slice(2));
206
+ try {
207
+ const { results, pass } = await runGold({ projectRoot, corpusPath });
208
+ printScorecard(results);
209
+ if (jsonPath) {
210
+ await writeScorecardJson(jsonPath, results);
211
+ console.log(`[bench] scorecard JSON written: ${jsonPath}`);
212
+ }
213
+ if (!pass) {
214
+ process.exitCode = 1;
215
+ }
216
+ } catch (error) {
217
+ console.error(`[bench] ${error?.message ?? error}`);
218
+ process.exitCode = 1;
219
+ }
220
+ }
@@ -45,6 +45,12 @@ const steps = [
45
45
  args: ['test:artifact'],
46
46
  env: process.env,
47
47
  },
48
+ {
49
+ label: 'Doc contracts',
50
+ command: 'yarn',
51
+ args: ['vitest', 'run', 'tests/consistency/docContracts.test.js'],
52
+ env: process.env,
53
+ },
48
54
  {
49
55
  label: 'Core test suite',
50
56
  command: 'yarn',
@@ -0,0 +1,182 @@
1
+ import { compileContext } from '../../core/codeintel/compiler.js';
2
+ import { packetToText, validatePacket } from '../../core/codeintel/packet.js';
3
+ import { IndexFileSyntaxProvider } from '../../core/codeintel/providers.js';
4
+
5
+ // `ukit code` — CLI surface over the codeintel plane (SPEC §11).
6
+ // Every subcommand emits a Context Packet v1: `--json` prints the raw packet,
7
+ // default prints `packetToText`. Unknown/missing subcommand → help + non-zero.
8
+
9
+ const HELP_FLAGS = new Set(['--help', '-h', 'help']);
10
+ const SUBCOMMANDS = new Set(['peek', 'search', 'context', 'impact']);
11
+
12
+ function extractFlag(args, flag) {
13
+ const index = args.indexOf(flag);
14
+ if (index < 0) return { value: null, rest: args };
15
+ return {
16
+ value: args[index + 1] ?? null,
17
+ rest: [...args.slice(0, index), ...args.slice(index + 2)],
18
+ };
19
+ }
20
+
21
+ function hasFlag(args, flag) {
22
+ return args.includes(flag);
23
+ }
24
+
25
+ function parsePositiveInt(value, fallback) {
26
+ const parsed = Number.parseInt(value ?? '', 10);
27
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : fallback;
28
+ }
29
+
30
+ function printPacket(packet, { json }) {
31
+ if (json) {
32
+ console.log(JSON.stringify(packet, null, 2));
33
+ return;
34
+ }
35
+ console.log(packetToText(packet));
36
+ const validation = validatePacket(packet);
37
+ if (!validation.ok) {
38
+ console.log(`[UKit] packet validation warnings: ${validation.errors.join('; ')}`);
39
+ }
40
+ }
41
+
42
+ async function printFileOutline(projectRoot, filePath) {
43
+ try {
44
+ const provider = new IndexFileSyntaxProvider({ rootDir: projectRoot });
45
+ if (!provider.supports(filePath)) {
46
+ return;
47
+ }
48
+ const symbols = await provider.symbols(filePath);
49
+ if (!Array.isArray(symbols) || symbols.length === 0) {
50
+ return;
51
+ }
52
+ console.log('');
53
+ console.log(`## Outline (${filePath})`);
54
+ for (const item of symbols) {
55
+ console.log(`${item.line ?? '?'}: ${item.name ?? item.kind ?? ''}${item.kind ? ` [${item.kind}]` : ''}`);
56
+ }
57
+ } catch {
58
+ // Outline is best-effort decoration; never fail the packet print for it.
59
+ }
60
+ }
61
+
62
+ export async function runCode({ projectRoot, argv = [] }) {
63
+ const [subcommandRaw, ...rest] = argv;
64
+ const subcommand = (subcommandRaw ?? '').toLowerCase();
65
+
66
+ if (!subcommand || HELP_FLAGS.has(subcommand)) {
67
+ printCodeHelp();
68
+ if (!subcommand) {
69
+ process.exitCode = 1;
70
+ }
71
+ return;
72
+ }
73
+
74
+ if (!SUBCOMMANDS.has(subcommand)) {
75
+ console.error(`[UKit] Unknown code subcommand: ${subcommand}`);
76
+ printCodeHelp();
77
+ process.exitCode = 1;
78
+ return;
79
+ }
80
+
81
+ const json = hasFlag(rest, '--json');
82
+ const withoutJson = rest.filter((arg) => arg !== '--json');
83
+
84
+ if (subcommand === 'peek') {
85
+ const { value: lines, rest: args } = extractFlag(withoutJson, '--lines');
86
+ const filePath = args[0];
87
+ if (!filePath) {
88
+ console.error('[UKit] Missing file path. Usage: ukit code peek <path> [--lines a-b] [--json]');
89
+ process.exitCode = 1;
90
+ return;
91
+ }
92
+ const packet = await compileContext(projectRoot, {
93
+ prompt: filePath,
94
+ filesHinted: [filePath],
95
+ }, { mode: 'peek' });
96
+ printPacket(packet, { json });
97
+ if (!json) {
98
+ await printFileOutline(projectRoot, filePath);
99
+ if (lines) {
100
+ console.log(`(note: --lines ${lines} accepted; v1 outline prints full symbol list)`);
101
+ }
102
+ }
103
+ return;
104
+ }
105
+
106
+ if (subcommand === 'search') {
107
+ const { value: limitArg, rest: args } = extractFlag(withoutJson, '--limit');
108
+ const limit = parsePositiveInt(limitArg, 20);
109
+ const query = args.join(' ').trim();
110
+ if (!query) {
111
+ console.error('[UKit] Missing query. Usage: ukit code search "<query>" [--limit n] [--json]');
112
+ process.exitCode = 1;
113
+ return;
114
+ }
115
+ const packet = await compileContext(projectRoot, query);
116
+ if (packet.evidence.length > limit) {
117
+ const dropped = packet.evidence.splice(limit);
118
+ for (const item of dropped) {
119
+ packet.omitted.push({ what: item.path ?? 'evidence', why: 'limit' });
120
+ }
121
+ }
122
+ if (packet.anchors.length > limit) {
123
+ packet.anchors = packet.anchors.slice(0, limit);
124
+ }
125
+ printPacket(packet, { json });
126
+ return;
127
+ }
128
+
129
+ if (subcommand === 'context') {
130
+ const { value: mode, rest: afterMode } = extractFlag(withoutJson, '--mode');
131
+ const { value: budgetArg, rest: afterBudget } = extractFlag(afterMode, '--budget');
132
+ const budget = budgetArg !== null ? parsePositiveInt(budgetArg, null) : undefined;
133
+ const diagnostics = hasFlag(afterBudget, '--diagnostics');
134
+ const args = afterBudget.filter((arg) => arg !== '--diagnostics');
135
+ const task = args.join(' ').trim();
136
+ if (!task) {
137
+ console.error('[UKit] Missing task. Usage: ukit code context "<task>" [--mode m] [--budget n] [--diagnostics] [--json]');
138
+ process.exitCode = 1;
139
+ return;
140
+ }
141
+ const packet = await compileContext(projectRoot, task, {
142
+ mode: mode ?? undefined,
143
+ budget: budget ?? undefined,
144
+ diagnostics: diagnostics || undefined,
145
+ });
146
+ printPacket(packet, { json });
147
+ return;
148
+ }
149
+
150
+ // subcommand === 'impact'
151
+ const { value: depthArg, rest: impactArgs } = extractFlag(withoutJson, '--depth');
152
+ const depth = depthArg !== null ? parsePositiveInt(depthArg, null) : undefined;
153
+ const target = impactArgs[0];
154
+ if (!target) {
155
+ console.error('[UKit] Missing target. Usage: ukit code impact <path|symbol> [--depth N] [--json]');
156
+ process.exitCode = 1;
157
+ return;
158
+ }
159
+ const packet = await compileContext(projectRoot, {
160
+ prompt: target,
161
+ filesHinted: target.includes('/') || /\.[a-z0-9]+$/i.test(target) ? [target] : [],
162
+ }, { mode: 'impact', depth: depth ?? undefined });
163
+ printPacket(packet, { json });
164
+ }
165
+
166
+ export function printCodeHelp() {
167
+ console.log('UKit Code Commands (codeintel plane)');
168
+ console.log('Usage: ukit code <peek|search|context|impact> [args]');
169
+ console.log('');
170
+ console.log('Subcommands:');
171
+ console.log(' peek <path> [--lines a-b] L0 outline-only packet + file outline');
172
+ console.log(' search "<query>" [--limit n] Routed retrieval packet over the index');
173
+ console.log(' context "<task>" [--mode m] [--budget n] [--diagnostics] Compiled context packet for a task');
174
+ console.log(' impact <path|symbol> [--depth N] L2 packet with multi-hop reverse-import edges');
175
+ console.log('');
176
+ console.log('Options:');
177
+ console.log(' --json Print raw Context Packet JSON (default: text render)');
178
+ console.log(' --mode <m> Force router mode: none|peek|targeted|explore|impact|deep_flow|analogy');
179
+ console.log(' --budget <n> Override routed token budget');
180
+ console.log(' --depth <N> Impact-mode hop depth (clamped 1..codeIntel.impact.maxDepth)');
181
+ console.log(' --diagnostics Include post-edit diagnostics in next_actions');
182
+ }