@ngockhoale/ukit 2.6.7 → 2.6.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/manifests/documentation.yaml +77 -6
- package/manifests/hostCapabilities.yaml +49 -0
- package/manifests/instructionRules.yaml +7 -0
- package/package.json +1 -1
- package/scripts/bench/goldTasks.json +38 -0
- package/scripts/bench/runGold.mjs +220 -0
- package/scripts/release/verify-release.mjs +6 -0
- package/src/cli/commands/code.js +182 -0
- package/src/cli/commands/doctor.js +35 -3
- package/src/cli/commands/indexTools.js +102 -1
- package/src/cli/commands/memory.js +137 -0
- package/src/cli/index.js +7 -0
- package/src/core/codeintel/compiler.js +316 -0
- package/src/core/codeintel/diagnostics.js +114 -0
- package/src/core/codeintel/freshness.js +295 -0
- package/src/core/codeintel/impact.js +251 -0
- package/src/core/codeintel/invalidation.js +150 -0
- package/src/core/codeintel/manifest.js +176 -0
- package/src/core/codeintel/packet.js +146 -0
- package/src/core/codeintel/providers.js +201 -0
- package/src/core/codeintel/retriever.js +372 -0
- package/src/core/codeintel/router.js +149 -0
- package/src/core/codeintel/semanticProvider.js +235 -0
- package/src/core/docContracts.js +723 -0
- package/src/core/memory/migrate.js +324 -0
- package/src/core/memory/records.js +172 -0
- package/src/core/memory/retrieval.js +161 -11
- package/src/core/memory/store.js +398 -0
- package/src/core/memory/storeV2.js +171 -0
- package/src/core/memory/storeV2Loader.js +22 -0
- package/src/core/runtimeConfig.js +125 -0
- package/src/core/runtimePaths.js +3 -0
- package/src/index/taskRouting.js +39 -0
- package/templates/.claude/ukit/index/route-task.mjs +40 -0
- package/templates/AGENTS.md +46 -99
- package/templates/CLAUDE.md +46 -99
- package/templates/docs/AI_HANDOFF/tasks/_TEMPLATE.md +5 -0
- package/templates/docs/BUGFIX.md +2 -19
- package/templates/docs/BUG_INDEX.md +43 -0
- package/templates/docs/BUG_METRICS.md +1 -5
- package/templates/docs/BUG_TEMPLATE.md +1 -11
- package/templates/docs/UKIT_INTERNALS.md +4 -0
- package/templates/instructions/core.md +46 -99
- package/templates/ukit/storage/config.json +30 -0
|
@@ -134,7 +134,8 @@ entries:
|
|
|
134
134
|
load_policy: task-routed
|
|
135
135
|
validation: [manual]
|
|
136
136
|
archive_policy: never
|
|
137
|
-
budget: { max_lines:
|
|
137
|
+
budget: { max_lines: 120, enforcement: warning }
|
|
138
|
+
notes: 'recalibrated TASK-217: actual 92 lines; 250 was >2.7× actual — 120 leaves ~30% headroom'
|
|
138
139
|
- id: docs-improvement-roadmap
|
|
139
140
|
path: docs/DOCUMENTATION_IMPROVEMENT_ROADMAP.md
|
|
140
141
|
class: canonical
|
|
@@ -175,7 +176,8 @@ entries:
|
|
|
175
176
|
load_policy: task-routed
|
|
176
177
|
validation: [manual]
|
|
177
178
|
archive_policy: never
|
|
178
|
-
budget: { max_lines:
|
|
179
|
+
budget: { max_lines: 110, enforcement: warning }
|
|
180
|
+
notes: 'recalibrated TASK-217: actual 82 lines; 200 was ~2.4× actual — 110 leaves ~34% headroom'
|
|
179
181
|
- id: docs-project
|
|
180
182
|
path: docs/PROJECT.md
|
|
181
183
|
class: canonical
|
|
@@ -186,6 +188,16 @@ entries:
|
|
|
186
188
|
load_policy: on-demand
|
|
187
189
|
validation: [manual]
|
|
188
190
|
archive_policy: never
|
|
191
|
+
- id: docs-project-description
|
|
192
|
+
path: docs/PROJECT_DESCRIPTION.md
|
|
193
|
+
class: canonical
|
|
194
|
+
audience: [agent, maintainer]
|
|
195
|
+
owner: product
|
|
196
|
+
source_of_truth: docs/PROJECT_DESCRIPTION.md
|
|
197
|
+
merge_strategy: none
|
|
198
|
+
load_policy: on-demand
|
|
199
|
+
validation: [manual]
|
|
200
|
+
archive_policy: never
|
|
189
201
|
- id: docs-project-important-spec
|
|
190
202
|
path: docs/PROJECT_IMPORTANT_SPEC.md
|
|
191
203
|
class: canonical
|
|
@@ -196,7 +208,8 @@ entries:
|
|
|
196
208
|
load_policy: on-demand
|
|
197
209
|
validation: [manual]
|
|
198
210
|
archive_policy: replace-with-snapshot
|
|
199
|
-
|
|
211
|
+
budget: { max_lines: 60, enforcement: warning }
|
|
212
|
+
notes: 'archived DOC-106 → docs/archive/specs/; live file is compact decision record. TASK-217: budget added (audit listed 44/150) — actual 44 lines, 60 leaves ~36% headroom'
|
|
200
213
|
- id: docs-context-budget
|
|
201
214
|
path: docs/CONTEXT_BUDGET.md
|
|
202
215
|
class: canonical
|
|
@@ -271,7 +284,7 @@ entries:
|
|
|
271
284
|
validation: [manual]
|
|
272
285
|
archive_policy: date-rotate
|
|
273
286
|
budget: { max_lines: 150, enforcement: error }
|
|
274
|
-
notes: enforcement flipped to error by TASK-005 (2026-09-19)
|
|
287
|
+
notes: enforcement flipped to error by TASK-005 (2026-09-19); TASK-217 audit — actual 83 lines (~55% headroom), value kept
|
|
275
288
|
- id: docs-worklog
|
|
276
289
|
path: docs/WORKLOG.md
|
|
277
290
|
class: runtime
|
|
@@ -283,7 +296,7 @@ entries:
|
|
|
283
296
|
validation: [manual]
|
|
284
297
|
archive_policy: never
|
|
285
298
|
budget: { max_lines: 600, enforcement: warning }
|
|
286
|
-
notes: rotation = DOC-105
|
|
299
|
+
notes: 'rotation = DOC-105; TASK-217 audit — actual 528 lines (~12% headroom), value kept pending rotation'
|
|
287
300
|
- id: docs-ai-handoff
|
|
288
301
|
path: docs/AI_HANDOFF/
|
|
289
302
|
class: runtime
|
|
@@ -295,6 +308,17 @@ entries:
|
|
|
295
308
|
validation: [manual]
|
|
296
309
|
archive_policy: never
|
|
297
310
|
notes: dir entry — covers all files beneath
|
|
311
|
+
- id: docs-ai-handoff-index
|
|
312
|
+
path: docs/AI_HANDOFF/INDEX.md
|
|
313
|
+
class: runtime
|
|
314
|
+
audience: [agent, maintainer]
|
|
315
|
+
owner: runtime
|
|
316
|
+
source_of_truth: docs/AI_HANDOFF/INDEX.md
|
|
317
|
+
merge_strategy: none
|
|
318
|
+
load_policy: task-routed
|
|
319
|
+
validation: [manual]
|
|
320
|
+
archive_policy: never
|
|
321
|
+
notes: handoff run cursor index; declared in context_layers.intents.handoff (DOC-201)
|
|
298
322
|
- id: docs-archive
|
|
299
323
|
path: docs/archive/
|
|
300
324
|
class: archive
|
|
@@ -382,7 +406,7 @@ entries:
|
|
|
382
406
|
validation: [manual]
|
|
383
407
|
archive_policy: never
|
|
384
408
|
budget: { max_lines: 160, enforcement: warning }
|
|
385
|
-
notes: canonical shared core;
|
|
409
|
+
notes: 'canonical shared core; TASK-217 shrink 210→157 lines — budget 160 kept per SPEC §14 (now obeyed, warning-only)'
|
|
386
410
|
- id: tpl-omp-rules
|
|
387
411
|
path: templates/.omp/RULES.md
|
|
388
412
|
class: generated
|
|
@@ -660,3 +684,50 @@ entries:
|
|
|
660
684
|
load_policy: on-demand
|
|
661
685
|
validation: [manual]
|
|
662
686
|
archive_policy: never
|
|
687
|
+
|
|
688
|
+
# ── Same-wave additions (DOC-201 batch; files authored by TASK-208/212) ────
|
|
689
|
+
- id: docs-host-capability-matrix
|
|
690
|
+
path: docs/HOST_CAPABILITY_MATRIX.md
|
|
691
|
+
class: canonical
|
|
692
|
+
audience: [maintainer, contributor]
|
|
693
|
+
owner: adapter
|
|
694
|
+
source_of_truth: docs/HOST_CAPABILITY_MATRIX.md
|
|
695
|
+
merge_strategy: none
|
|
696
|
+
load_policy: on-demand
|
|
697
|
+
validation: [manual]
|
|
698
|
+
archive_policy: never
|
|
699
|
+
notes: hand-authored from manifests/hostCapabilities.yaml (TASK-208 adds hostCapabilityMatrix.test.js enforcement)
|
|
700
|
+
- id: manifest-host-capabilities
|
|
701
|
+
path: manifests/hostCapabilities.yaml
|
|
702
|
+
class: canonical
|
|
703
|
+
audience: [maintainer]
|
|
704
|
+
owner: adapter
|
|
705
|
+
source_of_truth: manifests/hostCapabilities.yaml
|
|
706
|
+
merge_strategy: none
|
|
707
|
+
load_policy: on-demand
|
|
708
|
+
validation: [manual]
|
|
709
|
+
archive_policy: never
|
|
710
|
+
notes: machine-readable host capability states (DOC-202); matrix doc renders from it
|
|
711
|
+
- id: docs-prompt-caching-vendor-evidence
|
|
712
|
+
path: docs/research/PROMPT_CACHING_VENDOR_EVIDENCE.md
|
|
713
|
+
class: archive
|
|
714
|
+
audience: [maintainer]
|
|
715
|
+
owner: product
|
|
716
|
+
source_of_truth: docs/research/PROMPT_CACHING_VENDOR_EVIDENCE.md
|
|
717
|
+
merge_strategy: none
|
|
718
|
+
load_policy: never
|
|
719
|
+
validation: [manual]
|
|
720
|
+
archive_policy: immutable
|
|
721
|
+
notes: vendor evidence extracted from docs/PROMPT_CACHING.md (DOC-206, TASK-212)
|
|
722
|
+
|
|
723
|
+
# Declared complexity→docs mapping consumed by the router's `docs=` summary segment
|
|
724
|
+
# (src/index/taskRouting.js CONTEXT_LAYER_DOCS mirrors this block; DOC-201 FR-001/002).
|
|
725
|
+
# queued-task intent omitted v1: docs/TASKS.md is optional per-repo (SPEC §14).
|
|
726
|
+
context_layers:
|
|
727
|
+
task_types:
|
|
728
|
+
trivial: []
|
|
729
|
+
simple: [docs/MEMORY.md]
|
|
730
|
+
non-trivial: [docs/MEMORY.md, docs/PROJECT.md, docs/CODE_MAP.md]
|
|
731
|
+
intents:
|
|
732
|
+
open-ended: [docs/STATUS.md]
|
|
733
|
+
handoff: [docs/AI_HANDOFF/INDEX.md]
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# Host capability matrix (DOC-202 / FR-003).
|
|
2
|
+
# Every host × capability cell carries a state:
|
|
3
|
+
# tested — an automated test under tests/** exercises the shipped artifact (evidence required)
|
|
4
|
+
# observed — shipped artifact exists and is reviewed by hand; no dedicated automated test (evidence/note required)
|
|
5
|
+
# unknown — not verified in this repo
|
|
6
|
+
# Human table mirror: docs/HOST_CAPABILITY_MATRIX.md
|
|
7
|
+
# Consistency enforced by: tests/consistency/hostCapabilityMatrix.test.js
|
|
8
|
+
version: 1
|
|
9
|
+
hosts:
|
|
10
|
+
claude-code:
|
|
11
|
+
skills: { state: tested, evidence: tests/integration/installPipeline.test.js }
|
|
12
|
+
subagents: { state: tested, evidence: tests/consistency/mappingFingerprint.test.js }
|
|
13
|
+
hooks: { state: tested, evidence: tests/hooks/skillRouterHook.test.js }
|
|
14
|
+
session-start-injection: { state: tested, evidence: tests/hooks/reinjectContextHook.test.js }
|
|
15
|
+
stop-gate: { state: tested, evidence: tests/hooks/stopCoordinator.test.js }
|
|
16
|
+
slash-commands: { state: tested, evidence: tests/integration/artifactReferenceIntegrity.test.js }
|
|
17
|
+
owner-file-injection: { state: tested, evidence: tests/consistency/instructionRender.test.js }
|
|
18
|
+
settings-env: { state: tested, evidence: tests/core/repairBrokenHooks.test.js }
|
|
19
|
+
context-files: { state: tested, evidence: tests/integration/installPipeline.test.js }
|
|
20
|
+
codex:
|
|
21
|
+
skills: { state: observed, note: '.codex/skills mirror ships via install; no dedicated codex-skill test' }
|
|
22
|
+
subagents: { state: unknown }
|
|
23
|
+
hooks: { state: unknown }
|
|
24
|
+
session-start-injection: { state: unknown }
|
|
25
|
+
stop-gate: { state: unknown }
|
|
26
|
+
slash-commands: { state: unknown }
|
|
27
|
+
owner-file-injection: { state: observed, note: 'AGENTS.md rendered for codex; parity asserted only at artifact level' }
|
|
28
|
+
settings-env: { state: tested, evidence: tests/integration/installPipeline.test.js }
|
|
29
|
+
context-files: { state: tested, evidence: tests/integration/installPipeline.test.js }
|
|
30
|
+
opencode:
|
|
31
|
+
skills: { state: observed, note: 'reads .claude/skills via discovery; asserted indirectly by artifact tests' }
|
|
32
|
+
subagents: { state: unknown }
|
|
33
|
+
hooks: { state: unknown }
|
|
34
|
+
session-start-injection: { state: unknown }
|
|
35
|
+
stop-gate: { state: unknown }
|
|
36
|
+
slash-commands: { state: observed, note: 'ukit-* helper entrypoints declared in opencode.json' }
|
|
37
|
+
owner-file-injection: { state: observed, note: 'AGENTS.md loaded at session start; no injection test' }
|
|
38
|
+
settings-env: { state: tested, evidence: tests/integration/installPipeline.test.js }
|
|
39
|
+
context-files: { state: observed, note: 'AGENTS.md asserted by install pipeline; no opencode-specific parse test' }
|
|
40
|
+
omp:
|
|
41
|
+
skills: { state: observed, note: 'omp reads .claude/skills via claude discovery provider' }
|
|
42
|
+
subagents: { state: tested, evidence: tests/consistency/ompAgentParity.test.js }
|
|
43
|
+
hooks: { state: tested, evidence: tests/hooks/ompHookBridge.test.js }
|
|
44
|
+
session-start-injection: { state: tested, evidence: tests/hooks/ompHookBridge.test.js }
|
|
45
|
+
stop-gate: { state: unknown }
|
|
46
|
+
slash-commands: { state: observed, note: 'delegation parentheticals checked by tests/consistency/ompCommandParity.test.js' }
|
|
47
|
+
owner-file-injection: { state: unknown }
|
|
48
|
+
settings-env: { state: observed, note: 'templates/.omp/config.yml modelRoles reviewed by ompAgentParity' }
|
|
49
|
+
context-files: { state: observed, note: 'omp consumes AGENTS.md/CLAUDE.md context layer; no dedicated test' }
|
|
@@ -11,6 +11,11 @@ description: >-
|
|
|
11
11
|
# SHARED) — its anchor `PROJECT_IMPORTANT.md` does not exist in the CLAUDE
|
|
12
12
|
# files, and markers are comment-only (no prose edits allowed). AGENTS-set
|
|
13
13
|
# coverage matches the HOST-OC-01/HOST-OWN-01 rows that tag the same overlay.
|
|
14
|
+
# paths (optional, DOC-302): list of repo-relative glob strings naming the
|
|
15
|
+
# work-tree scope a rule applies to. `**` = any depth, `*` = one segment,
|
|
16
|
+
# `?` = one char; `/`-separated, no leading `/`, no `..`, no backslashes.
|
|
17
|
+
# Absent = file-scoped only (marker_in governs rendered coverage; paths is
|
|
18
|
+
# metadata for future router/doc tooling, consumed by rulesForPath).
|
|
14
19
|
rules:
|
|
15
20
|
- id: CORE-01
|
|
16
21
|
title: One remembered command — ukit install
|
|
@@ -268,6 +273,7 @@ rules:
|
|
|
268
273
|
The handoff quality gate activates only when a task goes through
|
|
269
274
|
docs/AI_HANDOFF; daily prompts keep the old flow.
|
|
270
275
|
marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
|
|
276
|
+
paths: [docs/AI_HANDOFF/**]
|
|
271
277
|
- id: HAND-02
|
|
272
278
|
title: RUN.md is the authoritative handoff run cursor
|
|
273
279
|
kind: critical
|
|
@@ -277,6 +283,7 @@ rules:
|
|
|
277
283
|
a run.
|
|
278
284
|
marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
|
|
279
285
|
anchor: 'RUN.md'
|
|
286
|
+
paths: [docs/AI_HANDOFF/**]
|
|
280
287
|
- id: BUDGET-01
|
|
281
288
|
title: Context budget per task class
|
|
282
289
|
kind: critical
|
package/package.json
CHANGED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"id": "gold-001-trivial-none",
|
|
4
|
+
"prompt": "fix typo in the header label",
|
|
5
|
+
"expectedMode": "none",
|
|
6
|
+
"expectedAnchors": [],
|
|
7
|
+
"maxTokens": 500
|
|
8
|
+
},
|
|
9
|
+
{
|
|
10
|
+
"id": "gold-002-peek",
|
|
11
|
+
"prompt": "peek src/cli/index.js",
|
|
12
|
+
"flags": { "mode": "peek" },
|
|
13
|
+
"expectedMode": "peek",
|
|
14
|
+
"expectedAnchors": ["src/cli/index.js"],
|
|
15
|
+
"maxTokens": 500
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"id": "gold-003-targeted",
|
|
19
|
+
"prompt": "update src/cli/index.js to add a code command entry",
|
|
20
|
+
"expectedMode": "targeted",
|
|
21
|
+
"expectedAnchors": ["src/cli/index.js"],
|
|
22
|
+
"maxTokens": 2000
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
"id": "gold-004-impact",
|
|
26
|
+
"prompt": "build fails with TypeError in src/index/buildIndex.js",
|
|
27
|
+
"expectedMode": "impact",
|
|
28
|
+
"expectedAnchors": ["src/index/buildIndex.js"],
|
|
29
|
+
"maxTokens": 3000
|
|
30
|
+
},
|
|
31
|
+
{
|
|
32
|
+
"id": "gold-005-explore",
|
|
33
|
+
"prompt": "where is index freshness handled",
|
|
34
|
+
"expectedMode": "explore",
|
|
35
|
+
"expectedAnchors": [],
|
|
36
|
+
"maxTokens": 4000
|
|
37
|
+
}
|
|
38
|
+
]
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Gold-task benchmark harness skeleton (SPEC §10).
|
|
3
|
+
// Loads a corpus of {id, prompt, expectedMode, expectedAnchors[], maxTokens}
|
|
4
|
+
// cases, runs routeTask + compileContext per case, prints a scorecard with a
|
|
5
|
+
// totals block, optionally writes an aggregated JSON scorecard (--json <path>),
|
|
6
|
+
// and exits 0 only when every case passes. CI-3xx gates on the JSON summary.
|
|
7
|
+
|
|
8
|
+
import fs from 'node:fs/promises';
|
|
9
|
+
import path from 'node:path';
|
|
10
|
+
import { fileURLToPath } from 'node:url';
|
|
11
|
+
|
|
12
|
+
const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..');
|
|
13
|
+
const DEFAULT_CORPUS = path.join(REPO_ROOT, 'scripts', 'bench', 'goldTasks.json');
|
|
14
|
+
|
|
15
|
+
function parseArgs(argv) {
|
|
16
|
+
let corpusPath = DEFAULT_CORPUS;
|
|
17
|
+
let projectRoot = REPO_ROOT;
|
|
18
|
+
let jsonPath = null;
|
|
19
|
+
for (let i = 0; i < argv.length; i += 1) {
|
|
20
|
+
if (argv[i] === '--corpus') {
|
|
21
|
+
corpusPath = path.resolve(argv[i + 1] ?? corpusPath);
|
|
22
|
+
i += 1;
|
|
23
|
+
} else if (argv[i] === '--project') {
|
|
24
|
+
projectRoot = path.resolve(argv[i + 1] ?? projectRoot);
|
|
25
|
+
i += 1;
|
|
26
|
+
} else if (argv[i] === '--json') {
|
|
27
|
+
jsonPath = argv[i + 1] ? path.resolve(argv[i + 1]) : null;
|
|
28
|
+
i += 1;
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
return { corpusPath, projectRoot, jsonPath };
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
async function loadCorpus(corpusPath) {
|
|
35
|
+
const raw = await fs.readFile(corpusPath, 'utf8');
|
|
36
|
+
const cases = JSON.parse(raw);
|
|
37
|
+
if (!Array.isArray(cases)) {
|
|
38
|
+
throw new Error(`bench corpus must be a JSON array: ${corpusPath}`);
|
|
39
|
+
}
|
|
40
|
+
return cases;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function anchorPathsOf(packet) {
|
|
44
|
+
const fromAnchors = (packet.anchors ?? []).map((a) => a.path).filter(Boolean);
|
|
45
|
+
const fromEvidence = (packet.evidence ?? []).map((e) => e.path).filter(Boolean);
|
|
46
|
+
return new Set([...fromAnchors, ...fromEvidence]);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
function normalizeMode(mode) {
|
|
50
|
+
return typeof mode === 'string' && mode.trim() !== '' ? mode.trim() : null;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export async function runGoldCase(projectRoot, testCase, { compileContext, routeTask }) {
|
|
54
|
+
const failures = [];
|
|
55
|
+
const expectedMode = normalizeMode(testCase.expectedMode);
|
|
56
|
+
const expectedAnchors = Array.isArray(testCase.expectedAnchors) ? testCase.expectedAnchors : [];
|
|
57
|
+
const maxTokens = typeof testCase.maxTokens === 'number' ? testCase.maxTokens : Infinity;
|
|
58
|
+
|
|
59
|
+
const routed = routeTask({
|
|
60
|
+
prompt: testCase.prompt ?? '',
|
|
61
|
+
hasError: testCase.hasError === true,
|
|
62
|
+
filesHinted: Array.isArray(testCase.filesHinted) ? testCase.filesHinted : undefined,
|
|
63
|
+
flags: testCase.flags && typeof testCase.flags === 'object' ? testCase.flags : {},
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
const packet = await compileContext(projectRoot, {
|
|
67
|
+
prompt: testCase.prompt ?? '',
|
|
68
|
+
hasError: testCase.hasError === true,
|
|
69
|
+
filesHinted: Array.isArray(testCase.filesHinted) ? testCase.filesHinted : undefined,
|
|
70
|
+
flags: testCase.flags && typeof testCase.flags === 'object' ? testCase.flags : {},
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
if (expectedMode && packet.task_type !== expectedMode) {
|
|
74
|
+
failures.push(`mode: expected ${expectedMode}, got ${packet.task_type || '(empty)'}`);
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const anchorPaths = anchorPathsOf(packet);
|
|
78
|
+
const anchorHits = [];
|
|
79
|
+
const anchorMisses = [];
|
|
80
|
+
for (const expected of expectedAnchors) {
|
|
81
|
+
if (anchorPaths.has(expected)) {
|
|
82
|
+
anchorHits.push(expected);
|
|
83
|
+
} else {
|
|
84
|
+
anchorMisses.push(expected);
|
|
85
|
+
failures.push(`anchor missing: ${expected}`);
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
if (packet.budget?.used > maxTokens) {
|
|
90
|
+
failures.push(`budget: used ${packet.budget.used} > maxTokens ${maxTokens}`);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
return {
|
|
94
|
+
id: testCase.id ?? '(unnamed)',
|
|
95
|
+
mode: packet.task_type,
|
|
96
|
+
expectedMode,
|
|
97
|
+
anchors: [...anchorPaths],
|
|
98
|
+
expectedAnchors,
|
|
99
|
+
anchorHits,
|
|
100
|
+
anchorMisses,
|
|
101
|
+
budgetUsed: packet.budget?.used ?? 0,
|
|
102
|
+
maxTokens,
|
|
103
|
+
routedMode: routed.mode,
|
|
104
|
+
pass: failures.length === 0,
|
|
105
|
+
failures,
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export async function runGold({ projectRoot, corpusPath } = {}) {
|
|
110
|
+
const rootDir = projectRoot ?? REPO_ROOT;
|
|
111
|
+
const { compileContext } = await import(path.join(REPO_ROOT, 'src/core/codeintel/compiler.js'));
|
|
112
|
+
const { routeTask } = await import(path.join(REPO_ROOT, 'src/core/codeintel/router.js'));
|
|
113
|
+
const { buildCodeIndex } = await import(path.join(REPO_ROOT, 'src/index/buildIndex.js'));
|
|
114
|
+
const { INDEX_ARTIFACTS, getArtifactPath } = await import(path.join(REPO_ROOT, 'src/index/paths.js'));
|
|
115
|
+
|
|
116
|
+
// The bench needs an index to score anchors; build once when missing.
|
|
117
|
+
try {
|
|
118
|
+
await fs.stat(getArtifactPath(rootDir, INDEX_ARTIFACTS.files));
|
|
119
|
+
} catch {
|
|
120
|
+
console.log('[bench] index missing — building code index first');
|
|
121
|
+
await buildCodeIndex({ rootDir });
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
const cases = await loadCorpus(corpusPath ?? DEFAULT_CORPUS);
|
|
125
|
+
const results = [];
|
|
126
|
+
for (const testCase of cases) {
|
|
127
|
+
results.push(await runGoldCase(rootDir, testCase, { compileContext, routeTask }));
|
|
128
|
+
}
|
|
129
|
+
return { results, pass: results.every((r) => r.pass) };
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Aggregate per-case results into a machine-readable summary for CI gating.
|
|
134
|
+
* Pooled anchor precision = hits/found, recall = hits/expected; cases with
|
|
135
|
+
* empty expectedAnchors contribute 0 to both denominators (excluded, not 1).
|
|
136
|
+
* @param {Array<object>} results per-case runGoldCase outputs
|
|
137
|
+
* @returns {{total:number, passed:number, passRate:number, modeAccuracy:number|null, anchorPrecision:number|null, anchorRecall:number|null, meanBudgetUsedPct:number|null}}
|
|
138
|
+
*/
|
|
139
|
+
export function summarizeResults(results) {
|
|
140
|
+
const total = results.length;
|
|
141
|
+
const passed = results.filter((r) => r.pass).length;
|
|
142
|
+
const passRate = total === 0 ? 0 : passed / total;
|
|
143
|
+
|
|
144
|
+
const withMode = results.filter((r) => r.expectedMode);
|
|
145
|
+
const modeCorrect = withMode.filter((r) => r.mode === r.expectedMode).length;
|
|
146
|
+
const modeAccuracy = withMode.length === 0 ? null : modeCorrect / withMode.length;
|
|
147
|
+
|
|
148
|
+
let hits = 0;
|
|
149
|
+
let found = 0;
|
|
150
|
+
let expected = 0;
|
|
151
|
+
for (const r of results) {
|
|
152
|
+
hits += Array.isArray(r.anchorHits) ? r.anchorHits.length : 0;
|
|
153
|
+
found += Array.isArray(r.anchors) ? r.anchors.length : 0;
|
|
154
|
+
expected += Array.isArray(r.expectedAnchors) ? r.expectedAnchors.length : 0;
|
|
155
|
+
}
|
|
156
|
+
const anchorPrecision = found === 0 ? null : hits / found;
|
|
157
|
+
const anchorRecall = expected === 0 ? null : hits / expected;
|
|
158
|
+
|
|
159
|
+
const budgeted = results.filter((r) => typeof r.maxTokens === 'number' && r.maxTokens > 0);
|
|
160
|
+
const meanBudgetUsedPct = budgeted.length === 0
|
|
161
|
+
? null
|
|
162
|
+
: budgeted.reduce((acc, r) => acc + (r.budgetUsed ?? 0) / r.maxTokens, 0) / budgeted.length;
|
|
163
|
+
|
|
164
|
+
return { total, passed, passRate, modeAccuracy, anchorPrecision, anchorRecall, meanBudgetUsedPct };
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Write { generatedAt, results, summary } scorecard JSON to `jsonPath` (pretty-printed).
|
|
169
|
+
* @param {string} jsonPath output file path
|
|
170
|
+
* @param {Array<object>} results per-case runGoldCase outputs
|
|
171
|
+
*/
|
|
172
|
+
export async function writeScorecardJson(jsonPath, results) {
|
|
173
|
+
const payload = {
|
|
174
|
+
generatedAt: new Date().toISOString(),
|
|
175
|
+
results,
|
|
176
|
+
summary: summarizeResults(results),
|
|
177
|
+
};
|
|
178
|
+
await fs.writeFile(jsonPath, `${JSON.stringify(payload, null, 2)}\n`, 'utf8');
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
function fmtPct(value) {
|
|
182
|
+
return value === null ? 'n/a' : `${(value * 100).toFixed(1)}%`;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
export function printScorecard(results) {
|
|
186
|
+
console.log('id | mode | expected | anchors | budget | verdict');
|
|
187
|
+
console.log('---|------|----------|---------|--------|--------');
|
|
188
|
+
for (const r of results) {
|
|
189
|
+
console.log(
|
|
190
|
+
`${r.id} | ${r.mode || '-'} | ${r.expectedMode || '-'} | ${r.anchors.length} | ${r.budgetUsed}/${r.maxTokens === Infinity ? '∞' : r.maxTokens} | ${r.pass ? 'PASS' : 'FAIL'}`,
|
|
191
|
+
);
|
|
192
|
+
for (const failure of r.failures) {
|
|
193
|
+
console.log(` ! ${failure}`);
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
const s = summarizeResults(results);
|
|
197
|
+
console.log(`scorecard: ${s.passed}/${s.total} passed`);
|
|
198
|
+
console.log(
|
|
199
|
+
`totals: passRate: ${fmtPct(s.passRate)} | modeAccuracy: ${fmtPct(s.modeAccuracy)} | anchorPrecision: ${fmtPct(s.anchorPrecision)} | anchorRecall: ${fmtPct(s.anchorRecall)} | meanBudgetUsedPct: ${fmtPct(s.meanBudgetUsedPct)}`,
|
|
200
|
+
);
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
|
|
204
|
+
if (isMain) {
|
|
205
|
+
const { corpusPath, projectRoot, jsonPath } = parseArgs(process.argv.slice(2));
|
|
206
|
+
try {
|
|
207
|
+
const { results, pass } = await runGold({ projectRoot, corpusPath });
|
|
208
|
+
printScorecard(results);
|
|
209
|
+
if (jsonPath) {
|
|
210
|
+
await writeScorecardJson(jsonPath, results);
|
|
211
|
+
console.log(`[bench] scorecard JSON written: ${jsonPath}`);
|
|
212
|
+
}
|
|
213
|
+
if (!pass) {
|
|
214
|
+
process.exitCode = 1;
|
|
215
|
+
}
|
|
216
|
+
} catch (error) {
|
|
217
|
+
console.error(`[bench] ${error?.message ?? error}`);
|
|
218
|
+
process.exitCode = 1;
|
|
219
|
+
}
|
|
220
|
+
}
|
|
@@ -45,6 +45,12 @@ const steps = [
|
|
|
45
45
|
args: ['test:artifact'],
|
|
46
46
|
env: process.env,
|
|
47
47
|
},
|
|
48
|
+
{
|
|
49
|
+
label: 'Doc contracts',
|
|
50
|
+
command: 'yarn',
|
|
51
|
+
args: ['vitest', 'run', 'tests/consistency/docContracts.test.js'],
|
|
52
|
+
env: process.env,
|
|
53
|
+
},
|
|
48
54
|
{
|
|
49
55
|
label: 'Core test suite',
|
|
50
56
|
command: 'yarn',
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
import { compileContext } from '../../core/codeintel/compiler.js';
|
|
2
|
+
import { packetToText, validatePacket } from '../../core/codeintel/packet.js';
|
|
3
|
+
import { IndexFileSyntaxProvider } from '../../core/codeintel/providers.js';
|
|
4
|
+
|
|
5
|
+
// `ukit code` — CLI surface over the codeintel plane (SPEC §11).
|
|
6
|
+
// Every subcommand emits a Context Packet v1: `--json` prints the raw packet,
|
|
7
|
+
// default prints `packetToText`. Unknown/missing subcommand → help + non-zero.
|
|
8
|
+
|
|
9
|
+
const HELP_FLAGS = new Set(['--help', '-h', 'help']);
|
|
10
|
+
const SUBCOMMANDS = new Set(['peek', 'search', 'context', 'impact']);
|
|
11
|
+
|
|
12
|
+
function extractFlag(args, flag) {
|
|
13
|
+
const index = args.indexOf(flag);
|
|
14
|
+
if (index < 0) return { value: null, rest: args };
|
|
15
|
+
return {
|
|
16
|
+
value: args[index + 1] ?? null,
|
|
17
|
+
rest: [...args.slice(0, index), ...args.slice(index + 2)],
|
|
18
|
+
};
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
function hasFlag(args, flag) {
|
|
22
|
+
return args.includes(flag);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
function parsePositiveInt(value, fallback) {
|
|
26
|
+
const parsed = Number.parseInt(value ?? '', 10);
|
|
27
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : fallback;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
function printPacket(packet, { json }) {
|
|
31
|
+
if (json) {
|
|
32
|
+
console.log(JSON.stringify(packet, null, 2));
|
|
33
|
+
return;
|
|
34
|
+
}
|
|
35
|
+
console.log(packetToText(packet));
|
|
36
|
+
const validation = validatePacket(packet);
|
|
37
|
+
if (!validation.ok) {
|
|
38
|
+
console.log(`[UKit] packet validation warnings: ${validation.errors.join('; ')}`);
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
async function printFileOutline(projectRoot, filePath) {
|
|
43
|
+
try {
|
|
44
|
+
const provider = new IndexFileSyntaxProvider({ rootDir: projectRoot });
|
|
45
|
+
if (!provider.supports(filePath)) {
|
|
46
|
+
return;
|
|
47
|
+
}
|
|
48
|
+
const symbols = await provider.symbols(filePath);
|
|
49
|
+
if (!Array.isArray(symbols) || symbols.length === 0) {
|
|
50
|
+
return;
|
|
51
|
+
}
|
|
52
|
+
console.log('');
|
|
53
|
+
console.log(`## Outline (${filePath})`);
|
|
54
|
+
for (const item of symbols) {
|
|
55
|
+
console.log(`${item.line ?? '?'}: ${item.name ?? item.kind ?? ''}${item.kind ? ` [${item.kind}]` : ''}`);
|
|
56
|
+
}
|
|
57
|
+
} catch {
|
|
58
|
+
// Outline is best-effort decoration; never fail the packet print for it.
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export async function runCode({ projectRoot, argv = [] }) {
|
|
63
|
+
const [subcommandRaw, ...rest] = argv;
|
|
64
|
+
const subcommand = (subcommandRaw ?? '').toLowerCase();
|
|
65
|
+
|
|
66
|
+
if (!subcommand || HELP_FLAGS.has(subcommand)) {
|
|
67
|
+
printCodeHelp();
|
|
68
|
+
if (!subcommand) {
|
|
69
|
+
process.exitCode = 1;
|
|
70
|
+
}
|
|
71
|
+
return;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
if (!SUBCOMMANDS.has(subcommand)) {
|
|
75
|
+
console.error(`[UKit] Unknown code subcommand: ${subcommand}`);
|
|
76
|
+
printCodeHelp();
|
|
77
|
+
process.exitCode = 1;
|
|
78
|
+
return;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
const json = hasFlag(rest, '--json');
|
|
82
|
+
const withoutJson = rest.filter((arg) => arg !== '--json');
|
|
83
|
+
|
|
84
|
+
if (subcommand === 'peek') {
|
|
85
|
+
const { value: lines, rest: args } = extractFlag(withoutJson, '--lines');
|
|
86
|
+
const filePath = args[0];
|
|
87
|
+
if (!filePath) {
|
|
88
|
+
console.error('[UKit] Missing file path. Usage: ukit code peek <path> [--lines a-b] [--json]');
|
|
89
|
+
process.exitCode = 1;
|
|
90
|
+
return;
|
|
91
|
+
}
|
|
92
|
+
const packet = await compileContext(projectRoot, {
|
|
93
|
+
prompt: filePath,
|
|
94
|
+
filesHinted: [filePath],
|
|
95
|
+
}, { mode: 'peek' });
|
|
96
|
+
printPacket(packet, { json });
|
|
97
|
+
if (!json) {
|
|
98
|
+
await printFileOutline(projectRoot, filePath);
|
|
99
|
+
if (lines) {
|
|
100
|
+
console.log(`(note: --lines ${lines} accepted; v1 outline prints full symbol list)`);
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
return;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
if (subcommand === 'search') {
|
|
107
|
+
const { value: limitArg, rest: args } = extractFlag(withoutJson, '--limit');
|
|
108
|
+
const limit = parsePositiveInt(limitArg, 20);
|
|
109
|
+
const query = args.join(' ').trim();
|
|
110
|
+
if (!query) {
|
|
111
|
+
console.error('[UKit] Missing query. Usage: ukit code search "<query>" [--limit n] [--json]');
|
|
112
|
+
process.exitCode = 1;
|
|
113
|
+
return;
|
|
114
|
+
}
|
|
115
|
+
const packet = await compileContext(projectRoot, query);
|
|
116
|
+
if (packet.evidence.length > limit) {
|
|
117
|
+
const dropped = packet.evidence.splice(limit);
|
|
118
|
+
for (const item of dropped) {
|
|
119
|
+
packet.omitted.push({ what: item.path ?? 'evidence', why: 'limit' });
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
if (packet.anchors.length > limit) {
|
|
123
|
+
packet.anchors = packet.anchors.slice(0, limit);
|
|
124
|
+
}
|
|
125
|
+
printPacket(packet, { json });
|
|
126
|
+
return;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
if (subcommand === 'context') {
|
|
130
|
+
const { value: mode, rest: afterMode } = extractFlag(withoutJson, '--mode');
|
|
131
|
+
const { value: budgetArg, rest: afterBudget } = extractFlag(afterMode, '--budget');
|
|
132
|
+
const budget = budgetArg !== null ? parsePositiveInt(budgetArg, null) : undefined;
|
|
133
|
+
const diagnostics = hasFlag(afterBudget, '--diagnostics');
|
|
134
|
+
const args = afterBudget.filter((arg) => arg !== '--diagnostics');
|
|
135
|
+
const task = args.join(' ').trim();
|
|
136
|
+
if (!task) {
|
|
137
|
+
console.error('[UKit] Missing task. Usage: ukit code context "<task>" [--mode m] [--budget n] [--diagnostics] [--json]');
|
|
138
|
+
process.exitCode = 1;
|
|
139
|
+
return;
|
|
140
|
+
}
|
|
141
|
+
const packet = await compileContext(projectRoot, task, {
|
|
142
|
+
mode: mode ?? undefined,
|
|
143
|
+
budget: budget ?? undefined,
|
|
144
|
+
diagnostics: diagnostics || undefined,
|
|
145
|
+
});
|
|
146
|
+
printPacket(packet, { json });
|
|
147
|
+
return;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// subcommand === 'impact'
|
|
151
|
+
const { value: depthArg, rest: impactArgs } = extractFlag(withoutJson, '--depth');
|
|
152
|
+
const depth = depthArg !== null ? parsePositiveInt(depthArg, null) : undefined;
|
|
153
|
+
const target = impactArgs[0];
|
|
154
|
+
if (!target) {
|
|
155
|
+
console.error('[UKit] Missing target. Usage: ukit code impact <path|symbol> [--depth N] [--json]');
|
|
156
|
+
process.exitCode = 1;
|
|
157
|
+
return;
|
|
158
|
+
}
|
|
159
|
+
const packet = await compileContext(projectRoot, {
|
|
160
|
+
prompt: target,
|
|
161
|
+
filesHinted: target.includes('/') || /\.[a-z0-9]+$/i.test(target) ? [target] : [],
|
|
162
|
+
}, { mode: 'impact', depth: depth ?? undefined });
|
|
163
|
+
printPacket(packet, { json });
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
export function printCodeHelp() {
|
|
167
|
+
console.log('UKit Code Commands (codeintel plane)');
|
|
168
|
+
console.log('Usage: ukit code <peek|search|context|impact> [args]');
|
|
169
|
+
console.log('');
|
|
170
|
+
console.log('Subcommands:');
|
|
171
|
+
console.log(' peek <path> [--lines a-b] L0 outline-only packet + file outline');
|
|
172
|
+
console.log(' search "<query>" [--limit n] Routed retrieval packet over the index');
|
|
173
|
+
console.log(' context "<task>" [--mode m] [--budget n] [--diagnostics] Compiled context packet for a task');
|
|
174
|
+
console.log(' impact <path|symbol> [--depth N] L2 packet with multi-hop reverse-import edges');
|
|
175
|
+
console.log('');
|
|
176
|
+
console.log('Options:');
|
|
177
|
+
console.log(' --json Print raw Context Packet JSON (default: text render)');
|
|
178
|
+
console.log(' --mode <m> Force router mode: none|peek|targeted|explore|impact|deep_flow|analogy');
|
|
179
|
+
console.log(' --budget <n> Override routed token budget');
|
|
180
|
+
console.log(' --depth <N> Impact-mode hop depth (clamped 1..codeIntel.impact.maxDepth)');
|
|
181
|
+
console.log(' --diagnostics Include post-edit diagnostics in next_actions');
|
|
182
|
+
}
|