@ngockhoale/ukit 2.6.6 → 2.6.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +37 -0
  2. package/README.md +40 -177
  3. package/manifests/documentation.yaml +143 -15
  4. package/manifests/hostCapabilities.yaml +49 -0
  5. package/manifests/instructionRules.yaml +383 -0
  6. package/manifests/platform.full.yaml +15 -0
  7. package/package.json +3 -1
  8. package/scripts/bench/goldTasks.json +38 -0
  9. package/scripts/bench/runGold.mjs +220 -0
  10. package/scripts/docs/render-instructions.mjs +42 -0
  11. package/scripts/release/verify-release.mjs +6 -0
  12. package/src/cli/commands/code.js +182 -0
  13. package/src/cli/commands/doctor.js +35 -3
  14. package/src/cli/commands/indexTools.js +102 -1
  15. package/src/cli/commands/memory.js +137 -0
  16. package/src/cli/index.js +7 -0
  17. package/src/core/codeintel/compiler.js +316 -0
  18. package/src/core/codeintel/diagnostics.js +114 -0
  19. package/src/core/codeintel/freshness.js +295 -0
  20. package/src/core/codeintel/impact.js +251 -0
  21. package/src/core/codeintel/invalidation.js +150 -0
  22. package/src/core/codeintel/manifest.js +176 -0
  23. package/src/core/codeintel/packet.js +146 -0
  24. package/src/core/codeintel/providers.js +201 -0
  25. package/src/core/codeintel/retriever.js +372 -0
  26. package/src/core/codeintel/router.js +149 -0
  27. package/src/core/codeintel/semanticProvider.js +235 -0
  28. package/src/core/docContracts.js +723 -0
  29. package/src/core/memory/migrate.js +324 -0
  30. package/src/core/memory/records.js +172 -0
  31. package/src/core/memory/retrieval.js +161 -11
  32. package/src/core/memory/store.js +398 -0
  33. package/src/core/memory/storeV2.js +171 -0
  34. package/src/core/memory/storeV2Loader.js +22 -0
  35. package/src/core/projectImportant.js +1 -1
  36. package/src/core/runtimeConfig.js +125 -0
  37. package/src/core/runtimePaths.js +3 -0
  38. package/src/core/uninstall.js +1 -1
  39. package/src/index/taskRouting.js +39 -0
  40. package/src/render/instructionRenderer.js +226 -0
  41. package/templates/.claude/ukit/index/route-task.mjs +40 -0
  42. package/templates/.gitignore +2 -2
  43. package/templates/.omp/RULES.md +1 -0
  44. package/templates/AGENTS.md +89 -218
  45. package/templates/CLAUDE.md +85 -212
  46. package/templates/docs/AI_HANDOFF/tasks/_TEMPLATE.md +5 -0
  47. package/templates/docs/BUGFIX.md +2 -19
  48. package/templates/docs/BUG_INDEX.md +43 -0
  49. package/templates/docs/BUG_METRICS.md +1 -5
  50. package/templates/docs/BUG_TEMPLATE.md +1 -11
  51. package/templates/docs/UKIT_INTERNALS.md +223 -0
  52. package/templates/instructions/core.md +157 -0
  53. package/templates/instructions/layout.yaml +149 -0
  54. package/templates/instructions/overlays/agents.md +15 -0
  55. package/templates/instructions/overlays/claude.md +3 -0
  56. package/templates/instructions/overlays/omp-rules.md +74 -0
  57. package/templates/instructions/overlays/repo.md +9 -0
  58. package/templates/instructions/repo-vars.yaml +23 -0
  59. package/templates/ukit/storage/config.json +30 -0
@@ -0,0 +1,383 @@
1
+ version: 1
2
+ generated: false
3
+ description: >-
4
+ Canonical rule-ID catalog for the root instruction contracts. Marker comments
5
+ <!-- RULE: ID --> tag normative behavior; semantic tests prove delivery.
6
+ # marker_in values are drawn from the fixed set of the four marker-bearing
7
+ # contract files {templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md}.
8
+ # templates/.omp/RULES.md and .omp/RULES.md are marker-free by SPEC §7.1 policy
9
+ # and must never appear here.
10
+ # DEVIATION from SPEC §7.2 table: OWN-01 is marker_in: AGENTS set only (not
11
+ # SHARED) — its anchor `PROJECT_IMPORTANT.md` does not exist in the CLAUDE
12
+ # files, and markers are comment-only (no prose edits allowed). AGENTS-set
13
+ # coverage matches the HOST-OC-01/HOST-OWN-01 rows that tag the same overlay.
14
+ # paths (optional, DOC-302): list of repo-relative glob strings naming the
15
+ # work-tree scope a rule applies to. `**` = any depth, `*` = one segment,
16
+ # `?` = one char; `/`-separated, no leading `/`, no `..`, no backslashes.
17
+ # Absent = file-scoped only (marker_in governs rendered coverage; paths is
18
+ # metadata for future router/doc tooling, consumed by rulesForPath).
19
+ rules:
20
+ - id: CORE-01
21
+ title: One remembered command — ukit install
22
+ kind: critical
23
+ statement: >-
24
+ Human-facing UKit workflow collapses to one remembered command:
25
+ `ukit install`; after install, work in natural language inside the
26
+ supported harnesses.
27
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
28
+ anchor: 'ukit install'
29
+ - id: CORE-02
30
+ title: Quality first, then speed, then token discipline
31
+ kind: supporting
32
+ statement: >-
33
+ Ordering of priorities for all work: quality first, then speed, then
34
+ token discipline.
35
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
36
+ - id: CLS-01
37
+ title: Trivial lane acts directly
38
+ kind: critical
39
+ statement: >-
40
+ Trivial tasks (typo, label, small rename, spacing, toggle flag, obvious
41
+ config change) act directly — no doc reads, planning, index, or agents.
42
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
43
+ anchor: 'Act directly'
44
+ - id: CLS-02
45
+ title: Simple lane pulls smallest useful context
46
+ kind: critical
47
+ statement: >-
48
+ Simple tasks (1-2 files, clear scope, existing pattern) are handled
49
+ directly, pulling only the smallest useful context.
50
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
51
+ anchor: 'smallest useful context'
52
+ - id: CLS-03
53
+ title: Non-trivial lane verifies harder
54
+ kind: supporting
55
+ statement: >-
56
+ Non-trivial/risky tasks read deeper, verify harder, and use the
57
+ index-first loop then skill activation then targeted verification.
58
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
59
+ - id: EXEC-01
60
+ title: Continue until write evidence or blocker
61
+ kind: critical
62
+ statement: >-
63
+ For explicit implement/apply/fix requests, continue until the actual edit
64
+ is made or a real blocker is found; never stop after read-only steps.
65
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
66
+ anchor: 'actual edit is made'
67
+ - id: EXEC-02
68
+ title: No done-claims without Edit/Write evidence
69
+ kind: critical
70
+ statement: >-
71
+ Completion wording requires concrete Edit/Write evidence in the current
72
+ turn, plus verification when the scope is risky.
73
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
74
+ anchor: 'concrete Edit/Write evidence'
75
+ - id: EXEC-03
76
+ title: Routed continuation is not a stopping point
77
+ kind: supporting
78
+ statement: >-
79
+ Routed states like pull-indexed-context or continuation-required are
80
+ internal continuation steps; finish the named milestone before widening.
81
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
82
+ - id: EXEC-04
83
+ title: Every stop names its reason
84
+ kind: critical
85
+ statement: >-
86
+ When a turn ends on a user-only action, open the reply with a
87
+ `WAITING ON YOU:` line and schedule a one-shot wakeup when available.
88
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
89
+ anchor: 'WAITING ON YOU'
90
+ - id: LONG-01
91
+ title: LAND one thing, DEFER the rest, DELEGATE broad work
92
+ kind: supporting
93
+ statement: >-
94
+ Near token-cap, land one in-flight item end-to-end, defer the rest into
95
+ docs/STATUS.md or bounded handoff tasks, delegate broad work to subagents.
96
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
97
+ - id: LONG-02
98
+ title: Post-compact continuity from disk state
99
+ kind: supporting
100
+ statement: >-
101
+ After any compact or handoff, continue from persisted disk state without
102
+ rereading pre-compact context; delegate and keep replies short.
103
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
104
+ - id: IDX-01
105
+ title: Check index freshness before code context
106
+ kind: critical
107
+ statement: >-
108
+ For tasks needing code context, check the index is fresh and refresh it
109
+ when stale or missing.
110
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
111
+ anchor: 'index is fresh'
112
+ - id: IDX-02
113
+ title: Open top 1-3 suspect files first
114
+ kind: critical
115
+ statement: >-
116
+ Open only the top 1-3 suspect files first, then widen if needed; use the
117
+ outline block to jump straight to the relevant region.
118
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
119
+ anchor: 'top 1-3 suspect files'
120
+ - id: IDX-03
121
+ title: Outlines locate code; code to be changed must still be read
122
+ kind: critical
123
+ statement: >-
124
+ The outline locates code but does not describe behaviour; any code about
125
+ to be changed must still be read.
126
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
127
+ anchor: 'must still be read'
128
+ - id: SKILL-01
129
+ title: Auto-activate the matching skill
130
+ kind: critical
131
+ statement: >-
132
+ On every non-trivial task, inspect installed project-local skills and
133
+ auto-activate the matching skill immediately; end users should not need
134
+ to know skill names.
135
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
136
+ anchor: 'auto-activate the matching skill'
137
+ - id: SKILL-02
138
+ title: Smallest effective skill set
139
+ kind: supporting
140
+ statement: >-
141
+ Use the smallest effective set of skills, usually 1-2.
142
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
143
+ - id: SKILL-03
144
+ title: Upgrade skill choice on sharper evidence
145
+ kind: supporting
146
+ statement: >-
147
+ If evidence becomes more specific than the original prompt, upgrade the
148
+ active skill choice immediately.
149
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
150
+ - id: HELP-01
151
+ title: Prefer internal helper commands for routing/context/verify
152
+ kind: supporting
153
+ statement: >-
154
+ Prefer the index helper scripts (route-task, resolve-context,
155
+ verify-context) for routing, related-file context, and verification.
156
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
157
+ - id: HELP-02
158
+ title: Never ask contributors to run internal helpers
159
+ kind: critical
160
+ statement: >-
161
+ Do not ask normal contributors to run internal helper commands or
162
+ memorize maintainer commands; run them yourself or tell them to rerun
163
+ `ukit install`.
164
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
165
+ anchor: 'Do not ask normal contributors'
166
+ - id: OWN-01
167
+ title: PROJECT_IMPORTANT.md is the canonical owner-instruction source
168
+ kind: critical
169
+ statement: >-
170
+ Read and follow the root PROJECT_IMPORTANT.md before project work; it is
171
+ the canonical project-owner instruction source.
172
+ marker_in: [templates/AGENTS.md, AGENTS.md]
173
+ anchor: 'PROJECT_IMPORTANT.md'
174
+ - id: SAFE-01
175
+ title: Preserve byte-level file details on edit
176
+ kind: critical
177
+ statement: >-
178
+ Preserve UTF-8 BOM/no-BOM and LF/CRLF for existing multilingual or
179
+ user-authored files.
180
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
181
+ anchor: 'BOM/no-BOM'
182
+ - id: SAFE-02
183
+ title: Prefer unique current-file anchors for risky edits
184
+ kind: critical
185
+ statement: >-
186
+ For risky/shared/large edits, prefer unique current-file anchors over
187
+ line numbers or stale pasted blocks.
188
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
189
+ anchor: 'unique current-file anchors'
190
+ - id: SAFE-03
191
+ title: Ask on missing or ambiguous stale specs
192
+ kind: supporting
193
+ statement: >-
194
+ Do not silently merge stale specs: if old_string is missing or ambiguous,
195
+ re-read current source and ask whether to apply, adapt, or skip.
196
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
197
+ - id: CTX-01
198
+ title: Deterministic segment bytes
199
+ kind: critical
200
+ statement: >-
201
+ Keep prompt-segment bytes deterministic so provider prompt prefixes can
202
+ be reused.
203
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
204
+ anchor: 'CTX-01'
205
+ - id: CTX-02
206
+ title: Keep roles and order
207
+ kind: critical
208
+ statement: >-
209
+ Keep message roles and order stable across calls.
210
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
211
+ anchor: 'CTX-02'
212
+ - id: CTX-03
213
+ title: Keep tool IDs and continuation state
214
+ kind: critical
215
+ statement: >-
216
+ Keep tool IDs and continuation state stable for prefix reuse.
217
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
218
+ anchor: 'CTX-03'
219
+ - id: CTX-04
220
+ title: No clock/random IDs in static blocks
221
+ kind: critical
222
+ statement: >-
223
+ Static prompt blocks must not contain clock values or random IDs.
224
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
225
+ anchor: 'CTX-04'
226
+ - id: CTX-05
227
+ title: Compaction starts a new epoch
228
+ kind: critical
229
+ statement: >-
230
+ Compaction starts a new cache epoch; do not expect prefix reuse across it.
231
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
232
+ anchor: 'CTX-05'
233
+ - id: CTX-06
234
+ title: Never change data to match a cache
235
+ kind: critical
236
+ statement: >-
237
+ Never mutate data to make it match a cache entry.
238
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
239
+ anchor: 'CTX-06'
240
+ - id: CTX-07
241
+ title: No unconfirmed cache fields
242
+ kind: critical
243
+ statement: >-
244
+ Do not rely on cache fields that have not been confirmed for the route in
245
+ use.
246
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
247
+ anchor: 'CTX-07'
248
+ - id: CTX-08
249
+ title: Tool-result reuse needs valid freshness
250
+ kind: critical
251
+ statement: >-
252
+ Reuse of tool results requires valid freshness.
253
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
254
+ anchor: 'CTX-08'
255
+ - id: CTX-09
256
+ title: Missing usage is unknown, not zero
257
+ kind: critical
258
+ statement: >-
259
+ Missing cache-usage reporting means unknown, not zero savings.
260
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
261
+ anchor: 'CTX-09'
262
+ - id: CTX-10
263
+ title: Never cut a required check to reduce calls
264
+ kind: critical
265
+ statement: >-
266
+ Never skip a required verification or check just to reduce call count.
267
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
268
+ anchor: 'CTX-10'
269
+ - id: HAND-01
270
+ title: Handoff Quality Gate is opt-in via docs/AI_HANDOFF
271
+ kind: supporting
272
+ statement: >-
273
+ The handoff quality gate activates only when a task goes through
274
+ docs/AI_HANDOFF; daily prompts keep the old flow.
275
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
276
+ paths: [docs/AI_HANDOFF/**]
277
+ - id: HAND-02
278
+ title: RUN.md is the authoritative handoff run cursor
279
+ kind: critical
280
+ statement: >-
281
+ docs/AI_HANDOFF/RUN.md is the authoritative run cursor; only the
282
+ HANDOFF FULLSTACK COMPLETE/BLOCKED markers after Phase done/blocked end
283
+ a run.
284
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
285
+ anchor: 'RUN.md'
286
+ paths: [docs/AI_HANDOFF/**]
287
+ - id: BUDGET-01
288
+ title: Context budget per task class
289
+ kind: critical
290
+ statement: >-
291
+ Context + verification budget rows: Trivial reads no docs, Simple reads
292
+ MEMORY only, Non-trivial adds PROJECT and CODE_MAP.
293
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
294
+ anchor: '**Trivial**: no docs'
295
+ - id: BUDGET-02
296
+ title: STATUS.md only for open-ended or continuation prompts
297
+ kind: supporting
298
+ statement: >-
299
+ docs/STATUS.md is used for open-ended status/continue prompts or
300
+ meaningful continuation context; stale status is orientation only.
301
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
302
+ - id: BUDGET-03
303
+ title: WORKLOG recent entries only
304
+ kind: supporting
305
+ statement: >-
306
+ docs/WORKLOG.md is read for recent relevant entries only; archive oldest
307
+ entries when over limits.
308
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
309
+ - id: STATUS-01
310
+ title: STATUS.md is not source truth
311
+ kind: supporting
312
+ statement: >-
313
+ docs/STATUS.md captures compact current state; it is not source truth and
314
+ must not replace source/index-first investigation.
315
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
316
+ - id: SUBAG-01
317
+ title: Delegate only on meaningful context or parallel gains
318
+ kind: supporting
319
+ statement: >-
320
+ Keep direct execution as default for trivial/simple work; delegate only
321
+ when it shrinks context or enables useful parallel progress.
322
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
323
+ - id: SUBAG-02
324
+ title: Small-task maintainer is a sidecar lane
325
+ kind: supporting
326
+ statement: >-
327
+ The ukit-small-task-maintainer subagent handles safe/reversible UKit
328
+ chores as a sidecar lane and hands risky work back to the main model.
329
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
330
+ - id: AUTO-01
331
+ title: autonomy.level controls how much UKit acts without asking
332
+ kind: supporting
333
+ statement: >-
334
+ autonomy.level in .ukit/storage/config.json controls how much UKit acts
335
+ without asking first; end users should not need to change it.
336
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
337
+ - id: TIER-01
338
+ title: Tiers bind through agent definitions
339
+ kind: critical
340
+ statement: >-
341
+ A model tier takes effect only when work is handed to an agent whose own
342
+ definition binds that model; the main session model never changes
343
+ mid-turn.
344
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
345
+ anchor: 'model:'
346
+ - id: TIER-02
347
+ title: Escalate one tier after repeated failure
348
+ kind: supporting
349
+ statement: >-
350
+ When the same file or symbol fails debugLoopThreshold times in one
351
+ session, route the next attempt one tier higher, capped at smart.
352
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
353
+ - id: HOST-OC-01
354
+ title: OpenCode must explicitly read the triggered SKILL.md
355
+ kind: critical
356
+ statement: >-
357
+ OpenCode reads AGENTS.md at session start only and does not auto-load
358
+ skills; the model must explicitly read the triggered SKILL.md.
359
+ marker_in: [templates/AGENTS.md, AGENTS.md]
360
+ anchor: 'read the triggered SKILL.md'
361
+ - id: HOST-OWN-01
362
+ title: Codex/OpenCode read the root owner instructions
363
+ kind: critical
364
+ statement: >-
365
+ When running in Codex or OpenCode, read and follow the root
366
+ PROJECT_IMPORTANT.md before doing project work.
367
+ marker_in: [templates/AGENTS.md, AGENTS.md]
368
+ anchor: 'read and follow the root'
369
+ - id: DURA-01
370
+ title: DuraOne skill activates only when installed
371
+ kind: supporting
372
+ statement: >-
373
+ The DuraOne skill is active only when the duraone pack is installed or
374
+ .claude/skills/duraone/SKILL.md exists; otherwise use generic standards.
375
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
376
+ - id: FALLBACK-01
377
+ title: Missing/corrupt runtime → rerun ukit install
378
+ kind: critical
379
+ statement: >-
380
+ If the workspace needs a refresh or runtime files are missing or corrupt,
381
+ tell maintainers to rerun `ukit install`.
382
+ marker_in: [templates/CLAUDE.md, templates/AGENTS.md, CLAUDE.md, AGENTS.md]
383
+ anchor: 'rerun `ukit install`'
@@ -246,6 +246,21 @@ items:
246
246
  packs:
247
247
  - core
248
248
 
249
+ # Shipped internal-orchestration detail (tier tables, subagent policy, sidecar review,
250
+ # small-task maintainer, helper/runtime/budget/status detail, DuraOne). Extracted from
251
+ # templates/instructions/core.md in DOC-104; `overwrite_with_backup` so `ukit update`
252
+ # refreshes it alongside the root contracts that point at it.
253
+ - id: docs-ukit-internals
254
+ type: config
255
+ sourceTemplate: docs/UKIT_INTERNALS.md
256
+ targetPath: docs/UKIT_INTERNALS.md
257
+ requires: []
258
+ mergeStrategy: overwrite_with_backup
259
+ variables: []
260
+ enabledByDefault: true
261
+ packs:
262
+ - core
263
+
249
264
  - id: core-skill-delivery
250
265
  type: skill
251
266
  sourceTemplate: .claude/skills/delivery/SKILL.md
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ngockhoale/ukit",
3
- "version": "2.6.6",
3
+ "version": "2.6.8",
4
4
  "description": "Install/update an index-first AI workspace for Claude Code, OpenAI Codex, OpenCode, and omp (Oh My Pi).",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -55,6 +55,8 @@
55
55
  "index:query": "node ./scripts/index/query-index.mjs",
56
56
  "bug:triage": "node ./scripts/bug/triage.mjs",
57
57
  "skill:audit": "node ./scripts/skill/audit-skill.mjs",
58
+ "docs:render": "node scripts/docs/render-instructions.mjs --write",
59
+ "docs:render:check": "node scripts/docs/render-instructions.mjs --check",
58
60
  "test:artifact": "vitest run tests/integration/packageArtifact.test.js tests/integration/artifactReferenceIntegrity.test.js",
59
61
  "test:release-core": "vitest run --exclude tests/integration/packageArtifact.test.js",
60
62
  "release:verify": "node ./scripts/release/verify-release.mjs",
@@ -0,0 +1,38 @@
1
+ [
2
+ {
3
+ "id": "gold-001-trivial-none",
4
+ "prompt": "fix typo in the header label",
5
+ "expectedMode": "none",
6
+ "expectedAnchors": [],
7
+ "maxTokens": 500
8
+ },
9
+ {
10
+ "id": "gold-002-peek",
11
+ "prompt": "peek src/cli/index.js",
12
+ "flags": { "mode": "peek" },
13
+ "expectedMode": "peek",
14
+ "expectedAnchors": ["src/cli/index.js"],
15
+ "maxTokens": 500
16
+ },
17
+ {
18
+ "id": "gold-003-targeted",
19
+ "prompt": "update src/cli/index.js to add a code command entry",
20
+ "expectedMode": "targeted",
21
+ "expectedAnchors": ["src/cli/index.js"],
22
+ "maxTokens": 2000
23
+ },
24
+ {
25
+ "id": "gold-004-impact",
26
+ "prompt": "build fails with TypeError in src/index/buildIndex.js",
27
+ "expectedMode": "impact",
28
+ "expectedAnchors": ["src/index/buildIndex.js"],
29
+ "maxTokens": 3000
30
+ },
31
+ {
32
+ "id": "gold-005-explore",
33
+ "prompt": "where is index freshness handled",
34
+ "expectedMode": "explore",
35
+ "expectedAnchors": [],
36
+ "maxTokens": 4000
37
+ }
38
+ ]
@@ -0,0 +1,220 @@
1
+ #!/usr/bin/env node
2
+ // Gold-task benchmark harness skeleton (SPEC §10).
3
+ // Loads a corpus of {id, prompt, expectedMode, expectedAnchors[], maxTokens}
4
+ // cases, runs routeTask + compileContext per case, prints a scorecard with a
5
+ // totals block, optionally writes an aggregated JSON scorecard (--json <path>),
6
+ // and exits 0 only when every case passes. CI-3xx gates on the JSON summary.
7
+
8
+ import fs from 'node:fs/promises';
9
+ import path from 'node:path';
10
+ import { fileURLToPath } from 'node:url';
11
+
12
+ const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..');
13
+ const DEFAULT_CORPUS = path.join(REPO_ROOT, 'scripts', 'bench', 'goldTasks.json');
14
+
15
+ function parseArgs(argv) {
16
+ let corpusPath = DEFAULT_CORPUS;
17
+ let projectRoot = REPO_ROOT;
18
+ let jsonPath = null;
19
+ for (let i = 0; i < argv.length; i += 1) {
20
+ if (argv[i] === '--corpus') {
21
+ corpusPath = path.resolve(argv[i + 1] ?? corpusPath);
22
+ i += 1;
23
+ } else if (argv[i] === '--project') {
24
+ projectRoot = path.resolve(argv[i + 1] ?? projectRoot);
25
+ i += 1;
26
+ } else if (argv[i] === '--json') {
27
+ jsonPath = argv[i + 1] ? path.resolve(argv[i + 1]) : null;
28
+ i += 1;
29
+ }
30
+ }
31
+ return { corpusPath, projectRoot, jsonPath };
32
+ }
33
+
34
+ async function loadCorpus(corpusPath) {
35
+ const raw = await fs.readFile(corpusPath, 'utf8');
36
+ const cases = JSON.parse(raw);
37
+ if (!Array.isArray(cases)) {
38
+ throw new Error(`bench corpus must be a JSON array: ${corpusPath}`);
39
+ }
40
+ return cases;
41
+ }
42
+
43
+ function anchorPathsOf(packet) {
44
+ const fromAnchors = (packet.anchors ?? []).map((a) => a.path).filter(Boolean);
45
+ const fromEvidence = (packet.evidence ?? []).map((e) => e.path).filter(Boolean);
46
+ return new Set([...fromAnchors, ...fromEvidence]);
47
+ }
48
+
49
+ function normalizeMode(mode) {
50
+ return typeof mode === 'string' && mode.trim() !== '' ? mode.trim() : null;
51
+ }
52
+
53
+ export async function runGoldCase(projectRoot, testCase, { compileContext, routeTask }) {
54
+ const failures = [];
55
+ const expectedMode = normalizeMode(testCase.expectedMode);
56
+ const expectedAnchors = Array.isArray(testCase.expectedAnchors) ? testCase.expectedAnchors : [];
57
+ const maxTokens = typeof testCase.maxTokens === 'number' ? testCase.maxTokens : Infinity;
58
+
59
+ const routed = routeTask({
60
+ prompt: testCase.prompt ?? '',
61
+ hasError: testCase.hasError === true,
62
+ filesHinted: Array.isArray(testCase.filesHinted) ? testCase.filesHinted : undefined,
63
+ flags: testCase.flags && typeof testCase.flags === 'object' ? testCase.flags : {},
64
+ });
65
+
66
+ const packet = await compileContext(projectRoot, {
67
+ prompt: testCase.prompt ?? '',
68
+ hasError: testCase.hasError === true,
69
+ filesHinted: Array.isArray(testCase.filesHinted) ? testCase.filesHinted : undefined,
70
+ flags: testCase.flags && typeof testCase.flags === 'object' ? testCase.flags : {},
71
+ });
72
+
73
+ if (expectedMode && packet.task_type !== expectedMode) {
74
+ failures.push(`mode: expected ${expectedMode}, got ${packet.task_type || '(empty)'}`);
75
+ }
76
+
77
+ const anchorPaths = anchorPathsOf(packet);
78
+ const anchorHits = [];
79
+ const anchorMisses = [];
80
+ for (const expected of expectedAnchors) {
81
+ if (anchorPaths.has(expected)) {
82
+ anchorHits.push(expected);
83
+ } else {
84
+ anchorMisses.push(expected);
85
+ failures.push(`anchor missing: ${expected}`);
86
+ }
87
+ }
88
+
89
+ if (packet.budget?.used > maxTokens) {
90
+ failures.push(`budget: used ${packet.budget.used} > maxTokens ${maxTokens}`);
91
+ }
92
+
93
+ return {
94
+ id: testCase.id ?? '(unnamed)',
95
+ mode: packet.task_type,
96
+ expectedMode,
97
+ anchors: [...anchorPaths],
98
+ expectedAnchors,
99
+ anchorHits,
100
+ anchorMisses,
101
+ budgetUsed: packet.budget?.used ?? 0,
102
+ maxTokens,
103
+ routedMode: routed.mode,
104
+ pass: failures.length === 0,
105
+ failures,
106
+ };
107
+ }
108
+
109
+ export async function runGold({ projectRoot, corpusPath } = {}) {
110
+ const rootDir = projectRoot ?? REPO_ROOT;
111
+ const { compileContext } = await import(path.join(REPO_ROOT, 'src/core/codeintel/compiler.js'));
112
+ const { routeTask } = await import(path.join(REPO_ROOT, 'src/core/codeintel/router.js'));
113
+ const { buildCodeIndex } = await import(path.join(REPO_ROOT, 'src/index/buildIndex.js'));
114
+ const { INDEX_ARTIFACTS, getArtifactPath } = await import(path.join(REPO_ROOT, 'src/index/paths.js'));
115
+
116
+ // The bench needs an index to score anchors; build once when missing.
117
+ try {
118
+ await fs.stat(getArtifactPath(rootDir, INDEX_ARTIFACTS.files));
119
+ } catch {
120
+ console.log('[bench] index missing — building code index first');
121
+ await buildCodeIndex({ rootDir });
122
+ }
123
+
124
+ const cases = await loadCorpus(corpusPath ?? DEFAULT_CORPUS);
125
+ const results = [];
126
+ for (const testCase of cases) {
127
+ results.push(await runGoldCase(rootDir, testCase, { compileContext, routeTask }));
128
+ }
129
+ return { results, pass: results.every((r) => r.pass) };
130
+ }
131
+
132
+ /**
133
+ * Aggregate per-case results into a machine-readable summary for CI gating.
134
+ * Pooled anchor precision = hits/found, recall = hits/expected; cases with
135
+ * empty expectedAnchors contribute 0 to both denominators (excluded, not 1).
136
+ * @param {Array<object>} results per-case runGoldCase outputs
137
+ * @returns {{total:number, passed:number, passRate:number, modeAccuracy:number|null, anchorPrecision:number|null, anchorRecall:number|null, meanBudgetUsedPct:number|null}}
138
+ */
139
+ export function summarizeResults(results) {
140
+ const total = results.length;
141
+ const passed = results.filter((r) => r.pass).length;
142
+ const passRate = total === 0 ? 0 : passed / total;
143
+
144
+ const withMode = results.filter((r) => r.expectedMode);
145
+ const modeCorrect = withMode.filter((r) => r.mode === r.expectedMode).length;
146
+ const modeAccuracy = withMode.length === 0 ? null : modeCorrect / withMode.length;
147
+
148
+ let hits = 0;
149
+ let found = 0;
150
+ let expected = 0;
151
+ for (const r of results) {
152
+ hits += Array.isArray(r.anchorHits) ? r.anchorHits.length : 0;
153
+ found += Array.isArray(r.anchors) ? r.anchors.length : 0;
154
+ expected += Array.isArray(r.expectedAnchors) ? r.expectedAnchors.length : 0;
155
+ }
156
+ const anchorPrecision = found === 0 ? null : hits / found;
157
+ const anchorRecall = expected === 0 ? null : hits / expected;
158
+
159
+ const budgeted = results.filter((r) => typeof r.maxTokens === 'number' && r.maxTokens > 0);
160
+ const meanBudgetUsedPct = budgeted.length === 0
161
+ ? null
162
+ : budgeted.reduce((acc, r) => acc + (r.budgetUsed ?? 0) / r.maxTokens, 0) / budgeted.length;
163
+
164
+ return { total, passed, passRate, modeAccuracy, anchorPrecision, anchorRecall, meanBudgetUsedPct };
165
+ }
166
+
167
+ /**
168
+ * Write { generatedAt, results, summary } scorecard JSON to `jsonPath` (pretty-printed).
169
+ * @param {string} jsonPath output file path
170
+ * @param {Array<object>} results per-case runGoldCase outputs
171
+ */
172
+ export async function writeScorecardJson(jsonPath, results) {
173
+ const payload = {
174
+ generatedAt: new Date().toISOString(),
175
+ results,
176
+ summary: summarizeResults(results),
177
+ };
178
+ await fs.writeFile(jsonPath, `${JSON.stringify(payload, null, 2)}\n`, 'utf8');
179
+ }
180
+
181
+ function fmtPct(value) {
182
+ return value === null ? 'n/a' : `${(value * 100).toFixed(1)}%`;
183
+ }
184
+
185
+ export function printScorecard(results) {
186
+ console.log('id | mode | expected | anchors | budget | verdict');
187
+ console.log('---|------|----------|---------|--------|--------');
188
+ for (const r of results) {
189
+ console.log(
190
+ `${r.id} | ${r.mode || '-'} | ${r.expectedMode || '-'} | ${r.anchors.length} | ${r.budgetUsed}/${r.maxTokens === Infinity ? '∞' : r.maxTokens} | ${r.pass ? 'PASS' : 'FAIL'}`,
191
+ );
192
+ for (const failure of r.failures) {
193
+ console.log(` ! ${failure}`);
194
+ }
195
+ }
196
+ const s = summarizeResults(results);
197
+ console.log(`scorecard: ${s.passed}/${s.total} passed`);
198
+ console.log(
199
+ `totals: passRate: ${fmtPct(s.passRate)} | modeAccuracy: ${fmtPct(s.modeAccuracy)} | anchorPrecision: ${fmtPct(s.anchorPrecision)} | anchorRecall: ${fmtPct(s.anchorRecall)} | meanBudgetUsedPct: ${fmtPct(s.meanBudgetUsedPct)}`,
200
+ );
201
+ }
202
+
203
+ const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
204
+ if (isMain) {
205
+ const { corpusPath, projectRoot, jsonPath } = parseArgs(process.argv.slice(2));
206
+ try {
207
+ const { results, pass } = await runGold({ projectRoot, corpusPath });
208
+ printScorecard(results);
209
+ if (jsonPath) {
210
+ await writeScorecardJson(jsonPath, results);
211
+ console.log(`[bench] scorecard JSON written: ${jsonPath}`);
212
+ }
213
+ if (!pass) {
214
+ process.exitCode = 1;
215
+ }
216
+ } catch (error) {
217
+ console.error(`[bench] ${error?.message ?? error}`);
218
+ process.exitCode = 1;
219
+ }
220
+ }