@ngockhoale/ukit 3.0.1 → 3.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/manifests/documentation.yaml +11 -0
- package/manifests/engineConformance.yaml +32 -0
- package/manifests/hostCapabilities.yaml +1 -1
- package/manifests/platform.full.yaml +26 -0
- package/package.json +1 -1
- package/src/cli/commands/doctor.js +38 -2
- package/src/core/executionContracts.js +40 -0
- package/src/core/handoffDocValidator.js +152 -0
- package/src/core/reviewPanelAggregate.js +120 -0
- package/src/index/taskRouting.js +50 -1
- package/template_project/.claude/agents/code-reviewer.md +25 -0
- package/template_project/.claude/agents/handoff-planner.md +15 -0
- package/template_project/.claude/commands/ukit/handoff-fullstack.md +25 -5
- package/template_project/.claude/commands/ukit/handoff-review.md +31 -6
- package/template_project/.claude/ukit/index/handoff-doc-validator.mjs +193 -0
- package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +174 -0
- package/template_project/.claude/ukit/index/route-task.mjs +91 -3
- package/template_project/.claude/ukit/runtime/execution-ledger.mjs +4 -1
- package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +187 -8
- package/template_project/.omp/agents/code-reviewer.md +27 -0
- package/template_project/.omp/agents/handoff-planner.md +15 -0
- package/template_project/ukit/storage/config.json +3 -3
|
@@ -59,7 +59,7 @@ resumable. If compaction does happen, the run resumes from the cursor automatica
|
|
|
59
59
|
|
|
60
60
|
### Run cursor — write after every step
|
|
61
61
|
|
|
62
|
-
Maintain `docs/AI_HANDOFF/RUN.md`. Rewrite it (whole file,
|
|
62
|
+
Maintain `docs/AI_HANDOFF/RUN.md`. Rewrite it (whole file, 7 lines) immediately after every
|
|
63
63
|
numbered step completes:
|
|
64
64
|
|
|
65
65
|
```
|
|
@@ -69,9 +69,15 @@ Base: <BASE branch>
|
|
|
69
69
|
Phase: <P0|P1|P2|P2.5|P3|I1|I2|I3|I4|R1|R2|R3|R4|R4.5|R5|F|done|blocked>
|
|
70
70
|
Cursor: wave <N> batch <M> — <what just finished>
|
|
71
71
|
Next: <the exact next step to run>
|
|
72
|
+
ExitPredicate: all(<term>[,<term>...]) # written once at P2; never relaxed
|
|
72
73
|
QuietScans: <n>/<required> # only while sweeping for stragglers near the end
|
|
73
74
|
```
|
|
74
75
|
|
|
76
|
+
`ExitPredicate` grammar: `all(<term>[,<term>...])` or a single term; terms are
|
|
77
|
+
`index:all-done`, `git:clean`, `file-exists:<relpath>`. The stop gate evaluates it
|
|
78
|
+
mechanically when the cursor claims `Phase: done` — a `done` claim with failing terms
|
|
79
|
+
is bounced back with the per-term state. Keep the line verbatim on every rewrite.
|
|
80
|
+
|
|
75
81
|
This file is the resume contract. It costs one small write per step and is what turns an
|
|
76
82
|
interrupted run into a continuable one.
|
|
77
83
|
|
|
@@ -172,9 +178,12 @@ The run must keep itself alive without the human watching.
|
|
|
172
178
|
|
|
173
179
|
- **Stop gate (Claude Code, automatic):** while `RUN.md` `Phase:` is not `done`/`blocked`,
|
|
174
180
|
`completion-gate.sh` refuses the stop with the cursor's `Next:` step. Recaps and
|
|
175
|
-
premature stops cannot park the run.
|
|
181
|
+
premature stops cannot park the run. When RUN.md carries `ExitPredicate:`, a
|
|
182
|
+
`Phase: done` whose predicate fails is refused too — `done` means the predicate
|
|
183
|
+
evaluates true. A stalled-cursor breaker
|
|
176
184
|
(`handoff.fullstack.stopGateMaxStalledBlocks`, default 12) releases the session if the
|
|
177
|
-
cursor has genuinely stopped advancing — the escape hatch, not the norm
|
|
185
|
+
cursor has genuinely stopped advancing — the escape hatch, not the norm — and its
|
|
186
|
+
release advisory reports the predicate state, so a plateau is visible, not silent.
|
|
178
187
|
- **Scheduled wakeup (when the harness offers one):** arm it at run start — Claude Code:
|
|
179
188
|
`CronCreate` a ~`handoff.fullstack.idleWatchdogMin`-minute (default 5) session job with
|
|
180
189
|
prompt `Resume HANDOFF FULLSTACK from docs/AI_HANDOFF/RUN.md — continue the Next: step`;
|
|
@@ -269,7 +278,13 @@ The planner agent does the following (use P1 summary — do NOT re-read files):
|
|
|
269
278
|
```
|
|
270
279
|
Then **verify the write** (read back the `Status:` line).
|
|
271
280
|
|
|
272
|
-
8. **
|
|
281
|
+
8. **Write `ExitPredicate:` into `docs/AI_HANDOFF/RUN.md`** — the mechanical definition of
|
|
282
|
+
done for this run: `all(<term>[,<term>...])` over `index:all-done`, `git:clean`,
|
|
283
|
+
`file-exists:<relpath>` (default `all(index:all-done, git:clean)`). Written once,
|
|
284
|
+
before iteration 1, and **never relaxed** — if the run cannot satisfy it, the plan is
|
|
285
|
+
wrong, not the gate. Then **verify the write** (read back the `ExitPredicate:` line).
|
|
286
|
+
|
|
287
|
+
9. **Report:** task IDs, dependency graph, recovery/superseded tasks, any
|
|
273
288
|
`needs_breakdown` tasks + reason.
|
|
274
289
|
|
|
275
290
|
### P2.5 — Independent plan review (strong model, separate agent)
|
|
@@ -455,7 +470,12 @@ Worktrees are **always deleted immediately** — no exceptions.
|
|
|
455
470
|
A wave boundary is the only safe place to shed context, because everything of value is
|
|
456
471
|
already on disk (task files hold the full logs, git holds the code). Do all four, in order:
|
|
457
472
|
|
|
458
|
-
1. **Checkpoint the code
|
|
473
|
+
1. **Checkpoint the code — commit-if-advanced.** One commit per wave, so every wave is
|
|
474
|
+
independently revertible — but only when the wave's diff advances the run's
|
|
475
|
+
`ExitPredicate` (a task moved toward `done`, a required file appeared, the worktree
|
|
476
|
+
got cleaner). A wave whose diff does not advance the predicate is **discarded**: its
|
|
477
|
+
copied-back changes are reverted and nothing is committed. A plateau is not a stop —
|
|
478
|
+
the run continues to the next wave or fix round; the predicate is never relaxed.
|
|
459
479
|
```bash
|
|
460
480
|
git add -A && git commit -m "handoff: wave <N> — TASK-00x, TASK-00y"
|
|
461
481
|
```
|
|
@@ -54,9 +54,11 @@ If that diff is empty → handoff-implement was not completed. Report which task
|
|
|
54
54
|
|
|
55
55
|
**Batch the review set — mandatory.** Read `handoff.maxParallelAgents` from `.ukit/storage/config.json`. If more `pending_review` tasks exist than that, split into consecutive batches of at most that many; finish one batch's verdicts (2a–2d, appended to each task file) before starting the next.
|
|
56
56
|
|
|
57
|
-
**
|
|
57
|
+
**Spawn the review panel — MANDATORY, do this before anything else:** read `modelRoles['review-panel']` from `.ukit/storage/config.json` → `modelRoles` (default `['smart','code','lite']`; omp tier aliases `@smol`/`@default`/`@slow`). For each `pending_review` task in the batch, call the Agent tool **once per panel entry** (in parallel, bounded by `handoff.maxParallelAgents`), each with `subagent_type: "code-reviewer"` (omp: the `task` tool with `agent: "code-reviewer"`, likewise invoked once per panel entry per task) — each member bound to its panel tier and setting `PANEL_MEMBER: <tier>` in its verdict. Reviewer agents only read the diff and append a verdict to their own task file — no worktree, no shared write target — so running them in parallel carries none of Phase 3's file-conflict risk. Do NOT review the diff yourself in the current session — this step is contracted to the panel tiers and MUST differ from the executor's model, which only the spawned agent's frontmatter model guarantees. Pass each agent: the task file path, the executor's report, and the diff.
|
|
58
58
|
|
|
59
|
-
|
|
59
|
+
Degenerate panel: if `modelRoles['review-panel']` is empty or has one entry, spawn a single reviewer and require `PANEL_MEMBER: solo` in its verdict.
|
|
60
|
+
|
|
61
|
+
Each spawned reviewer agent performs 2a–2d below per task:
|
|
60
62
|
|
|
61
63
|
### 2a — Model isolation check (always first)
|
|
62
64
|
|
|
@@ -98,6 +100,7 @@ matches the task file's wording.
|
|
|
98
100
|
VERDICT: approved | approved_minor | changes_requested | critical_block
|
|
99
101
|
REVIEWER_MODEL: <your exact model ID>
|
|
100
102
|
EXECUTOR_MODEL: <from Executor Report>
|
|
103
|
+
PANEL_MEMBER: <panel tier | solo | ->
|
|
101
104
|
VERIFICATION_RERUN: PASS | FAIL
|
|
102
105
|
FINDINGS:
|
|
103
106
|
critical: <file:line — what fails> | none
|
|
@@ -116,7 +119,29 @@ BLOCKING: <one line per critical/important finding, or "none">
|
|
|
116
119
|
```
|
|
117
120
|
Do not paste the diff, the findings prose, or verification logs back into the orchestrator.
|
|
118
121
|
|
|
119
|
-
### 2e —
|
|
122
|
+
### 2e — Aggregate the panel + lead bucketing
|
|
123
|
+
|
|
124
|
+
Once every panel member for a task has appended its `## Reviewer Verdict`, run:
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
node .claude/ukit/index/review-panel-aggregate.mjs docs/AI_HANDOFF/tasks/TASK-xxx.md
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
The script emits an Agreement Map (finding → set of panel members reporting it) and a
|
|
131
|
+
consensus verdict (`approved` when ≥2 members approve and no critical;
|
|
132
|
+
`changes_requested` when ≥1 critical or ≥2 request changes; else `approved_minor`).
|
|
133
|
+
`consensus≥2 identical findings = high signal` — a finding reported by two or more
|
|
134
|
+
members is high-signal. The consensus verdict, not any single member's vote, drives
|
|
135
|
+
`NEXT_STATUS_FOR_INDEX`.
|
|
136
|
+
|
|
137
|
+
Then the **lead member** (first entry in `modelRoles['review-panel']` = the judgment
|
|
138
|
+
tier) applies the lead-judgment buckets to every finding — **Act on** / **Consider** /
|
|
139
|
+
**Noted** / **Dismissed** — and appends a `## Panel Consensus` block to the task file
|
|
140
|
+
carrying the Agreement Map and the bucketed findings. `Act on` findings feed Step 3's
|
|
141
|
+
auto-fix loop; `Consider`/`Noted`/`Dismissed` are advisory. A high-signal finding must
|
|
142
|
+
not be bucketed below `Consider` without a stated reason.
|
|
143
|
+
|
|
144
|
+
### 2f — Orchestrator updates INDEX.md
|
|
120
145
|
|
|
121
146
|
Set `status = NEXT_STATUS_FOR_INDEX`. Orchestrator writes INDEX, not the reviewer.
|
|
122
147
|
|
|
@@ -144,7 +169,7 @@ For round in 1..2:
|
|
|
144
169
|
```bash
|
|
145
170
|
git add -A && git commit -m "handoff: fix round <N> — TASK-00x"
|
|
146
171
|
```
|
|
147
|
-
4. Re-review **only the tasks touched this round**, in parallel, per 2a–
|
|
172
|
+
4. Re-review **only the tasks touched this round**, in parallel, per 2a–2f — panel spawn, aggregation, and lead bucketing included. The reviewers must
|
|
148
173
|
still differ from the executor model and still re-runs verification itself.
|
|
149
174
|
5. Update `INDEX.md` and `RUN.md`. Exit when everything is approved.
|
|
150
175
|
|
|
@@ -190,5 +215,5 @@ Then, as the last line:
|
|
|
190
215
|
This is the only place this command may ask for a compaction. All state lives in git,
|
|
191
216
|
`INDEX.md` and `RUN.md`, so a compacted or fresh session resumes with nothing lost.
|
|
192
217
|
|
|
193
|
-
> Orchestrator (this session) handles Step 1, 2e, and 3 directly — those are not delegated.
|
|
194
|
-
> **For the human operator, on a tool with no agent support (Codex) — not an instruction to the model:** manually switch to the strong model, run 2a–2d per task (one session at a time).
|
|
218
|
+
> Orchestrator (this session) handles Step 1, 2e, 2f, and 3 directly — those are not delegated.
|
|
219
|
+
> **For the human operator, on a tool with no agent support (Codex) — not an instruction to the model:** manually switch to the strong model, run 2a–2d per task (one session at a time, panel members sequentially), then run `review-panel-aggregate.mjs` and apply the lead buckets yourself.
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// handoff-doc-validator.mjs — TASK-004 / Cycle C50 (shipped CLI twin)
|
|
3
|
+
//
|
|
4
|
+
// Self-contained twin of src/core/handoffDocValidator.js for user installs (no `src/`).
|
|
5
|
+
// The logic below is a literal port — keep the two files behavior-identical when editing.
|
|
6
|
+
//
|
|
7
|
+
// REAL GATE — unlike task-budget-validator.mjs (advisory, always exit 0), this
|
|
8
|
+
// script exits 1 on any violation. Do not "fix" the exit code.
|
|
9
|
+
//
|
|
10
|
+
// Usage:
|
|
11
|
+
// node .claude/ukit/index/handoff-doc-validator.mjs <file.md> [file.md...]
|
|
12
|
+
//
|
|
13
|
+
// Kind detection per file: basename PLAN.md → 'plan'; TASK-*.md → 'task';
|
|
14
|
+
// otherwise content sniffing (`## §1` → plan, `## Goal` → task, else
|
|
15
|
+
// executor-report). Task files carrying `## Executor Report` are also checked
|
|
16
|
+
// against the report header contract.
|
|
17
|
+
//
|
|
18
|
+
// Output:
|
|
19
|
+
// `PASS <file>` when clean; one `FAIL <file>: <rule>` line per violation.
|
|
20
|
+
// Exit 0 when every file passes, 1 otherwise, 2 on usage/read errors.
|
|
21
|
+
|
|
22
|
+
import fs from 'node:fs';
|
|
23
|
+
import path from 'node:path';
|
|
24
|
+
import process from 'node:process';
|
|
25
|
+
import { pathToFileURL } from 'node:url';
|
|
26
|
+
|
|
27
|
+
const PLAN_SECTIONS = ['§1', '§2', '§3', '§4', '§5', '§6'];
|
|
28
|
+
|
|
29
|
+
const TASK_SECTIONS = [
|
|
30
|
+
'Goal',
|
|
31
|
+
'Target Files',
|
|
32
|
+
'Test Cases',
|
|
33
|
+
'Test Files',
|
|
34
|
+
'Verification Commands',
|
|
35
|
+
'Acceptance Criteria',
|
|
36
|
+
'Dependencies',
|
|
37
|
+
'Interfaces',
|
|
38
|
+
];
|
|
39
|
+
|
|
40
|
+
const REPORT_FIELDS = ['EXECUTOR_TOOL', 'EXECUTOR_MODEL', 'EXECUTOR_SUBAGENT', 'RED_OUTPUT'];
|
|
41
|
+
|
|
42
|
+
// Section helpers. Any line starting with `## ` opens a section; the body runs
|
|
43
|
+
// until the next `## ` line or EOF. Returns [{ name, body, pos }] in file order.
|
|
44
|
+
function splitSections(markdown) {
|
|
45
|
+
const sections = [];
|
|
46
|
+
const lines = String(markdown).split('\n');
|
|
47
|
+
let current = null;
|
|
48
|
+
lines.forEach((line, idx) => {
|
|
49
|
+
const m = line.match(/^##\s+(.+?)\s*$/);
|
|
50
|
+
if (m) {
|
|
51
|
+
current = { name: m[1], body: '', pos: idx };
|
|
52
|
+
sections.push(current);
|
|
53
|
+
} else if (current) {
|
|
54
|
+
current.body += line + '\n';
|
|
55
|
+
}
|
|
56
|
+
});
|
|
57
|
+
return sections;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// Positions of `## §<n>` headings, keyed by the literal token ('§1'..'§6').
|
|
61
|
+
// A heading matches when its name starts with the token (`## §1 Intent`).
|
|
62
|
+
function planSectionPositions(sections) {
|
|
63
|
+
const positions = new Map();
|
|
64
|
+
for (const s of sections) {
|
|
65
|
+
for (const token of PLAN_SECTIONS) {
|
|
66
|
+
if (!positions.has(token) && s.name.startsWith(token)) positions.set(token, s.pos);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
return positions;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function hasSection(sections, name) {
|
|
73
|
+
return sections.some((s) => s.name === name || s.name.startsWith(`${name} `) || s.name.startsWith(`${name}(`));
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
function getSection(sections, name) {
|
|
77
|
+
return sections.find((s) => s.name === name || s.name.startsWith(`${name} `) || s.name.startsWith(`${name}(`));
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function validatePlan(markdown) {
|
|
81
|
+
const failures = [];
|
|
82
|
+
const sections = splitSections(markdown);
|
|
83
|
+
const positions = planSectionPositions(sections);
|
|
84
|
+
|
|
85
|
+
for (const token of PLAN_SECTIONS) {
|
|
86
|
+
if (!positions.has(token)) failures.push(`missing required section heading "## ${token}"`);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// Order check over the sections that exist: each must appear after the
|
|
90
|
+
// highest position seen so far.
|
|
91
|
+
let maxPos = -1;
|
|
92
|
+
for (const token of PLAN_SECTIONS) {
|
|
93
|
+
const pos = positions.get(token);
|
|
94
|
+
if (pos === undefined) continue;
|
|
95
|
+
if (pos < maxPos) {
|
|
96
|
+
failures.push(`section order violation: "## ${token}" appears after a later section`);
|
|
97
|
+
} else {
|
|
98
|
+
maxPos = pos;
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
const report = getSection(sections, 'Planner Report');
|
|
103
|
+
if (!report) {
|
|
104
|
+
failures.push('missing required section "## Planner Report"');
|
|
105
|
+
} else if (!/^PLANNER_MODEL\s*:/m.test(report.body)) {
|
|
106
|
+
failures.push('missing required field "PLANNER_MODEL:" inside "## Planner Report"');
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
return failures;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function validateExecutorReportFields(markdown, prefix) {
|
|
113
|
+
const failures = [];
|
|
114
|
+
for (const field of REPORT_FIELDS) {
|
|
115
|
+
if (!new RegExp(`^${field}\\s*:`, 'm').test(markdown)) {
|
|
116
|
+
failures.push(`${prefix}missing required field "${field}:"`);
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
return failures;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
function validateTask(markdown) {
|
|
123
|
+
const failures = [];
|
|
124
|
+
const sections = splitSections(markdown);
|
|
125
|
+
|
|
126
|
+
for (const name of TASK_SECTIONS) {
|
|
127
|
+
if (!hasSection(sections, name)) failures.push(`missing required section "## ${name}"`);
|
|
128
|
+
}
|
|
129
|
+
if (!/^\s*-?\s*Status\s*:/m.test(markdown)) {
|
|
130
|
+
failures.push('missing required field "Status:"');
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
// A task file that already carries an Executor Report must satisfy the
|
|
134
|
+
// report header contract too (FR-007).
|
|
135
|
+
const report = getSection(sections, 'Executor Report');
|
|
136
|
+
if (report) {
|
|
137
|
+
failures.push(...validateExecutorReportFields(report.body, 'executor-report: '));
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
return failures;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
export function validateHandoffDoc(markdown, kind) {
|
|
144
|
+
const text = String(markdown ?? '');
|
|
145
|
+
let failures;
|
|
146
|
+
if (kind === 'plan') failures = validatePlan(text);
|
|
147
|
+
else if (kind === 'task') failures = validateTask(text);
|
|
148
|
+
else if (kind === 'executor-report') failures = validateExecutorReportFields(text, '');
|
|
149
|
+
else failures = [`unknown kind "${kind}" (expected plan|task|executor-report)`];
|
|
150
|
+
return { ok: failures.length === 0, failures };
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// CLI-only helpers below (not part of the canonical module).
|
|
154
|
+
|
|
155
|
+
function detectKind(filePath, markdown) {
|
|
156
|
+
const base = path.basename(filePath);
|
|
157
|
+
if (/^PLAN\.md$/i.test(base)) return 'plan';
|
|
158
|
+
if (/^TASK-.*\.md$/i.test(base)) return 'task';
|
|
159
|
+
if (/^##\s+§1/m.test(markdown)) return 'plan';
|
|
160
|
+
if (/^##\s+Goal\b/m.test(markdown)) return 'task';
|
|
161
|
+
return 'executor-report';
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
function main() {
|
|
165
|
+
const files = process.argv.slice(2);
|
|
166
|
+
if (files.length === 0) {
|
|
167
|
+
console.error('usage: handoff-doc-validator.mjs <file.md> [file.md...]');
|
|
168
|
+
process.exit(2);
|
|
169
|
+
}
|
|
170
|
+
let failed = false;
|
|
171
|
+
for (const file of files) {
|
|
172
|
+
let markdown;
|
|
173
|
+
try {
|
|
174
|
+
markdown = fs.readFileSync(file, 'utf8');
|
|
175
|
+
} catch (err) {
|
|
176
|
+
console.log(`FAIL ${file}: unreadable (${err.message})`);
|
|
177
|
+
failed = true;
|
|
178
|
+
continue;
|
|
179
|
+
}
|
|
180
|
+
const { ok, failures } = validateHandoffDoc(markdown, detectKind(file, markdown));
|
|
181
|
+
if (ok) {
|
|
182
|
+
console.log(`PASS ${file}`);
|
|
183
|
+
} else {
|
|
184
|
+
failed = true;
|
|
185
|
+
for (const f of failures) console.log(`FAIL ${file}: ${f}`);
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
process.exit(failed ? 1 : 0);
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
if (process.argv[1] && pathToFileURL(process.argv[1]).href === import.meta.url) {
|
|
192
|
+
main();
|
|
193
|
+
}
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// review-panel-aggregate.mjs — TASK-004 / Cycle C50 (shipped CLI twin)
|
|
3
|
+
//
|
|
4
|
+
// Self-contained twin of src/core/reviewPanelAggregate.js for user installs (no `src/`).
|
|
5
|
+
// The logic below is a literal port — keep the two files behavior-identical when editing.
|
|
6
|
+
//
|
|
7
|
+
// ADVISORY — always exits 0. Aggregation output feeds the lead reviewer's
|
|
8
|
+
// AGREEMENT_MAP field and the orchestrator's INDEX status update; it must never
|
|
9
|
+
// break a pipeline.
|
|
10
|
+
//
|
|
11
|
+
// Usage:
|
|
12
|
+
// node .claude/ukit/index/review-panel-aggregate.mjs <TASK-xxx.md> [TASK-xxx.md...]
|
|
13
|
+
//
|
|
14
|
+
// Reads every `## Reviewer Verdict` block from the given task files, then prints:
|
|
15
|
+
// AGREEMENT_MAP: finding → member, member (one line per finding)
|
|
16
|
+
// HIGH_SIGNAL: finding → member, member (≥2 members or any critical)
|
|
17
|
+
// UNPARSED: <file>#verdict[<i>] (blocks with no VERDICT line)
|
|
18
|
+
// VERDICT: approved | approved_minor | changes_requested
|
|
19
|
+
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import process from 'node:process';
|
|
22
|
+
import { pathToFileURL } from 'node:url';
|
|
23
|
+
|
|
24
|
+
const SEVERITIES = ['critical', 'important', 'minor'];
|
|
25
|
+
|
|
26
|
+
// Parse one `## Reviewer Verdict` block (format: code-reviewer.md §Output).
|
|
27
|
+
// Returns null when no recognizable VERDICT line exists — the block is then
|
|
28
|
+
// counted as absent and named in `unparsed`.
|
|
29
|
+
export function parseVerdictBlock(text) {
|
|
30
|
+
const src = String(text ?? '');
|
|
31
|
+
const verdictMatch = src.match(/^VERDICT\s*:\s*(APPROVED-WITH-MINOR|APPROVED|CHANGES-REQUESTED|CRITICAL)\s*$/im);
|
|
32
|
+
if (!verdictMatch) return null;
|
|
33
|
+
|
|
34
|
+
const memberMatch = src.match(/^PANEL_MEMBER\s*:\s*(\S+)\s*$/im);
|
|
35
|
+
const member = memberMatch && memberMatch[1] !== '-' ? memberMatch[1] : null;
|
|
36
|
+
|
|
37
|
+
const findings = { critical: [], important: [], minor: [] };
|
|
38
|
+
const findingsMatch = src.match(/^FINDINGS:\s*$/im);
|
|
39
|
+
if (findingsMatch) {
|
|
40
|
+
const rest = src.slice(findingsMatch.index + findingsMatch[0].length);
|
|
41
|
+
let severity = null;
|
|
42
|
+
for (const line of rest.split('\n')) {
|
|
43
|
+
if (/^\S/.test(line)) break; // next top-level field ends FINDINGS
|
|
44
|
+
const sev = line.match(/^\s{2}(critical|important|minor)\s*:/);
|
|
45
|
+
if (sev) {
|
|
46
|
+
severity = sev[1];
|
|
47
|
+
continue;
|
|
48
|
+
}
|
|
49
|
+
const item = line.match(/^\s{4}-\s+(.+?)\s*$/);
|
|
50
|
+
if (item && severity) {
|
|
51
|
+
const text2 = item[1];
|
|
52
|
+
if (text2 !== 'none' && text2 !== '-') findings[severity].push(text2);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
return { member, verdict: verdictMatch[1].toUpperCase(), findings };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// Normalize a finding bullet for identity comparison: two members report the
|
|
61
|
+
// "identical finding" when the normalized text matches.
|
|
62
|
+
function normalizeFinding(text) {
|
|
63
|
+
return String(text).replace(/\s+/g, ' ').trim().toLowerCase();
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export function aggregatePanelVerdicts(verdicts) {
|
|
67
|
+
const list = Array.isArray(verdicts) ? verdicts : [];
|
|
68
|
+
const parsed = [];
|
|
69
|
+
const unparsed = [];
|
|
70
|
+
|
|
71
|
+
list.forEach((block, i) => {
|
|
72
|
+
const p = parseVerdictBlock(block);
|
|
73
|
+
if (p) parsed.push({ ...p, label: p.member || `verdict[${i}]` });
|
|
74
|
+
else unparsed.push(`verdict[${i}]`);
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
// Agreement Map: finding → set of panel members reporting it.
|
|
78
|
+
const map = new Map(); // normalized → { finding, members:Set, critical:boolean }
|
|
79
|
+
for (const p of parsed) {
|
|
80
|
+
for (const sev of SEVERITIES) {
|
|
81
|
+
for (const f of p.findings[sev]) {
|
|
82
|
+
const key = normalizeFinding(f);
|
|
83
|
+
let entry = map.get(key);
|
|
84
|
+
if (!entry) {
|
|
85
|
+
entry = { finding: f, members: new Set(), critical: sev === 'critical' };
|
|
86
|
+
map.set(key, entry);
|
|
87
|
+
}
|
|
88
|
+
entry.members.add(p.label);
|
|
89
|
+
if (sev === 'critical') entry.critical = true;
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
const agreementMap = [...map.values()].map((e) => ({
|
|
95
|
+
finding: e.finding,
|
|
96
|
+
members: [...e.members],
|
|
97
|
+
}));
|
|
98
|
+
const highSignal = [...map.values()]
|
|
99
|
+
.filter((e) => e.members.size >= 2 || e.critical)
|
|
100
|
+
.map((e) => ({ finding: e.finding, members: [...e.members] }));
|
|
101
|
+
|
|
102
|
+
const approvals = parsed.filter((p) => p.verdict === 'APPROVED').length;
|
|
103
|
+
const changeRequests = parsed.filter((p) => p.verdict === 'CHANGES-REQUESTED').length;
|
|
104
|
+
const hasCritical =
|
|
105
|
+
parsed.some((p) => p.verdict === 'CRITICAL') || parsed.some((p) => p.findings.critical.length > 0);
|
|
106
|
+
|
|
107
|
+
let verdict;
|
|
108
|
+
if (hasCritical || changeRequests >= 2) verdict = 'changes_requested';
|
|
109
|
+
else if (approvals >= 2) verdict = 'approved';
|
|
110
|
+
else verdict = 'approved_minor';
|
|
111
|
+
|
|
112
|
+
return { agreementMap, verdict, highSignal, unparsed };
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// CLI-only helpers below (not part of the canonical module).
|
|
116
|
+
|
|
117
|
+
// Extract every `## Reviewer Verdict` block from a task file. A block runs from
|
|
118
|
+
// its heading to the next `## ` heading or EOF.
|
|
119
|
+
function extractVerdictBlocks(markdown) {
|
|
120
|
+
const blocks = [];
|
|
121
|
+
const re = /^##\s+Reviewer Verdict\s*$/gm;
|
|
122
|
+
let m;
|
|
123
|
+
while ((m = re.exec(markdown)) !== null) {
|
|
124
|
+
const start = m.index;
|
|
125
|
+
const rest = markdown.slice(start + m[0].length);
|
|
126
|
+
const next = rest.search(/^##\s+/m);
|
|
127
|
+
blocks.push(next === -1 ? rest : rest.slice(0, next));
|
|
128
|
+
}
|
|
129
|
+
return blocks;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
function main() {
|
|
133
|
+
const files = process.argv.slice(2);
|
|
134
|
+
if (files.length === 0) {
|
|
135
|
+
console.error('usage: review-panel-aggregate.mjs <TASK-xxx.md> [TASK-xxx.md...]');
|
|
136
|
+
process.exit(0); // advisory — never break a pipeline
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
const blocks = [];
|
|
140
|
+
const labels = [];
|
|
141
|
+
for (const file of files) {
|
|
142
|
+
let markdown;
|
|
143
|
+
try {
|
|
144
|
+
markdown = fs.readFileSync(file, 'utf8');
|
|
145
|
+
} catch (err) {
|
|
146
|
+
console.log(`UNPARSED: ${file} (unreadable: ${err.message})`);
|
|
147
|
+
continue;
|
|
148
|
+
}
|
|
149
|
+
for (const block of extractVerdictBlocks(markdown)) {
|
|
150
|
+
labels.push(file);
|
|
151
|
+
blocks.push(block);
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const { agreementMap, verdict, highSignal, unparsed } = aggregatePanelVerdicts(blocks);
|
|
156
|
+
|
|
157
|
+
console.log('AGREEMENT_MAP:');
|
|
158
|
+
for (const e of agreementMap) console.log(` ${e.finding} → ${e.members.join(', ')}`);
|
|
159
|
+
if (agreementMap.length === 0) console.log(' (none)');
|
|
160
|
+
console.log('HIGH_SIGNAL:');
|
|
161
|
+
for (const e of highSignal) console.log(` ${e.finding} → ${e.members.join(', ')}`);
|
|
162
|
+
if (highSignal.length === 0) console.log(' (none)');
|
|
163
|
+
if (unparsed.length === 0) console.log('UNPARSED: (none)');
|
|
164
|
+
for (const u of unparsed) {
|
|
165
|
+
const idx = Number(u.slice('verdict['.length, -1));
|
|
166
|
+
console.log(`UNPARSED: ${labels[idx] ?? '?'}#${u}`);
|
|
167
|
+
}
|
|
168
|
+
console.log(`VERDICT: ${verdict}`);
|
|
169
|
+
process.exit(0);
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
if (process.argv[1] && pathToFileURL(process.argv[1]).href === import.meta.url) {
|
|
173
|
+
main();
|
|
174
|
+
}
|
|
@@ -70,6 +70,7 @@ export const ROUTE_MUTABILITIES = new Set(['read-only', 'mutating', 'mixed']);
|
|
|
70
70
|
export const ROUTE_RIGOR_LEVELS = new Set(['R0', 'R1', 'R2', 'R3', 'R4']);
|
|
71
71
|
export const ROUTE_MODEL_TIERS = new Set(['lite', 'code', 'smart']);
|
|
72
72
|
export const ROUTE_RISK_FLOORS = new Set(['none', 'high-risk']);
|
|
73
|
+
export const ROUTE_EFFORTS = new Set(['low', 'medium', 'high']);
|
|
73
74
|
// SPEC §5 FR-003: fixed table order — codes are emitted and printed in this order.
|
|
74
75
|
// The first six raise the floor to 'high-risk'; the last two are informational only.
|
|
75
76
|
export const ROUTE_RISK_REASON_CODES = Object.freeze([
|
|
@@ -354,6 +355,37 @@ function formatRiskFloorSegment(riskFloor = null) {
|
|
|
354
355
|
return printed.length > 0 ? `risk=${riskFloor.floor}(${printed.join(',')})` : null;
|
|
355
356
|
}
|
|
356
357
|
|
|
358
|
+
// FR-001 (M01.2' limits fragment): compile the contract's numeric budget keys
|
|
359
|
+
// into an advisory map. Only finite numbers survive — a non-numeric or missing
|
|
360
|
+
// key is simply absent. Zero is a real budget ("no read passes"), never
|
|
361
|
+
// filtered. Same key set as the ceremonyBudget.limits builder below.
|
|
362
|
+
export function deriveCeremonyLimits(executionContract = null) {
|
|
363
|
+
if (executionContract === null || typeof executionContract !== 'object') {
|
|
364
|
+
return {};
|
|
365
|
+
}
|
|
366
|
+
return Object.fromEntries(
|
|
367
|
+
['maxReadPasses', 'maxContextPulls', 'maxReadPassesBeforeReassess']
|
|
368
|
+
.filter((key) => Number.isFinite(executionContract[key]))
|
|
369
|
+
.map((key) => [key, executionContract[key]]),
|
|
370
|
+
);
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
// Route-line segment (FR-002): fixed reads,ctx,reassess order; only present
|
|
374
|
+
// keys print. Empty/absent map → null so the segment never appears.
|
|
375
|
+
export function formatLimitsSegment(limits = null) {
|
|
376
|
+
if (limits === null || typeof limits !== 'object') {
|
|
377
|
+
return null;
|
|
378
|
+
}
|
|
379
|
+
const parts = [
|
|
380
|
+
['reads', 'maxReadPasses'],
|
|
381
|
+
['ctx', 'maxContextPulls'],
|
|
382
|
+
['reassess', 'maxReadPassesBeforeReassess'],
|
|
383
|
+
]
|
|
384
|
+
.filter(([, key]) => Number.isFinite(limits[key]))
|
|
385
|
+
.map(([label, key]) => `${label}:${limits[key]}`);
|
|
386
|
+
return parts.length > 0 ? `limits=${parts.join(',')}` : null;
|
|
387
|
+
}
|
|
388
|
+
|
|
357
389
|
// FR-004 (M01.3'): Fast Path eligibility predicate. Returns null when no riskFloor
|
|
358
390
|
// was supplied — eligibility must never be derived without the floor check, so a
|
|
359
391
|
// missing floor means "not computed", not "none". Eligible iff the lane is
|
|
@@ -460,6 +492,10 @@ function buildResolvedRouteFields({
|
|
|
460
492
|
completionState = null,
|
|
461
493
|
riskFloor = null,
|
|
462
494
|
} = {}) {
|
|
495
|
+
// Decision table v2 (FR-001/FR-002): tier + effort resolve together from the
|
|
496
|
+
// contract lane and the additive riskFloor. The router cannot observe host
|
|
497
|
+
// binding capabilities, so the emitted pair is advisory text by definition.
|
|
498
|
+
const tierDecision = resolveModelTier({ executionMode, riskFloor });
|
|
463
499
|
const signalText = buildNormalizedRouteSignalText(routingContext.promptText, routingContext.commandText);
|
|
464
500
|
const deliveryOnly = isDeliveryOnlyRequest({
|
|
465
501
|
signalText,
|
|
@@ -484,7 +520,8 @@ function buildResolvedRouteFields({
|
|
|
484
520
|
riskFloor: riskFloor?.floor ?? null,
|
|
485
521
|
phase: null,
|
|
486
522
|
contractVersion: ROUTE_CONTRACT_VERSION,
|
|
487
|
-
modelTier:
|
|
523
|
+
modelTier: tierDecision.tier,
|
|
524
|
+
effort: tierDecision.effort,
|
|
488
525
|
},
|
|
489
526
|
evidence: {
|
|
490
527
|
observations: [],
|
|
@@ -567,6 +604,10 @@ export function validateResolvedRoute(route = null) {
|
|
|
567
604
|
if (route.execution.modelTier !== null && !ROUTE_MODEL_TIERS.has(route.execution.modelTier)) {
|
|
568
605
|
errors.push(`execution.modelTier must be null or one of: ${[...ROUTE_MODEL_TIERS].join(', ')}.`);
|
|
569
606
|
}
|
|
607
|
+
if (route.execution.effort !== null && route.execution.effort !== undefined
|
|
608
|
+
&& !ROUTE_EFFORTS.has(route.execution.effort)) {
|
|
609
|
+
errors.push(`execution.effort must be null or one of: ${[...ROUTE_EFFORTS].join(', ')}.`);
|
|
610
|
+
}
|
|
570
611
|
}
|
|
571
612
|
if (!isObject(route.evidence)) {
|
|
572
613
|
errors.push('evidence must be an object.');
|
|
@@ -2879,6 +2920,12 @@ function buildRouteSummary({
|
|
|
2879
2920
|
})
|
|
2880
2921
|
: null;
|
|
2881
2922
|
const riskSegment = formatRiskFloorSegment(riskFloor);
|
|
2923
|
+
// FR-003 (M01.2' limits fragment): the rigor advisory — emitted whenever
|
|
2924
|
+
// rigor.stage is on, independent of fastPath/escalation. Stage off → no
|
|
2925
|
+
// segment, byte-identical route.
|
|
2926
|
+
const limitsSegment = resolveRouteStage(runtimeConfig, 'rigor') !== 'off'
|
|
2927
|
+
? formatLimitsSegment(deriveCeremonyLimits(executionContract))
|
|
2928
|
+
: null;
|
|
2882
2929
|
// FR-004 (M01.3'): fastPath is emitted whenever its own stage is on — the field
|
|
2883
2930
|
// is always set then (eligible or not) so telemetry/harness can read it; the
|
|
2884
2931
|
// route-line segment prints only for eligible routes.
|
|
@@ -2951,6 +2998,7 @@ function buildRouteSummary({
|
|
|
2951
2998
|
formatCompactSegment('styles', styleFiles),
|
|
2952
2999
|
editGuardHint ? `editGuard=${editGuardHint}` : null,
|
|
2953
3000
|
riskSegment,
|
|
3001
|
+
limitsSegment,
|
|
2954
3002
|
fastPathSegment,
|
|
2955
3003
|
delegationRecommendation?.hint ? `delegate=${delegationRecommendation.hint}` : null,
|
|
2956
3004
|
policyMode ? `policy=${policyMode}` : null,
|
|
@@ -3442,6 +3490,45 @@ function buildExecutionContract(executionMode = null) {
|
|
|
3442
3490
|
return contracts[executionMode] ? { ...contracts[executionMode] } : null;
|
|
3443
3491
|
}
|
|
3444
3492
|
|
|
3493
|
+
// Decision table v2 (V3_RESHAPE §5): the contracts table's modelTier column is the
|
|
3494
|
+
// base row; resolveModelTier adds the (riskFloor, hostCapabilities) columns.
|
|
3495
|
+
// Deterministic — no registry, no leases, no outbound capability calls.
|
|
3496
|
+
// Literal mirror of src/core/executionContracts.js resolveModelTier; parity is
|
|
3497
|
+
// locked by tests/consistency/executionContractSync.test.js.
|
|
3498
|
+
const MODEL_TIER_ORDER = ['lite', 'code', 'smart'];
|
|
3499
|
+
const MODEL_TIER_EFFORT = {
|
|
3500
|
+
lite: 'low',
|
|
3501
|
+
code: 'medium',
|
|
3502
|
+
smart: 'high',
|
|
3503
|
+
};
|
|
3504
|
+
|
|
3505
|
+
/**
|
|
3506
|
+
* Resolves {tier, effort, advisoryOnly} for a route. A 'high-risk' floor escalates
|
|
3507
|
+
* the tier one band (lite→code→smart, capped at smart) and forces effort 'high'.
|
|
3508
|
+
* Unknown mode → {tier:null, effort:null, advisoryOnly:true}. Missing/unknown
|
|
3509
|
+
* riskFloor → 'none'. Missing hostCapabilities → advisoryOnly:true (the route text
|
|
3510
|
+
* is all the host gets when it cannot bind model and effort).
|
|
3511
|
+
*/
|
|
3512
|
+
export function resolveModelTier({
|
|
3513
|
+
executionMode = null,
|
|
3514
|
+
riskFloor = null,
|
|
3515
|
+
hostCapabilities = null,
|
|
3516
|
+
} = {}) {
|
|
3517
|
+
const baseTier = buildExecutionContract(executionMode)?.modelTier ?? null;
|
|
3518
|
+
if (!baseTier) {
|
|
3519
|
+
return { tier: null, effort: null, advisoryOnly: true };
|
|
3520
|
+
}
|
|
3521
|
+
const highRisk = riskFloor?.floor === 'high-risk';
|
|
3522
|
+
const tier = highRisk
|
|
3523
|
+
? MODEL_TIER_ORDER[Math.min(MODEL_TIER_ORDER.indexOf(baseTier) + 1, MODEL_TIER_ORDER.length - 1)]
|
|
3524
|
+
: baseTier;
|
|
3525
|
+
return {
|
|
3526
|
+
tier,
|
|
3527
|
+
effort: highRisk ? 'high' : MODEL_TIER_EFFORT[tier],
|
|
3528
|
+
advisoryOnly: !(hostCapabilities?.canBindModel === true && hostCapabilities?.canBindEffort === true),
|
|
3529
|
+
};
|
|
3530
|
+
}
|
|
3531
|
+
|
|
3445
3532
|
function buildApproachSelectorResult({
|
|
3446
3533
|
executionMode = null,
|
|
3447
3534
|
executionScores = null,
|
|
@@ -3726,7 +3813,8 @@ export function computeRiskEscalation({
|
|
|
3726
3813
|
previousRouteSummary = null,
|
|
3727
3814
|
config = null,
|
|
3728
3815
|
} = {}) {
|
|
3729
|
-
|
|
3816
|
+
const stage = resolveRouteStage(config, 'escalation');
|
|
3817
|
+
if (stage === 'off') {
|
|
3730
3818
|
return null;
|
|
3731
3819
|
}
|
|
3732
3820
|
const sourceFlags = config?.routing?.escalation?.sources ?? {};
|
|
@@ -3747,7 +3835,7 @@ export function computeRiskEscalation({
|
|
|
3747
3835
|
const level = RISK_ESCALATION_LEVELS[derivedLevel] >= previousLevel
|
|
3748
3836
|
? derivedLevel
|
|
3749
3837
|
: previous.level;
|
|
3750
|
-
return { level, sources };
|
|
3838
|
+
return { level, sources, stage };
|
|
3751
3839
|
}
|
|
3752
3840
|
|
|
3753
3841
|
// FR-005: buildRouteSummary runs before riskEscalation exists, so the
|