@ngockhoale/ukit 3.1.0 → 3.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,309 @@
1
+ /**
2
+ * reviewVerdict.js (S2 / UNIC_DECISION_MIGRATION)
3
+ *
4
+ * Review-verdict producer for the `review` decision family. Reviewers (the
5
+ * general-LLM code-reviewer agents) still generate findings; when
6
+ * `decisionPlane.families.review.stage` is promoted past 'off', the FINAL
7
+ * verdict and per-finding bucket pass runs through unic-decision instead.
8
+ * Deterministic severity aggregation stays authoritative on 'off' and on
9
+ * every failure path — never another LLM for the verdict.
10
+ *
11
+ * Registered keys (src/decision/registry.js):
12
+ * review.verdict.v1 — solo reviewer verdict (choice)
13
+ * review.finding-bucket.v1 — per-finding triage bucket (choice; asked once
14
+ * per finding, wire name carries `#<id>`)
15
+ * review.panel-verdict.v1 — aggregated panel verdict (choice)
16
+ *
17
+ * Transport discipline (C13/statePacket conventions): only whitelisted,
18
+ * truncated finding fields cross the wire — {id, severity, location, claim}.
19
+ * No diff hunks, no prose, no secrets. Everything below is pure — no I/O —
20
+ * and never throws (client.js precedent).
21
+ */
22
+
23
+ export const VERDICT_CANDIDATES = Object.freeze([
24
+ 'APPROVED',
25
+ 'APPROVED-WITH-MINOR',
26
+ 'CHANGES-REQUESTED',
27
+ 'CRITICAL',
28
+ ]);
29
+
30
+ export const BUCKET_CANDIDATES = Object.freeze([
31
+ 'Act on',
32
+ 'Consider',
33
+ 'Noted',
34
+ 'Dismissed',
35
+ ]);
36
+
37
+ export const REVIEW_VERDICT_KEYS = Object.freeze({
38
+ solo: 'review.verdict.v1',
39
+ bucket: 'review.finding-bucket.v1',
40
+ panel: 'review.panel-verdict.v1',
41
+ });
42
+
43
+ // Bounds keep the batch inside the state budget and the tool list small.
44
+ // MAX_FINDINGS mirrors statePacket.js MAX_ARRAY_ITEMS (24); a question plus a
45
+ // verdict slot means at most MAX_FINDINGS bucket questions per batch.
46
+ export const MAX_FINDINGS = 24;
47
+ export const MAX_CLAIM_LENGTH = 160;
48
+ const MAX_LOCATION_LENGTH = 160;
49
+ const MAX_ID_LENGTH = 64;
50
+ const MAX_SEVERITY_LENGTH = 24;
51
+
52
+ const KNOWN_SEVERITIES = new Set(['critical', 'important', 'minor']);
53
+
54
+ const BUCKET_INSTRUCTION =
55
+ 'Triage this single review finding into exactly one bucket: '
56
+ + '"Act on" feeds the auto-fix loop; "Consider" is advisory but worth the '
57
+ + 'author\'s attention; "Noted" is informational only; "Dismissed" is '
58
+ + 'rejected as not actionable.';
59
+
60
+ function isPlainObject(value) {
61
+ return value !== null && typeof value === 'object' && !Array.isArray(value);
62
+ }
63
+
64
+ function collapseWhitespace(text) {
65
+ return String(text).replace(/\s+/g, ' ').trim();
66
+ }
67
+
68
+ function truncate(text, max) {
69
+ return text.length > max ? `${text.slice(0, max - 1)}…` : text;
70
+ }
71
+
72
+ function normalizeSeverity(raw) {
73
+ const severity = typeof raw === 'string'
74
+ ? raw.trim().toLowerCase().slice(0, MAX_SEVERITY_LENGTH)
75
+ : '';
76
+ return KNOWN_SEVERITIES.has(severity) ? severity : 'unknown';
77
+ }
78
+
79
+ // Whitelist one finding down to the exact fields the verdict pass is allowed
80
+ // to see. Location collapses file:line into a single label; every other
81
+ // field on the input object is dropped silently.
82
+ function sanitizeFinding(raw, index, usedIds) {
83
+ const source = isPlainObject(raw) ? raw : {};
84
+ let id = typeof source.id === 'string' && source.id.trim().length > 0
85
+ ? source.id.trim().replace(/[^A-Za-z0-9._-]/g, '_').slice(0, MAX_ID_LENGTH)
86
+ : `F${index + 1}`;
87
+ if (id.length === 0 || usedIds.has(id)) {
88
+ // Collision fallback must itself be unique — two findings both claiming
89
+ // 'F2' must not collapse into one wire name / one bucket row.
90
+ let suffix = '';
91
+ while (usedIds.has(`F${index + 1}${suffix}`)) suffix = `${suffix}x`;
92
+ id = `F${index + 1}${suffix}`;
93
+ }
94
+ usedIds.add(id);
95
+
96
+ let file = typeof source.file === 'string' ? collapseWhitespace(source.file) : '';
97
+ const line = Number.isInteger(source.line) && source.line > 0 ? source.line : null;
98
+ if (file.length > 0) file = truncate(file, MAX_LOCATION_LENGTH - 8);
99
+ const location = file.length > 0
100
+ ? (line !== null ? `${file}:${line}` : file)
101
+ : (line !== null ? `line ${line}` : '-');
102
+
103
+ const claimSource = [source.claim, source.title, source.summary, source.message]
104
+ .find((v) => typeof v === 'string' && v.trim().length > 0);
105
+
106
+ return {
107
+ id,
108
+ severity: normalizeSeverity(source.severity),
109
+ location,
110
+ claim: truncate(collapseWhitespace(claimSource ?? ''), MAX_CLAIM_LENGTH),
111
+ };
112
+ }
113
+
114
+ // Stable short fingerprint for the batch id — same findings → same batchId
115
+ // across runs, so receipts and logs reconcile.
116
+ function stableBatchId(findings) {
117
+ let h = 0x811c9dc5;
118
+ const text = JSON.stringify(findings.map((f) => [f.id, f.severity, f.location, f.claim]));
119
+ for (let i = 0; i < text.length; i += 1) {
120
+ h ^= text.charCodeAt(i);
121
+ h = Math.imul(h, 0x01000193);
122
+ }
123
+ return `rv-${(h >>> 0).toString(16).padStart(8, '0')}`;
124
+ }
125
+
126
+ /**
127
+ * Build the C13 question batch for the review verdict pass.
128
+ *
129
+ * @param {object} [input]
130
+ * @param {Array} [input.findings] extracted review findings (whitelisted
131
+ * fields only: id/severity/file/line/claim|title|summary|message).
132
+ * @param {string} [input.mode] 'solo' → review.verdict.v1 question;
133
+ * 'panel' → review.panel-verdict.v1 instead.
134
+ * @returns {{questions: object[], statePacket: object, batch: object,
135
+ * findings: object[], verdictDecisionKey: string}}
136
+ * Never throws — garbage input sanitizes to an empty finding list.
137
+ */
138
+ export function buildReviewVerdictBatch({ findings = [], mode = 'solo' } = {}) {
139
+ const usedIds = new Set();
140
+ const sanitized = (Array.isArray(findings) ? findings : [])
141
+ .slice(0, MAX_FINDINGS)
142
+ .map((raw, index) => sanitizeFinding(raw, index, usedIds));
143
+
144
+ const verdictDecisionKey = mode === 'panel'
145
+ ? REVIEW_VERDICT_KEYS.panel
146
+ : REVIEW_VERDICT_KEYS.solo;
147
+
148
+ const counts = { critical: 0, important: 0, minor: 0, unknown: 0 };
149
+ for (const f of sanitized) counts[f.severity] += 1;
150
+
151
+ const verdictQuestion = {
152
+ decisionKey: verdictDecisionKey,
153
+ schemaVersion: 1,
154
+ family: 'review',
155
+ owner: 'handoffReview',
156
+ kind: 'choice',
157
+ instruction:
158
+ 'Pick the bounded review verdict for this finding set. '
159
+ + 'APPROVED = clean; APPROVED-WITH-MINOR = ship with nits logged; '
160
+ + 'CHANGES-REQUESTED = must fix before merge; CRITICAL = unsafe to ship.',
161
+ candidates: [...VERDICT_CANDIDATES],
162
+ hardConstraintRefs: ['verdict-vocabulary'],
163
+ };
164
+
165
+ // One bucket question per finding. The shared registered key cannot repeat
166
+ // in one batch (name-based reconciliation), so each wire name carries
167
+ // `#<findingId>`; parseVerdictAnswer splits it back off.
168
+ const bucketQuestions = sanitized.map((f) => ({
169
+ decisionKey: `${REVIEW_VERDICT_KEYS.bucket}#${f.id}`,
170
+ schemaVersion: 1,
171
+ family: 'review',
172
+ owner: 'handoffReview',
173
+ kind: 'choice',
174
+ instruction: `${BUCKET_INSTRUCTION} Finding ${f.id} [${f.severity}] ${f.location}: ${f.claim || '(no claim text)'}`,
175
+ candidates: [...BUCKET_CANDIDATES],
176
+ hardConstraintRefs: ['bucket-vocabulary'],
177
+ }));
178
+
179
+ const questions = [verdictQuestion, ...bucketQuestions];
180
+
181
+ // Producer-owned packet — only sanitized whitelisted fields. `findingCount`
182
+ // is the only aggregate the model needs; severities ride the findings array.
183
+ const statePacket = {
184
+ kind: 'review-verdict',
185
+ mode: mode === 'panel' ? 'panel' : 'solo',
186
+ findingCount: sanitized.length,
187
+ severityCounts: counts,
188
+ findings: sanitized,
189
+ };
190
+
191
+ const batch = {
192
+ batchVersion: 1,
193
+ batchId: stableBatchId(sanitized),
194
+ boundary: 'review-verdict',
195
+ questions,
196
+ statePacket,
197
+ };
198
+
199
+ return { questions, statePacket, batch, findings: sanitized, verdictDecisionKey };
200
+ }
201
+
202
+ /**
203
+ * Deterministic verdict from finding severities — the authoritative path at
204
+ * stage 'off' and the fallback on every failure (unic-decision unavailable,
205
+ * malformed/invalid answers). Rule (documented in review-verdict.mjs):
206
+ * any critical → CHANGES-REQUESTED
207
+ * ≥1 important or ≥1 unclassified → APPROVED-WITH-MINOR
208
+ * else → APPROVED
209
+ */
210
+ export function deriveDeterministicVerdict(findings) {
211
+ const list = Array.isArray(findings) ? findings : [];
212
+ let importantish = false;
213
+ for (const raw of list) {
214
+ const severity = isPlainObject(raw) ? normalizeSeverity(raw.severity) : 'unknown';
215
+ if (severity === 'critical') return 'CHANGES-REQUESTED';
216
+ if (severity === 'important' || severity === 'unknown') importantish = true;
217
+ }
218
+ return importantish ? 'APPROVED-WITH-MINOR' : 'APPROVED';
219
+ }
220
+
221
+ /**
222
+ * Deterministic per-finding bucket — the fallback the lead reviewer's
223
+ * bucketing judgment replaces only when the family stage is enabled.
224
+ * critical | important | unclassified → 'Act on'
225
+ * minor → 'Noted'
226
+ * Conservative: an unknown severity feeds the auto-fix loop rather than
227
+ * being silently dismissed.
228
+ */
229
+ export function deriveDeterministicBucket(finding) {
230
+ const severity = normalizeSeverity(isPlainObject(finding) ? finding.severity : null);
231
+ return severity === 'minor' ? 'Noted' : 'Act on';
232
+ }
233
+
234
+ // Wire name → {decisionKey, findingId}: 'review.finding-bucket.v1#F3' splits
235
+ // into the registered key plus the finding the bucket answer belongs to.
236
+ function splitWireName(name) {
237
+ const text = String(name ?? '');
238
+ const hash = text.lastIndexOf('#');
239
+ if (hash > 0 && text.slice(0, hash) === REVIEW_VERDICT_KEYS.bucket) {
240
+ return { decisionKey: REVIEW_VERDICT_KEYS.bucket, findingId: text.slice(hash + 1) };
241
+ }
242
+ return { decisionKey: text, findingId: null };
243
+ }
244
+
245
+ /**
246
+ * Fold a DecisionBatchResult (from client.requestBatch or an offline
247
+ * parseBatchResponse) into the review-verdict shape. Never throws; every
248
+ * invalid/missing answer degrades to `null`/deterministic-bucket so callers
249
+ * can substitute deriveDeterministicVerdict as the authoritative fallback.
250
+ *
251
+ * @param {object} [result] DecisionBatchResult — {status, answers}.
252
+ * @param {object} [opts]
253
+ * @param {Array} [opts.findings] sanitized or raw findings — supplies the
254
+ * bucket fold order and per-finding fallback values.
255
+ * @returns {{status: string, verdict: string|null, verdictStatus: string,
256
+ * buckets: Array<{findingId, bucket, bucketStatus}>}}
257
+ */
258
+ export function parseVerdictAnswer(result, { findings = [] } = {}) {
259
+ const status = typeof result?.status === 'string' ? result.status : 'invalid';
260
+ const answers = Array.isArray(result?.answers) ? result.answers : [];
261
+ const list = Array.isArray(findings) ? findings : [];
262
+ let verdict = null;
263
+ let verdictStatus = 'missing';
264
+ // findingId → {bucket, bucketStatus}; 'default' marks the deterministic
265
+ // fallback standing in because the batch never produced an answer for it.
266
+ const bucketByFinding = new Map();
267
+ for (const raw of list) {
268
+ if (!isPlainObject(raw)) continue;
269
+ const id = typeof raw.id === 'string' && raw.id.length > 0 ? raw.id : null;
270
+ if (id === null) continue;
271
+ bucketByFinding.set(id, {
272
+ bucket: deriveDeterministicBucket(raw),
273
+ bucketStatus: 'default',
274
+ });
275
+ }
276
+
277
+ for (const answer of answers) {
278
+ if (!isPlainObject(answer)) continue;
279
+ const { decisionKey, findingId } = splitWireName(answer.decisionKey);
280
+ const valid = answer.validationStatus === 'valid'
281
+ && typeof answer.value === 'string';
282
+ if (decisionKey === REVIEW_VERDICT_KEYS.solo
283
+ || decisionKey === REVIEW_VERDICT_KEYS.panel) {
284
+ if (valid && VERDICT_CANDIDATES.includes(answer.value)) {
285
+ verdict = answer.value;
286
+ verdictStatus = 'valid';
287
+ } else {
288
+ verdictStatus = 'invalid';
289
+ }
290
+ continue;
291
+ }
292
+ if (decisionKey === REVIEW_VERDICT_KEYS.bucket && findingId !== null) {
293
+ bucketByFinding.set(
294
+ findingId,
295
+ valid && BUCKET_CANDIDATES.includes(answer.value)
296
+ ? { bucket: answer.value, bucketStatus: 'valid' }
297
+ : { bucket: deriveDeterministicBucket({ severity: 'unknown' }), bucketStatus: 'invalid' },
298
+ );
299
+ }
300
+ }
301
+
302
+ const buckets = [...bucketByFinding.entries()].map(([findingId, entry]) => ({
303
+ findingId,
304
+ bucket: entry.bucket,
305
+ bucketStatus: entry.bucketStatus,
306
+ }));
307
+
308
+ return { status, verdict, verdictStatus, buckets };
309
+ }
@@ -49,6 +49,26 @@ If any input is missing, return `CHANGES-REQUESTED` with reason "incomplete hand
49
49
  - **APPROVED-WITH-MINOR** — Minor naming / doc / style issues. Logged on task file but handoff allowed.
50
50
  - **APPROVED** — Clean.
51
51
 
52
+ ### Stage gate — verdict emission (unic-decision)
53
+
54
+ You generate the findings; the FINAL verdict/bucket classification is a bounded
55
+ decision owned by the local decision model when staged. Read
56
+ `decisionPlane.families.review.stage` from `.ukit/storage/config.json`:
57
+
58
+ - `off` (default) or any adapter failure (`outcomeClass` ≠ `accepted`/`partial`) →
59
+ you emit the verdict directly, exactly as today; the deterministic severity
60
+ aggregate (any critical → CHANGES-REQUESTED; ≥1 important or any
61
+ unclassified/unknown severity → APPROVED-WITH-MINOR; else APPROVED) stays
62
+ authoritative.
63
+ - not `off` → emit the verdict via `node .claude/ukit/index/review-verdict.mjs`
64
+ (registered key `review.verdict.v1`): pipe your extracted findings as JSON on
65
+ stdin — whitelisted fields only (`id`, `severity`, `file`, `line`, short
66
+ `claim`; never diff hunks or secrets). Use the adapter's `verdict`/`buckets`
67
+ fields in the `## Reviewer Verdict` block.
68
+
69
+ Model isolation is preserved: the verdict pass runs on the local unic-decision
70
+ model, so reviewer ≠ executor still stands.
71
+
52
72
  ### Output (append to task file as `## Reviewer Verdict`)
53
73
 
54
74
  ```
@@ -138,7 +158,11 @@ verdict block — with these additions:
138
158
  `node .claude/ukit/index/review-panel-aggregate.mjs <TASK-xxx.md...>` and hands you
139
159
  the output, the lead fills `AGREEMENT_MAP` with the emitted finding → members map
140
160
  and applies the lead-judgment buckets to every finding: **Act on** / **Consider** /
141
- **Noted** / **Dismissed**.
161
+ **Noted** / **Dismissed**. When `decisionPlane.families.review.stage` is not `off`,
162
+ the lead's bucketing and panel verdict go through
163
+ `node .claude/ukit/index/review-verdict.mjs --panel` (`review.finding-bucket.v1` /
164
+ `review.panel-verdict.v1`) per the stage gate above — the deterministic aggregate
165
+ stays authoritative at `off` or on adapter failure.
142
166
  - `consensus≥2 identical findings = high signal`: a finding reported by two or more
143
167
  panel members is high-signal and must not be bucketed below **Consider** without a
144
168
  stated reason.
@@ -27,6 +27,22 @@ You are UKit's internal small-task maintainer. You run as a sidecar/parallel/non
27
27
  - Keeping agent context compact without removing existing lanes: Claude PreCompact/reinject stays active and Codex Desktop soft handoffs use `compact.codexContext.compactTarget` (default 150 lines; preferred 120-150; hard max 170) while preserving critical state.
28
28
  - Small, reversible UKit runtime maintenance decisions.
29
29
 
30
+ ## Bounded Decisions (stage-gated unic-decision)
31
+
32
+ The six bounded decisions enumerated in `.codex/settings.json` `smallTaskModel.decisionPolicy.decisions` — `fast-vs-slow-lane`, `safe-vs-risky-lane`, `skill-routing-needed`, `step-budget-enough`, `compact-now-or-later`, `summarize-docs-or-keep-detail` — consult `unic-decision` first via the installed CLI when the decision-plane stage allows:
33
+
34
+ ```
35
+ node .claude/ukit/index/sidecar-decision.mjs --decision <name> [--root <dir>] # context JSON on stdin
36
+ ```
37
+
38
+ The CLI resolves `decisionPlane.families.workflow.stage` and owns the name → `workflow.*` decision-key mapping (`--list` prints it):
39
+
40
+ - `off` (default): the CLI's deterministic rules answer directly — zero transport, authoritative.
41
+ - `shadow`: run the batch for comparison only; the deterministic rule answer stays authoritative.
42
+ - `canary`/`default`: a valid unic-decision answer wins; an invalid or unavailable adapter falls back to the deterministic rule (`outcomeClass: 'unavailable'`).
43
+
44
+ `unic-decision` is the only model allowed in these decision steps — on any failure the deterministic rule is the fallback, never another LLM for the verdict. This lane's own model (`unic-lite`) keeps generative work only: summarization, doc maintenance, and cleanup — it never emits a verdict for these decisions.
45
+
30
46
  ## Never Use For
31
47
 
32
48
  - Security/auth/permission/secrets work.
@@ -576,6 +576,8 @@ git status # overview of modified/new/deleted files
576
576
 
577
577
  Review as one unified diff — correctness, regression risk, security, edge cases, maintainability. Cross-reference each task's intent in `docs/AI_HANDOFF/tasks/TASK-xxx.md`.
578
578
 
579
+ Stage gate (unic-decision): reviewers still generate findings; when `decisionPlane.families.review.stage` is not `off`, the emitted verdict goes through `node .claude/ukit/index/review-verdict.mjs` (`review.verdict.v1`; `--panel` for `review.panel-verdict.v1`/`review.finding-bucket.v1`) — the deterministic severity aggregate stays authoritative at `off` or on adapter failure.
580
+
579
581
  Append verdict to each task file:
580
582
  ```
581
583
  ## Reviewer Verdict
@@ -141,6 +141,18 @@ carrying the Agreement Map and the bucketed findings. `Act on` findings feed Ste
141
141
  auto-fix loop; `Consider`/`Noted`/`Dismissed` are advisory. A high-signal finding must
142
142
  not be bucketed below `Consider` without a stated reason.
143
143
 
144
+ Stage gate (unic-decision): findings stay generated by the general-LLM panel — only
145
+ the FINAL verdict/bucket pass consults the local decision model. When
146
+ `decisionPlane.families.review.stage` (`.ukit/storage/config.json`) is not `off`, the
147
+ lead member's finding bucketing and the panel verdict go through
148
+ `node .claude/ukit/index/review-verdict.mjs --panel` (registered keys
149
+ `review.panel-verdict.v1` + `review.finding-bucket.v1`) instead of free judgment;
150
+ pipe the extracted findings (id/severity/file:line/short claim — never diff hunks)
151
+ as JSON on stdin. At stage `off` or on any `outcomeClass` ≠ `accepted`/`partial`
152
+ (gateway unavailable, invalid answers) the deterministic severity aggregate above
153
+ stays authoritative. Model isolation is preserved: the verdict pass is the local
154
+ decision model, so reviewer ≠ executor still stands.
155
+
144
156
  ### 2f — Orchestrator updates INDEX.md
145
157
 
146
158
  Set `status = NEXT_STATUS_FOR_INDEX`. Orchestrator writes INDEX, not the reviewer.