@ngockhoale/ukit 3.0.12 → 3.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +1 -0
- package/manifests/documentation.yaml +12 -0
- package/manifests/platform.full.yaml +24 -0
- package/package.json +1 -1
- package/scripts/bench/data-foundation.mjs +368 -50
- package/src/cli/commands/doctor.js +232 -3
- package/src/cli/commands/feedback.js +64 -1
- package/src/cli/commands/install.js +18 -0
- package/src/cli/commands/memory.js +42 -37
- package/src/cli/commands/telemetry.js +460 -0
- package/src/cli/index.js +7 -0
- package/src/core/agentRuntime/adapters.js +83 -2
- package/src/core/agentRuntime/diagnostics.js +104 -0
- package/src/core/agentRuntime/supervisor.js +137 -0
- package/src/core/agentRuntime/telemetry.js +204 -0
- package/src/core/memory/memoryEmit.js +131 -0
- package/src/core/memory/memoryHit.js +1 -1
- package/src/core/memory/migrate.js +18 -11
- package/src/core/memory/migrateMapping.js +15 -7
- package/src/core/memory/mutateMemory.js +22 -4
- package/src/core/memory/recordIndex.js +10 -3
- package/src/core/memory/recordStore.js +28 -3
- package/src/core/memory/retrieval.js +79 -38
- package/src/core/memory/store.js +37 -37
- package/src/core/memory/storeV2.js +28 -25
- package/src/core/memory/storeV2Loader.js +2 -2
- package/src/core/observability/adapters/ingest.js +576 -0
- package/src/core/observability/analytics/anomalies.js +415 -0
- package/src/core/observability/analytics/summary.js +16 -1
- package/src/core/observability/emit/config.js +69 -1
- package/src/core/observability/emit/crash.js +434 -0
- package/src/core/observability/emit/lifecycle.js +349 -0
- package/src/core/observability/emit/recorder.js +135 -9
- package/src/core/observability/evaluation/aiPacket.js +52 -10
- package/src/core/observability/evaluation/outcomes.js +95 -0
- package/src/core/observability/evaluation/runner.js +225 -0
- package/src/core/observability/privacy/allowlist.js +23 -3
- package/src/core/observability/schema/compatibility.js +48 -3
- package/src/core/observability/schema/constants.js +5 -0
- package/src/core/observability/schema/registry.js +57 -0
- package/src/core/observability/schema/validate.js +68 -6
- package/src/core/observability/segments/internal.js +42 -8
- package/src/core/observability/segments/readSegments.js +35 -1
- package/src/core/observability/segments/recovery.js +3 -2
- package/src/core/observability/segments/retention.js +137 -33
- package/src/core/observability/support/projector.js +88 -18
- package/src/core/observability/support/provision.js +160 -0
- package/src/core/observability/support/renderer.js +2 -2
- package/src/core/observability/support/schedule.js +174 -0
- package/src/decision/registry.js +144 -0
- package/src/decision/reviewVerdict.js +309 -0
- package/template_project/.claude/agents/code-reviewer.md +25 -1
- package/template_project/.claude/agents/ukit-small-task-maintainer.md +16 -0
- package/template_project/.claude/commands/ukit/handoff-fullstack.md +2 -0
- package/template_project/.claude/commands/ukit/handoff-review.md +12 -0
- package/template_project/.claude/hooks/auto-allow-bash.sh +7 -1
- package/template_project/.claude/hooks/auto-prune-bash.sh +16 -7
- package/template_project/.claude/hooks/verification-guard.sh +13 -4
- package/template_project/.claude/ukit/index/review-verdict.mjs +592 -0
- package/template_project/.claude/ukit/index/sidecar-decision.mjs +595 -0
- package/template_project/.claude/ukit/index/unic-decision.mjs +10 -1
- package/template_project/.claude/ukit/runtime/async-lock.mjs +26 -0
- package/template_project/.codex/settings.json +3 -0
- package/template_project/.omp/agents/code-reviewer.md +25 -1
- package/template_project/.omp/agents/ukit-small-task-maintainer.md +16 -0
package/src/decision/registry.js
CHANGED
|
@@ -269,6 +269,150 @@ export const DECISION_REGISTRY = Object.freeze([
|
|
|
269
269
|
cacheSensitivity: 'none',
|
|
270
270
|
leasePolicy: 'none',
|
|
271
271
|
},
|
|
272
|
+
{
|
|
273
|
+
decisionKey: 'review.verdict.v1',
|
|
274
|
+
schemaVersion: 1,
|
|
275
|
+
family: 'review',
|
|
276
|
+
owner: 'handoffReview',
|
|
277
|
+
kind: 'choice',
|
|
278
|
+
description: 'Bounded review verdict among APPROVED | APPROVED-WITH-MINOR | CHANGES-REQUESTED | CRITICAL.',
|
|
279
|
+
candidatePolicy: 'fixed-verdict-vocabulary',
|
|
280
|
+
hardConstraints: ['verdict-vocabulary'],
|
|
281
|
+
probabilityPolicy: 'raw-label',
|
|
282
|
+
fallbackPolicy: 'deterministic-human-review',
|
|
283
|
+
telemetryClass: 'decision',
|
|
284
|
+
rolloutStage: 'off',
|
|
285
|
+
cacheSensitivity: 'none',
|
|
286
|
+
leasePolicy: 'none',
|
|
287
|
+
},
|
|
288
|
+
{
|
|
289
|
+
decisionKey: 'review.finding-bucket.v1',
|
|
290
|
+
schemaVersion: 1,
|
|
291
|
+
family: 'review',
|
|
292
|
+
owner: 'handoffReview',
|
|
293
|
+
kind: 'choice',
|
|
294
|
+
description: 'Triage bucket for one review finding among Act on | Consider | Noted | Dismissed.',
|
|
295
|
+
candidatePolicy: 'fixed-bucket-vocabulary',
|
|
296
|
+
hardConstraints: ['bucket-vocabulary'],
|
|
297
|
+
probabilityPolicy: 'raw-label',
|
|
298
|
+
fallbackPolicy: 'deterministic-human-review',
|
|
299
|
+
telemetryClass: 'decision',
|
|
300
|
+
rolloutStage: 'off',
|
|
301
|
+
cacheSensitivity: 'none',
|
|
302
|
+
leasePolicy: 'none',
|
|
303
|
+
},
|
|
304
|
+
{
|
|
305
|
+
decisionKey: 'review.panel-verdict.v1',
|
|
306
|
+
schemaVersion: 1,
|
|
307
|
+
family: 'review',
|
|
308
|
+
owner: 'handoffReview',
|
|
309
|
+
kind: 'choice',
|
|
310
|
+
description: 'Aggregated panel verdict for a review artifact.',
|
|
311
|
+
candidatePolicy: 'fixed-verdict-vocabulary',
|
|
312
|
+
hardConstraints: ['verdict-vocabulary'],
|
|
313
|
+
probabilityPolicy: 'raw-label',
|
|
314
|
+
fallbackPolicy: 'deterministic-human-review',
|
|
315
|
+
telemetryClass: 'decision',
|
|
316
|
+
rolloutStage: 'off',
|
|
317
|
+
cacheSensitivity: 'none',
|
|
318
|
+
leasePolicy: 'none',
|
|
319
|
+
},
|
|
320
|
+
{
|
|
321
|
+
decisionKey: 'workflow.sidecar-lane.v1',
|
|
322
|
+
schemaVersion: 1,
|
|
323
|
+
family: 'workflow',
|
|
324
|
+
owner: 'smallTaskMaintainer',
|
|
325
|
+
kind: 'choice',
|
|
326
|
+
description: 'Bounded decision whether a chore belongs to the sidecar lane.',
|
|
327
|
+
candidatePolicy: 'owner-shortlist',
|
|
328
|
+
hardConstraints: ['lane-eligibility'],
|
|
329
|
+
probabilityPolicy: 'raw-label',
|
|
330
|
+
fallbackPolicy: 'deterministic-lane-rules',
|
|
331
|
+
telemetryClass: 'decision',
|
|
332
|
+
rolloutStage: 'off',
|
|
333
|
+
cacheSensitivity: 'none',
|
|
334
|
+
leasePolicy: 'none',
|
|
335
|
+
},
|
|
336
|
+
{
|
|
337
|
+
decisionKey: 'workflow.sidecar-risk.v1',
|
|
338
|
+
schemaVersion: 1,
|
|
339
|
+
family: 'workflow',
|
|
340
|
+
owner: 'smallTaskMaintainer',
|
|
341
|
+
kind: 'choice',
|
|
342
|
+
description: 'Bounded risk class for a sidecar candidate (safe/reversible vs escalate).',
|
|
343
|
+
candidatePolicy: 'owner-shortlist',
|
|
344
|
+
hardConstraints: ['risk-floor'],
|
|
345
|
+
probabilityPolicy: 'raw-label',
|
|
346
|
+
fallbackPolicy: 'deterministic-lane-rules',
|
|
347
|
+
telemetryClass: 'decision',
|
|
348
|
+
rolloutStage: 'off',
|
|
349
|
+
cacheSensitivity: 'none',
|
|
350
|
+
leasePolicy: 'none',
|
|
351
|
+
},
|
|
352
|
+
{
|
|
353
|
+
decisionKey: 'workflow.routing-needed.v1',
|
|
354
|
+
schemaVersion: 1,
|
|
355
|
+
family: 'workflow',
|
|
356
|
+
owner: 'smallTaskMaintainer',
|
|
357
|
+
kind: 'choice',
|
|
358
|
+
description: 'Whether routing/intent classification is needed for a prompt.',
|
|
359
|
+
candidatePolicy: 'owner-shortlist',
|
|
360
|
+
hardConstraints: ['lane-eligibility'],
|
|
361
|
+
probabilityPolicy: 'raw-label',
|
|
362
|
+
fallbackPolicy: 'deterministic-route',
|
|
363
|
+
telemetryClass: 'decision',
|
|
364
|
+
rolloutStage: 'off',
|
|
365
|
+
cacheSensitivity: 'none',
|
|
366
|
+
leasePolicy: 'none',
|
|
367
|
+
},
|
|
368
|
+
{
|
|
369
|
+
decisionKey: 'workflow.step-budget.v1',
|
|
370
|
+
schemaVersion: 1,
|
|
371
|
+
family: 'workflow',
|
|
372
|
+
owner: 'smallTaskMaintainer',
|
|
373
|
+
kind: 'choice',
|
|
374
|
+
description: 'Bounded step-budget class for an in-flight task.',
|
|
375
|
+
candidatePolicy: 'owner-shortlist',
|
|
376
|
+
hardConstraints: ['task-boundary'],
|
|
377
|
+
probabilityPolicy: 'raw-label',
|
|
378
|
+
fallbackPolicy: 'deterministic-budget',
|
|
379
|
+
telemetryClass: 'decision',
|
|
380
|
+
rolloutStage: 'off',
|
|
381
|
+
cacheSensitivity: 'none',
|
|
382
|
+
leasePolicy: 'none',
|
|
383
|
+
},
|
|
384
|
+
{
|
|
385
|
+
decisionKey: 'workflow.compact-now.v1',
|
|
386
|
+
schemaVersion: 1,
|
|
387
|
+
family: 'workflow',
|
|
388
|
+
owner: 'smallTaskMaintainer',
|
|
389
|
+
kind: 'choice',
|
|
390
|
+
description: 'Whether to compact context at the current boundary.',
|
|
391
|
+
candidatePolicy: 'owner-shortlist',
|
|
392
|
+
hardConstraints: ['task-boundary'],
|
|
393
|
+
probabilityPolicy: 'raw-label',
|
|
394
|
+
fallbackPolicy: 'deterministic-threshold',
|
|
395
|
+
telemetryClass: 'decision',
|
|
396
|
+
rolloutStage: 'off',
|
|
397
|
+
cacheSensitivity: 'none',
|
|
398
|
+
leasePolicy: 'none',
|
|
399
|
+
},
|
|
400
|
+
{
|
|
401
|
+
decisionKey: 'workflow.summarize-vs-keep.v1',
|
|
402
|
+
schemaVersion: 1,
|
|
403
|
+
family: 'workflow',
|
|
404
|
+
owner: 'smallTaskMaintainer',
|
|
405
|
+
kind: 'choice',
|
|
406
|
+
description: 'Summarize vs keep for a stored artifact/message.',
|
|
407
|
+
candidatePolicy: 'owner-shortlist',
|
|
408
|
+
hardConstraints: ['task-boundary', 'source-freshness'],
|
|
409
|
+
probabilityPolicy: 'raw-label',
|
|
410
|
+
fallbackPolicy: 'deterministic-threshold',
|
|
411
|
+
telemetryClass: 'decision',
|
|
412
|
+
rolloutStage: 'off',
|
|
413
|
+
cacheSensitivity: 'none',
|
|
414
|
+
leasePolicy: 'none',
|
|
415
|
+
},
|
|
272
416
|
]);
|
|
273
417
|
|
|
274
418
|
// Deep-freeze the shipped catalog — entries are shared metadata consumed by
|
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* reviewVerdict.js (S2 / UNIC_DECISION_MIGRATION)
|
|
3
|
+
*
|
|
4
|
+
* Review-verdict producer for the `review` decision family. Reviewers (the
|
|
5
|
+
* general-LLM code-reviewer agents) still generate findings; when
|
|
6
|
+
* `decisionPlane.families.review.stage` is promoted past 'off', the FINAL
|
|
7
|
+
* verdict and per-finding bucket pass runs through unic-decision instead.
|
|
8
|
+
* Deterministic severity aggregation stays authoritative on 'off' and on
|
|
9
|
+
* every failure path — never another LLM for the verdict.
|
|
10
|
+
*
|
|
11
|
+
* Registered keys (src/decision/registry.js):
|
|
12
|
+
* review.verdict.v1 — solo reviewer verdict (choice)
|
|
13
|
+
* review.finding-bucket.v1 — per-finding triage bucket (choice; asked once
|
|
14
|
+
* per finding, wire name carries `#<id>`)
|
|
15
|
+
* review.panel-verdict.v1 — aggregated panel verdict (choice)
|
|
16
|
+
*
|
|
17
|
+
* Transport discipline (C13/statePacket conventions): only whitelisted,
|
|
18
|
+
* truncated finding fields cross the wire — {id, severity, location, claim}.
|
|
19
|
+
* No diff hunks, no prose, no secrets. Everything below is pure — no I/O —
|
|
20
|
+
* and never throws (client.js precedent).
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
export const VERDICT_CANDIDATES = Object.freeze([
|
|
24
|
+
'APPROVED',
|
|
25
|
+
'APPROVED-WITH-MINOR',
|
|
26
|
+
'CHANGES-REQUESTED',
|
|
27
|
+
'CRITICAL',
|
|
28
|
+
]);
|
|
29
|
+
|
|
30
|
+
export const BUCKET_CANDIDATES = Object.freeze([
|
|
31
|
+
'Act on',
|
|
32
|
+
'Consider',
|
|
33
|
+
'Noted',
|
|
34
|
+
'Dismissed',
|
|
35
|
+
]);
|
|
36
|
+
|
|
37
|
+
export const REVIEW_VERDICT_KEYS = Object.freeze({
|
|
38
|
+
solo: 'review.verdict.v1',
|
|
39
|
+
bucket: 'review.finding-bucket.v1',
|
|
40
|
+
panel: 'review.panel-verdict.v1',
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
// Bounds keep the batch inside the state budget and the tool list small.
|
|
44
|
+
// MAX_FINDINGS mirrors statePacket.js MAX_ARRAY_ITEMS (24); a question plus a
|
|
45
|
+
// verdict slot means at most MAX_FINDINGS bucket questions per batch.
|
|
46
|
+
export const MAX_FINDINGS = 24;
|
|
47
|
+
export const MAX_CLAIM_LENGTH = 160;
|
|
48
|
+
const MAX_LOCATION_LENGTH = 160;
|
|
49
|
+
const MAX_ID_LENGTH = 64;
|
|
50
|
+
const MAX_SEVERITY_LENGTH = 24;
|
|
51
|
+
|
|
52
|
+
const KNOWN_SEVERITIES = new Set(['critical', 'important', 'minor']);
|
|
53
|
+
|
|
54
|
+
const BUCKET_INSTRUCTION =
|
|
55
|
+
'Triage this single review finding into exactly one bucket: '
|
|
56
|
+
+ '"Act on" feeds the auto-fix loop; "Consider" is advisory but worth the '
|
|
57
|
+
+ 'author\'s attention; "Noted" is informational only; "Dismissed" is '
|
|
58
|
+
+ 'rejected as not actionable.';
|
|
59
|
+
|
|
60
|
+
function isPlainObject(value) {
|
|
61
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
function collapseWhitespace(text) {
|
|
65
|
+
return String(text).replace(/\s+/g, ' ').trim();
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function truncate(text, max) {
|
|
69
|
+
return text.length > max ? `${text.slice(0, max - 1)}…` : text;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function normalizeSeverity(raw) {
|
|
73
|
+
const severity = typeof raw === 'string'
|
|
74
|
+
? raw.trim().toLowerCase().slice(0, MAX_SEVERITY_LENGTH)
|
|
75
|
+
: '';
|
|
76
|
+
return KNOWN_SEVERITIES.has(severity) ? severity : 'unknown';
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// Whitelist one finding down to the exact fields the verdict pass is allowed
|
|
80
|
+
// to see. Location collapses file:line into a single label; every other
|
|
81
|
+
// field on the input object is dropped silently.
|
|
82
|
+
function sanitizeFinding(raw, index, usedIds) {
|
|
83
|
+
const source = isPlainObject(raw) ? raw : {};
|
|
84
|
+
let id = typeof source.id === 'string' && source.id.trim().length > 0
|
|
85
|
+
? source.id.trim().replace(/[^A-Za-z0-9._-]/g, '_').slice(0, MAX_ID_LENGTH)
|
|
86
|
+
: `F${index + 1}`;
|
|
87
|
+
if (id.length === 0 || usedIds.has(id)) {
|
|
88
|
+
// Collision fallback must itself be unique — two findings both claiming
|
|
89
|
+
// 'F2' must not collapse into one wire name / one bucket row.
|
|
90
|
+
let suffix = '';
|
|
91
|
+
while (usedIds.has(`F${index + 1}${suffix}`)) suffix = `${suffix}x`;
|
|
92
|
+
id = `F${index + 1}${suffix}`;
|
|
93
|
+
}
|
|
94
|
+
usedIds.add(id);
|
|
95
|
+
|
|
96
|
+
let file = typeof source.file === 'string' ? collapseWhitespace(source.file) : '';
|
|
97
|
+
const line = Number.isInteger(source.line) && source.line > 0 ? source.line : null;
|
|
98
|
+
if (file.length > 0) file = truncate(file, MAX_LOCATION_LENGTH - 8);
|
|
99
|
+
const location = file.length > 0
|
|
100
|
+
? (line !== null ? `${file}:${line}` : file)
|
|
101
|
+
: (line !== null ? `line ${line}` : '-');
|
|
102
|
+
|
|
103
|
+
const claimSource = [source.claim, source.title, source.summary, source.message]
|
|
104
|
+
.find((v) => typeof v === 'string' && v.trim().length > 0);
|
|
105
|
+
|
|
106
|
+
return {
|
|
107
|
+
id,
|
|
108
|
+
severity: normalizeSeverity(source.severity),
|
|
109
|
+
location,
|
|
110
|
+
claim: truncate(collapseWhitespace(claimSource ?? ''), MAX_CLAIM_LENGTH),
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
// Stable short fingerprint for the batch id — same findings → same batchId
|
|
115
|
+
// across runs, so receipts and logs reconcile.
|
|
116
|
+
function stableBatchId(findings) {
|
|
117
|
+
let h = 0x811c9dc5;
|
|
118
|
+
const text = JSON.stringify(findings.map((f) => [f.id, f.severity, f.location, f.claim]));
|
|
119
|
+
for (let i = 0; i < text.length; i += 1) {
|
|
120
|
+
h ^= text.charCodeAt(i);
|
|
121
|
+
h = Math.imul(h, 0x01000193);
|
|
122
|
+
}
|
|
123
|
+
return `rv-${(h >>> 0).toString(16).padStart(8, '0')}`;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Build the C13 question batch for the review verdict pass.
|
|
128
|
+
*
|
|
129
|
+
* @param {object} [input]
|
|
130
|
+
* @param {Array} [input.findings] extracted review findings (whitelisted
|
|
131
|
+
* fields only: id/severity/file/line/claim|title|summary|message).
|
|
132
|
+
* @param {string} [input.mode] 'solo' → review.verdict.v1 question;
|
|
133
|
+
* 'panel' → review.panel-verdict.v1 instead.
|
|
134
|
+
* @returns {{questions: object[], statePacket: object, batch: object,
|
|
135
|
+
* findings: object[], verdictDecisionKey: string}}
|
|
136
|
+
* Never throws — garbage input sanitizes to an empty finding list.
|
|
137
|
+
*/
|
|
138
|
+
export function buildReviewVerdictBatch({ findings = [], mode = 'solo' } = {}) {
|
|
139
|
+
const usedIds = new Set();
|
|
140
|
+
const sanitized = (Array.isArray(findings) ? findings : [])
|
|
141
|
+
.slice(0, MAX_FINDINGS)
|
|
142
|
+
.map((raw, index) => sanitizeFinding(raw, index, usedIds));
|
|
143
|
+
|
|
144
|
+
const verdictDecisionKey = mode === 'panel'
|
|
145
|
+
? REVIEW_VERDICT_KEYS.panel
|
|
146
|
+
: REVIEW_VERDICT_KEYS.solo;
|
|
147
|
+
|
|
148
|
+
const counts = { critical: 0, important: 0, minor: 0, unknown: 0 };
|
|
149
|
+
for (const f of sanitized) counts[f.severity] += 1;
|
|
150
|
+
|
|
151
|
+
const verdictQuestion = {
|
|
152
|
+
decisionKey: verdictDecisionKey,
|
|
153
|
+
schemaVersion: 1,
|
|
154
|
+
family: 'review',
|
|
155
|
+
owner: 'handoffReview',
|
|
156
|
+
kind: 'choice',
|
|
157
|
+
instruction:
|
|
158
|
+
'Pick the bounded review verdict for this finding set. '
|
|
159
|
+
+ 'APPROVED = clean; APPROVED-WITH-MINOR = ship with nits logged; '
|
|
160
|
+
+ 'CHANGES-REQUESTED = must fix before merge; CRITICAL = unsafe to ship.',
|
|
161
|
+
candidates: [...VERDICT_CANDIDATES],
|
|
162
|
+
hardConstraintRefs: ['verdict-vocabulary'],
|
|
163
|
+
};
|
|
164
|
+
|
|
165
|
+
// One bucket question per finding. The shared registered key cannot repeat
|
|
166
|
+
// in one batch (name-based reconciliation), so each wire name carries
|
|
167
|
+
// `#<findingId>`; parseVerdictAnswer splits it back off.
|
|
168
|
+
const bucketQuestions = sanitized.map((f) => ({
|
|
169
|
+
decisionKey: `${REVIEW_VERDICT_KEYS.bucket}#${f.id}`,
|
|
170
|
+
schemaVersion: 1,
|
|
171
|
+
family: 'review',
|
|
172
|
+
owner: 'handoffReview',
|
|
173
|
+
kind: 'choice',
|
|
174
|
+
instruction: `${BUCKET_INSTRUCTION} Finding ${f.id} [${f.severity}] ${f.location}: ${f.claim || '(no claim text)'}`,
|
|
175
|
+
candidates: [...BUCKET_CANDIDATES],
|
|
176
|
+
hardConstraintRefs: ['bucket-vocabulary'],
|
|
177
|
+
}));
|
|
178
|
+
|
|
179
|
+
const questions = [verdictQuestion, ...bucketQuestions];
|
|
180
|
+
|
|
181
|
+
// Producer-owned packet — only sanitized whitelisted fields. `findingCount`
|
|
182
|
+
// is the only aggregate the model needs; severities ride the findings array.
|
|
183
|
+
const statePacket = {
|
|
184
|
+
kind: 'review-verdict',
|
|
185
|
+
mode: mode === 'panel' ? 'panel' : 'solo',
|
|
186
|
+
findingCount: sanitized.length,
|
|
187
|
+
severityCounts: counts,
|
|
188
|
+
findings: sanitized,
|
|
189
|
+
};
|
|
190
|
+
|
|
191
|
+
const batch = {
|
|
192
|
+
batchVersion: 1,
|
|
193
|
+
batchId: stableBatchId(sanitized),
|
|
194
|
+
boundary: 'review-verdict',
|
|
195
|
+
questions,
|
|
196
|
+
statePacket,
|
|
197
|
+
};
|
|
198
|
+
|
|
199
|
+
return { questions, statePacket, batch, findings: sanitized, verdictDecisionKey };
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Deterministic verdict from finding severities — the authoritative path at
|
|
204
|
+
* stage 'off' and the fallback on every failure (unic-decision unavailable,
|
|
205
|
+
* malformed/invalid answers). Rule (documented in review-verdict.mjs):
|
|
206
|
+
* any critical → CHANGES-REQUESTED
|
|
207
|
+
* ≥1 important or ≥1 unclassified → APPROVED-WITH-MINOR
|
|
208
|
+
* else → APPROVED
|
|
209
|
+
*/
|
|
210
|
+
export function deriveDeterministicVerdict(findings) {
|
|
211
|
+
const list = Array.isArray(findings) ? findings : [];
|
|
212
|
+
let importantish = false;
|
|
213
|
+
for (const raw of list) {
|
|
214
|
+
const severity = isPlainObject(raw) ? normalizeSeverity(raw.severity) : 'unknown';
|
|
215
|
+
if (severity === 'critical') return 'CHANGES-REQUESTED';
|
|
216
|
+
if (severity === 'important' || severity === 'unknown') importantish = true;
|
|
217
|
+
}
|
|
218
|
+
return importantish ? 'APPROVED-WITH-MINOR' : 'APPROVED';
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Deterministic per-finding bucket — the fallback the lead reviewer's
|
|
223
|
+
* bucketing judgment replaces only when the family stage is enabled.
|
|
224
|
+
* critical | important | unclassified → 'Act on'
|
|
225
|
+
* minor → 'Noted'
|
|
226
|
+
* Conservative: an unknown severity feeds the auto-fix loop rather than
|
|
227
|
+
* being silently dismissed.
|
|
228
|
+
*/
|
|
229
|
+
export function deriveDeterministicBucket(finding) {
|
|
230
|
+
const severity = normalizeSeverity(isPlainObject(finding) ? finding.severity : null);
|
|
231
|
+
return severity === 'minor' ? 'Noted' : 'Act on';
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
// Wire name → {decisionKey, findingId}: 'review.finding-bucket.v1#F3' splits
|
|
235
|
+
// into the registered key plus the finding the bucket answer belongs to.
|
|
236
|
+
function splitWireName(name) {
|
|
237
|
+
const text = String(name ?? '');
|
|
238
|
+
const hash = text.lastIndexOf('#');
|
|
239
|
+
if (hash > 0 && text.slice(0, hash) === REVIEW_VERDICT_KEYS.bucket) {
|
|
240
|
+
return { decisionKey: REVIEW_VERDICT_KEYS.bucket, findingId: text.slice(hash + 1) };
|
|
241
|
+
}
|
|
242
|
+
return { decisionKey: text, findingId: null };
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
/**
|
|
246
|
+
* Fold a DecisionBatchResult (from client.requestBatch or an offline
|
|
247
|
+
* parseBatchResponse) into the review-verdict shape. Never throws; every
|
|
248
|
+
* invalid/missing answer degrades to `null`/deterministic-bucket so callers
|
|
249
|
+
* can substitute deriveDeterministicVerdict as the authoritative fallback.
|
|
250
|
+
*
|
|
251
|
+
* @param {object} [result] DecisionBatchResult — {status, answers}.
|
|
252
|
+
* @param {object} [opts]
|
|
253
|
+
* @param {Array} [opts.findings] sanitized or raw findings — supplies the
|
|
254
|
+
* bucket fold order and per-finding fallback values.
|
|
255
|
+
* @returns {{status: string, verdict: string|null, verdictStatus: string,
|
|
256
|
+
* buckets: Array<{findingId, bucket, bucketStatus}>}}
|
|
257
|
+
*/
|
|
258
|
+
export function parseVerdictAnswer(result, { findings = [] } = {}) {
|
|
259
|
+
const status = typeof result?.status === 'string' ? result.status : 'invalid';
|
|
260
|
+
const answers = Array.isArray(result?.answers) ? result.answers : [];
|
|
261
|
+
const list = Array.isArray(findings) ? findings : [];
|
|
262
|
+
let verdict = null;
|
|
263
|
+
let verdictStatus = 'missing';
|
|
264
|
+
// findingId → {bucket, bucketStatus}; 'default' marks the deterministic
|
|
265
|
+
// fallback standing in because the batch never produced an answer for it.
|
|
266
|
+
const bucketByFinding = new Map();
|
|
267
|
+
for (const raw of list) {
|
|
268
|
+
if (!isPlainObject(raw)) continue;
|
|
269
|
+
const id = typeof raw.id === 'string' && raw.id.length > 0 ? raw.id : null;
|
|
270
|
+
if (id === null) continue;
|
|
271
|
+
bucketByFinding.set(id, {
|
|
272
|
+
bucket: deriveDeterministicBucket(raw),
|
|
273
|
+
bucketStatus: 'default',
|
|
274
|
+
});
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
for (const answer of answers) {
|
|
278
|
+
if (!isPlainObject(answer)) continue;
|
|
279
|
+
const { decisionKey, findingId } = splitWireName(answer.decisionKey);
|
|
280
|
+
const valid = answer.validationStatus === 'valid'
|
|
281
|
+
&& typeof answer.value === 'string';
|
|
282
|
+
if (decisionKey === REVIEW_VERDICT_KEYS.solo
|
|
283
|
+
|| decisionKey === REVIEW_VERDICT_KEYS.panel) {
|
|
284
|
+
if (valid && VERDICT_CANDIDATES.includes(answer.value)) {
|
|
285
|
+
verdict = answer.value;
|
|
286
|
+
verdictStatus = 'valid';
|
|
287
|
+
} else {
|
|
288
|
+
verdictStatus = 'invalid';
|
|
289
|
+
}
|
|
290
|
+
continue;
|
|
291
|
+
}
|
|
292
|
+
if (decisionKey === REVIEW_VERDICT_KEYS.bucket && findingId !== null) {
|
|
293
|
+
bucketByFinding.set(
|
|
294
|
+
findingId,
|
|
295
|
+
valid && BUCKET_CANDIDATES.includes(answer.value)
|
|
296
|
+
? { bucket: answer.value, bucketStatus: 'valid' }
|
|
297
|
+
: { bucket: deriveDeterministicBucket({ severity: 'unknown' }), bucketStatus: 'invalid' },
|
|
298
|
+
);
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
const buckets = [...bucketByFinding.entries()].map(([findingId, entry]) => ({
|
|
303
|
+
findingId,
|
|
304
|
+
bucket: entry.bucket,
|
|
305
|
+
bucketStatus: entry.bucketStatus,
|
|
306
|
+
}));
|
|
307
|
+
|
|
308
|
+
return { status, verdict, verdictStatus, buckets };
|
|
309
|
+
}
|
|
@@ -49,6 +49,26 @@ If any input is missing, return `CHANGES-REQUESTED` with reason "incomplete hand
|
|
|
49
49
|
- **APPROVED-WITH-MINOR** — Minor naming / doc / style issues. Logged on task file but handoff allowed.
|
|
50
50
|
- **APPROVED** — Clean.
|
|
51
51
|
|
|
52
|
+
### Stage gate — verdict emission (unic-decision)
|
|
53
|
+
|
|
54
|
+
You generate the findings; the FINAL verdict/bucket classification is a bounded
|
|
55
|
+
decision owned by the local decision model when staged. Read
|
|
56
|
+
`decisionPlane.families.review.stage` from `.ukit/storage/config.json`:
|
|
57
|
+
|
|
58
|
+
- `off` (default) or any adapter failure (`outcomeClass` ≠ `accepted`/`partial`) →
|
|
59
|
+
you emit the verdict directly, exactly as today; the deterministic severity
|
|
60
|
+
aggregate (any critical → CHANGES-REQUESTED; ≥1 important or any
|
|
61
|
+
unclassified/unknown severity → APPROVED-WITH-MINOR; else APPROVED) stays
|
|
62
|
+
authoritative.
|
|
63
|
+
- not `off` → emit the verdict via `node .claude/ukit/index/review-verdict.mjs`
|
|
64
|
+
(registered key `review.verdict.v1`): pipe your extracted findings as JSON on
|
|
65
|
+
stdin — whitelisted fields only (`id`, `severity`, `file`, `line`, short
|
|
66
|
+
`claim`; never diff hunks or secrets). Use the adapter's `verdict`/`buckets`
|
|
67
|
+
fields in the `## Reviewer Verdict` block.
|
|
68
|
+
|
|
69
|
+
Model isolation is preserved: the verdict pass runs on the local unic-decision
|
|
70
|
+
model, so reviewer ≠ executor still stands.
|
|
71
|
+
|
|
52
72
|
### Output (append to task file as `## Reviewer Verdict`)
|
|
53
73
|
|
|
54
74
|
```
|
|
@@ -138,7 +158,11 @@ verdict block — with these additions:
|
|
|
138
158
|
`node .claude/ukit/index/review-panel-aggregate.mjs <TASK-xxx.md...>` and hands you
|
|
139
159
|
the output, the lead fills `AGREEMENT_MAP` with the emitted finding → members map
|
|
140
160
|
and applies the lead-judgment buckets to every finding: **Act on** / **Consider** /
|
|
141
|
-
**Noted** / **Dismissed**.
|
|
161
|
+
**Noted** / **Dismissed**. When `decisionPlane.families.review.stage` is not `off`,
|
|
162
|
+
the lead's bucketing and panel verdict go through
|
|
163
|
+
`node .claude/ukit/index/review-verdict.mjs --panel` (`review.finding-bucket.v1` /
|
|
164
|
+
`review.panel-verdict.v1`) per the stage gate above — the deterministic aggregate
|
|
165
|
+
stays authoritative at `off` or on adapter failure.
|
|
142
166
|
- `consensus≥2 identical findings = high signal`: a finding reported by two or more
|
|
143
167
|
panel members is high-signal and must not be bucketed below **Consider** without a
|
|
144
168
|
stated reason.
|
|
@@ -27,6 +27,22 @@ You are UKit's internal small-task maintainer. You run as a sidecar/parallel/non
|
|
|
27
27
|
- Keeping agent context compact without removing existing lanes: Claude PreCompact/reinject stays active and Codex Desktop soft handoffs use `compact.codexContext.compactTarget` (default 150 lines; preferred 120-150; hard max 170) while preserving critical state.
|
|
28
28
|
- Small, reversible UKit runtime maintenance decisions.
|
|
29
29
|
|
|
30
|
+
## Bounded Decisions (stage-gated unic-decision)
|
|
31
|
+
|
|
32
|
+
The six bounded decisions enumerated in `.codex/settings.json` `smallTaskModel.decisionPolicy.decisions` — `fast-vs-slow-lane`, `safe-vs-risky-lane`, `skill-routing-needed`, `step-budget-enough`, `compact-now-or-later`, `summarize-docs-or-keep-detail` — consult `unic-decision` first via the installed CLI when the decision-plane stage allows:
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
node .claude/ukit/index/sidecar-decision.mjs --decision <name> [--root <dir>] # context JSON on stdin
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
The CLI resolves `decisionPlane.families.workflow.stage` and owns the name → `workflow.*` decision-key mapping (`--list` prints it):
|
|
39
|
+
|
|
40
|
+
- `off` (default): the CLI's deterministic rules answer directly — zero transport, authoritative.
|
|
41
|
+
- `shadow`: run the batch for comparison only; the deterministic rule answer stays authoritative.
|
|
42
|
+
- `canary`/`default`: a valid unic-decision answer wins; an invalid or unavailable adapter falls back to the deterministic rule (`outcomeClass: 'unavailable'`).
|
|
43
|
+
|
|
44
|
+
`unic-decision` is the only model allowed in these decision steps — on any failure the deterministic rule is the fallback, never another LLM for the verdict. This lane's own model (`unic-lite`) keeps generative work only: summarization, doc maintenance, and cleanup — it never emits a verdict for these decisions.
|
|
45
|
+
|
|
30
46
|
## Never Use For
|
|
31
47
|
|
|
32
48
|
- Security/auth/permission/secrets work.
|
|
@@ -576,6 +576,8 @@ git status # overview of modified/new/deleted files
|
|
|
576
576
|
|
|
577
577
|
Review as one unified diff — correctness, regression risk, security, edge cases, maintainability. Cross-reference each task's intent in `docs/AI_HANDOFF/tasks/TASK-xxx.md`.
|
|
578
578
|
|
|
579
|
+
Stage gate (unic-decision): reviewers still generate findings; when `decisionPlane.families.review.stage` is not `off`, the emitted verdict goes through `node .claude/ukit/index/review-verdict.mjs` (`review.verdict.v1`; `--panel` for `review.panel-verdict.v1`/`review.finding-bucket.v1`) — the deterministic severity aggregate stays authoritative at `off` or on adapter failure.
|
|
580
|
+
|
|
579
581
|
Append verdict to each task file:
|
|
580
582
|
```
|
|
581
583
|
## Reviewer Verdict
|
|
@@ -141,6 +141,18 @@ carrying the Agreement Map and the bucketed findings. `Act on` findings feed Ste
|
|
|
141
141
|
auto-fix loop; `Consider`/`Noted`/`Dismissed` are advisory. A high-signal finding must
|
|
142
142
|
not be bucketed below `Consider` without a stated reason.
|
|
143
143
|
|
|
144
|
+
Stage gate (unic-decision): findings stay generated by the general-LLM panel — only
|
|
145
|
+
the FINAL verdict/bucket pass consults the local decision model. When
|
|
146
|
+
`decisionPlane.families.review.stage` (`.ukit/storage/config.json`) is not `off`, the
|
|
147
|
+
lead member's finding bucketing and the panel verdict go through
|
|
148
|
+
`node .claude/ukit/index/review-verdict.mjs --panel` (registered keys
|
|
149
|
+
`review.panel-verdict.v1` + `review.finding-bucket.v1`) instead of free judgment;
|
|
150
|
+
pipe the extracted findings (id/severity/file:line/short claim — never diff hunks)
|
|
151
|
+
as JSON on stdin. At stage `off` or on any `outcomeClass` ≠ `accepted`/`partial`
|
|
152
|
+
(gateway unavailable, invalid answers) the deterministic severity aggregate above
|
|
153
|
+
stays authoritative. Model isolation is preserved: the verdict pass is the local
|
|
154
|
+
decision model, so reviewer ≠ executor still stands.
|
|
155
|
+
|
|
144
156
|
### 2f — Orchestrator updates INDEX.md
|
|
145
157
|
|
|
146
158
|
Set `status = NEXT_STATUS_FOR_INDEX`. Orchestrator writes INDEX, not the reviewer.
|
|
@@ -149,7 +149,13 @@ UKIT_RUNTIME_DIR="$SCRIPT_DIR/../ukit/runtime" node -e '
|
|
|
149
149
|
// is still legally waiting for the lock.
|
|
150
150
|
const HOOK_DEADLINE_MS = Number.parseInt(process.env.UKIT_HOOK_DEADLINE_MS || "", 10) || 8000;
|
|
151
151
|
const LOCK_STARTED_AT = Date.now();
|
|
152
|
-
setTimeout(() =>
|
|
152
|
+
setTimeout(() => {
|
|
153
|
+
// A wedged fs op (fifo/NFS as a state file) parks a libuv threadpool thread and
|
|
154
|
+
// the atexit teardown of process.exit()/reallyExit() waits for it forever — the
|
|
155
|
+
// deadline can only escape by killing itself. The bash call is `|| true`-masked,
|
|
156
|
+
// so the observable contract stays "exit 0".
|
|
157
|
+
try { process.kill(process.pid, "SIGKILL"); } catch {}
|
|
158
|
+
}, HOOK_DEADLINE_MS).unref();
|
|
153
159
|
const fsp = require("fs").promises;
|
|
154
160
|
const path = require("path");
|
|
155
161
|
const { pathToFileURL } = require("url");
|
|
@@ -54,18 +54,27 @@ const LOCK_STARTED_AT = Date.now();
|
|
|
54
54
|
// SPEC §8: a failed prune abandons configured work — announce it (success stays silent).
|
|
55
55
|
// Unpruned rules are harmless (stale entries simply never match again), so degrade is
|
|
56
56
|
// advisory-only and still exits 0.
|
|
57
|
-
function emitDegrade(reason) {
|
|
57
|
+
function emitDegrade(reason, onDone) {
|
|
58
58
|
try {
|
|
59
|
-
process.stdout.write(
|
|
60
|
-
systemMessage: `UKit auto-prune-bash: ${reason}
|
|
61
|
-
|
|
62
|
-
|
|
59
|
+
process.stdout.write(
|
|
60
|
+
JSON.stringify({ systemMessage: `UKit auto-prune-bash: ${reason}` }) + "\n",
|
|
61
|
+
() => { try { onDone?.(); } catch {} },
|
|
62
|
+
);
|
|
63
|
+
} catch {
|
|
64
|
+
try { onDone?.(); } catch {}
|
|
65
|
+
}
|
|
63
66
|
}
|
|
64
67
|
|
|
65
68
|
setTimeout(() => {
|
|
66
|
-
emitDegrade("prune exceeded its deadline; stale Bash auto-allow rules were not pruned this session — the next session retries.")
|
|
67
|
-
|
|
69
|
+
emitDegrade("prune exceeded its deadline; stale Bash auto-allow rules were not pruned this session — the next session retries.", () => {
|
|
70
|
+
// The degrade line must be flushed BEFORE the kill, and process.exit() cannot
|
|
71
|
+
// be used here: a wedged fs op parks a libuv threadpool thread and atexit
|
|
72
|
+
// teardown waits for it forever. SIGKILL is the only guaranteed escape; the
|
|
73
|
+
// bash call is `|| true`-masked, so the observable contract stays "exit 0".
|
|
74
|
+
try { process.kill(process.pid, "SIGKILL"); } catch {}
|
|
75
|
+
});
|
|
68
76
|
}, HOOK_DEADLINE_MS).unref();
|
|
77
|
+
|
|
69
78
|
const fsp = require("fs").promises;
|
|
70
79
|
const path = require("path");
|
|
71
80
|
const { pathToFileURL } = require("url");
|
|
@@ -71,9 +71,16 @@ setTimeout(() => {
|
|
|
71
71
|
try {
|
|
72
72
|
process.stdout.write(JSON.stringify({
|
|
73
73
|
systemMessage: `UKit verification-guard: evaluation exceeded its ${HOOK_DEADLINE_MS}ms deadline; this command was allowed without verification tracking.`,
|
|
74
|
-
}) + '\n')
|
|
75
|
-
|
|
76
|
-
|
|
74
|
+
}) + '\n', () => {
|
|
75
|
+
// The degrade line must be flushed BEFORE the kill, and process.exit() cannot
|
|
76
|
+
// be used here: a wedged fs op parks a libuv threadpool thread and atexit
|
|
77
|
+
// teardown waits for it forever. SIGKILL is the only guaranteed escape; the
|
|
78
|
+
// wrapper exits 0 regardless of node status (advisory contract).
|
|
79
|
+
try { process.kill(process.pid, 'SIGKILL'); } catch {}
|
|
80
|
+
});
|
|
81
|
+
} catch {
|
|
82
|
+
try { process.kill(process.pid, 'SIGKILL'); } catch {}
|
|
83
|
+
}
|
|
77
84
|
}, HOOK_DEADLINE_MS).unref();
|
|
78
85
|
// TASK-015 fix round 1: the awaitable fs surface. Every state/progress read and
|
|
79
86
|
// the atomic progress mutation is awaited so the unref'd self-deadline above can
|
|
@@ -570,4 +577,6 @@ process.exit(0);
|
|
|
570
577
|
});
|
|
571
578
|
NODE
|
|
572
579
|
|
|
573
|
-
|
|
580
|
+
# Advisory hook: the deadline may SIGKILL node past a wedged fs op — the hook
|
|
581
|
+
# contract is still always-exit-0 regardless of the child status.
|
|
582
|
+
exit 0
|