@ngockhoale/ukit 3.3.3 → 3.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/manifests/engineConformance.yaml +17 -1
- package/manifests/hostCapabilities.yaml +68 -1
- package/manifests/platform.full.yaml +138 -0
- package/manifests/platform.user.yaml +255 -3
- package/package.json +1 -1
- package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
- package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
- package/scripts/probe/codex-capability-probe.mjs +169 -0
- package/src/cli/commands/doctor.js +168 -0
- package/src/cli/commands/indexTools.js +7 -0
- package/src/cli/commands/metrics.js +66 -2
- package/src/cli/commands/playbook.js +4 -4
- package/src/cli/commands/vm.js +49 -8
- package/src/core/agentRuntime/adapters.js +328 -27
- package/src/core/agentRuntime/artifacts.js +89 -0
- package/src/core/agentRuntime/context.js +345 -1
- package/src/core/agentRuntime/contract.js +296 -0
- package/src/core/agentRuntime/eventStore.js +176 -0
- package/src/core/agentRuntime/shadowRun.js +481 -5
- package/src/core/agentRuntime/telemetry.js +121 -0
- package/src/core/observability/emit/lifecycle.js +68 -1
- package/src/core/observability/emit/sessionBoot.js +393 -0
- package/src/core/observability/privacy/allowlist.js +10 -1
- package/src/core/observability/schema/registry.js +10 -0
- package/src/core/runtimeConfig.js +133 -0
- package/src/core/userPlaybooks.js +18 -3
- package/src/decision/registry.js +19 -0
- package/src/diagnostics/feedbackEvents.js +7 -4
- package/src/diagnostics/routeOutcomes.js +51 -6
- package/src/diagnostics/skillAccuracy.js +43 -3
- package/src/index/crossCheckMatrix.js +412 -0
- package/src/index/fixLoopEscalation.js +453 -0
- package/src/index/playbookRegistry.js +691 -0
- package/src/index/reviewPolicy.js +368 -0
- package/src/index/routeResolver.js +915 -0
- package/src/index/sessionHistoryExtractor.js +359 -0
- package/src/index/taskRouting.js +764 -581
- package/src/index/tierSelection.js +308 -0
- package/src/index/verificationMap.js +404 -0
- package/template_project/.claude/hooks/observability-emit.mjs +14 -0
- package/template_project/.claude/hooks/record-execution.mjs +19 -1
- package/template_project/.claude/hooks/skill-router.sh +691 -25
- package/template_project/.claude/hooks/verification-guard.sh +230 -1
- package/template_project/.claude/settings.json +2 -2
- package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
- package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
- package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
- package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
- package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
- package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
- package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
- package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
- package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
- package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
- package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
- package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
- package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
- package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
- package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
- package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
- package/template_project/.codex/README.md +8 -0
- package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
- package/template_project/ukit/README.md +1 -1
- package/template_project/ukit/storage/config.json +20 -0
- package/template_user/playbooks/architecture-decision.md +28 -0
- package/template_user/playbooks/autonomous-run.md +43 -0
- package/template_user/playbooks/autopilot-full.md +59 -0
- package/template_user/playbooks/autopilot-stack.md +54 -0
- package/template_user/playbooks/babysit.md +39 -0
- package/template_user/playbooks/bug-fix.md +3 -1
- package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
- package/template_user/playbooks/hillclimb.md +44 -0
- package/template_user/playbooks/investigation.md +21 -0
- package/template_user/playbooks/migration.md +21 -0
- package/template_user/playbooks/open-pr.md +48 -0
- package/template_user/playbooks/orchestrate.md +45 -0
- package/template_user/playbooks/performance.md +33 -0
- package/template_user/playbooks/prototype.md +28 -0
- package/template_user/playbooks/refactor.md +19 -0
- package/template_user/playbooks/release.md +28 -0
- package/template_user/playbooks/runtime-forensics.md +23 -0
- package/template_user/playbooks/session-pickup.md +31 -0
- package/template_user/playbooks/shipping.md +53 -0
- package/template_user/playbooks/skill-evaluation.md +48 -0
- package/template_user/playbooks/small-feature.md +20 -0
- package/template_user/playbooks/verification-map.json +153 -0
- package/template_user/playbooks/verification.md +22 -0
- package/template_user/playbooks/worktree-cleanup.md +37 -0
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
// reviewPolicy.js — deterministic review decision policy evaluator (BL-015 /
|
|
2
|
+
// SPEC FR-003 / ARCH §Review Decision Policy).
|
|
3
|
+
//
|
|
4
|
+
// The evaluator answers "is one more verification round needed?" at checkpoints
|
|
5
|
+
// — playbook step boundaries and the pre-Stop gate (stop-coordinator.mjs) —
|
|
6
|
+
// from checkable signals only. It is the TRIGGER for the existing review lane
|
|
7
|
+
// (`review-verdict.mjs` + `review-panel-aggregate.mjs`, GAP M05 reuse), never a
|
|
8
|
+
// new reviewer and never a second gate: consumers record the verdict as an
|
|
9
|
+
// advisory while the completion gate stays the single Stop authority.
|
|
10
|
+
//
|
|
11
|
+
// Contract:
|
|
12
|
+
// evaluateReviewPolicy({signals, doneCriteria, evidenceClasses, fixLoopCount,
|
|
13
|
+
// round, cap, escalationDepth?, crossCheck?, projectRoot?,
|
|
14
|
+
// appendDecisionReceipt?})
|
|
15
|
+
// → {action, namedCheck?, signalsHit[], rationale, reviewRounds,
|
|
16
|
+
// verified[], notVerified[], missingEvidence[]}
|
|
17
|
+
//
|
|
18
|
+
// action ∈ REVIEW_POLICY_ACTIONS — exactly five:
|
|
19
|
+
// 'finish' stop; done predicates evidenced, no suspicion.
|
|
20
|
+
// 'run-targeted-check' name the command/surface + predicate to satisfy.
|
|
21
|
+
// 'independent-review' name what the reviewer must verify; the caller
|
|
22
|
+
// bundles {original request, diff, context files,
|
|
23
|
+
// run evidence} for the review-verdict lane — never
|
|
24
|
+
// the agent's summary alone (SPEC FR-004d).
|
|
25
|
+
// 'escalate' name the deeper lane/question (unic-decision
|
|
26
|
+
// laneDeepening/verificationDepth family) or the
|
|
27
|
+
// human authorization boundary. Terminal hand-off:
|
|
28
|
+
// it is NOT another round, so the cap never traps it.
|
|
29
|
+
// 'inconclusive' honest stop: report verified-vs-not, name the
|
|
30
|
+
// missing evidence. Round cap + timeout land here —
|
|
31
|
+
// never an infinite block.
|
|
32
|
+
//
|
|
33
|
+
// Safety boundary (ARCH §Safety boundary): no self-declared confidence exists
|
|
34
|
+
// in the input contract — keys like `confidence`/`statedConfidence` are ignored
|
|
35
|
+
// by design, so identical signal sets decide identically whether or not the
|
|
36
|
+
// agent narrates certainty.
|
|
37
|
+
//
|
|
38
|
+
// Calibration (ARCH §Calibration plan): when a caller supplies `projectRoot` +
|
|
39
|
+
// an `appendDecisionReceipt` implementation (execution-ledger.mjs signature,
|
|
40
|
+
// consumed unchanged), every evaluation emits one `kind:'review-policy'`
|
|
41
|
+
// receipt carrying action + namedCheck + signalsHit + round so decision↔outcome
|
|
42
|
+
// joins can measure unnecessary-review and escaped-defect rates.
|
|
43
|
+
//
|
|
44
|
+
// Parity: tests/consistency/reviewPolicyParity.test.js locks this module and
|
|
45
|
+
// the installed mirror to identical exports and identical verdicts.
|
|
46
|
+
|
|
47
|
+
import { resolveCrossCheckDepth } from './crossCheckMatrix.js';
|
|
48
|
+
|
|
49
|
+
// Exactly five actions — the closed enum consumers may switch on.
|
|
50
|
+
export const REVIEW_POLICY_ACTIONS = Object.freeze([
|
|
51
|
+
'finish',
|
|
52
|
+
'run-targeted-check',
|
|
53
|
+
'independent-review',
|
|
54
|
+
'escalate',
|
|
55
|
+
'inconclusive',
|
|
56
|
+
]);
|
|
57
|
+
|
|
58
|
+
// Action severity for multi-signal precedence and the escalationDepth bias.
|
|
59
|
+
// A higher-severity fired signal wins the action; the namedCheck travels with
|
|
60
|
+
// the winning signal.
|
|
61
|
+
const ACTION_SEVERITY = Object.freeze({
|
|
62
|
+
finish: 0,
|
|
63
|
+
'run-targeted-check': 1,
|
|
64
|
+
'independent-review': 2,
|
|
65
|
+
escalate: 3,
|
|
66
|
+
inconclusive: 0,
|
|
67
|
+
});
|
|
68
|
+
|
|
69
|
+
// Suspicion signal → action the policy fires. 'targeted' resolves to
|
|
70
|
+
// 'run-targeted-check', 'review' to 'independent-review'.
|
|
71
|
+
const SIGNAL_ACTION = Object.freeze({
|
|
72
|
+
targeted: 'run-targeted-check',
|
|
73
|
+
review: 'independent-review',
|
|
74
|
+
escalate: 'escalate',
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
// The 8-signal suspicion catalogue (ARCH §Suspicion-signal catalogue), kept as
|
|
78
|
+
// data so calibration + docs share one list. `check` names the exact check the
|
|
79
|
+
// action demands — command/surface + predicate, or what the reviewer verifies.
|
|
80
|
+
export const SUSPICION_SIGNAL_CATALOGUE = Object.freeze([
|
|
81
|
+
{
|
|
82
|
+
id: 'unchecked-request-details',
|
|
83
|
+
fires: 'targeted',
|
|
84
|
+
check: 'cross-check every request detail against the diff — each stated requirement must have a corresponding hunk or an explicit, evidenced skip',
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
id: 'wider-than-goal',
|
|
88
|
+
fires: 'review',
|
|
89
|
+
check: 'independent reviewer verifies each out-of-scope hunk is either required for the stated goal or reverted',
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
id: 'unwired-component',
|
|
93
|
+
fires: 'targeted',
|
|
94
|
+
check: 'verify the component is reachable in the usage flow — import/callsite present + render/import observation, not just a file on disk',
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
id: 'claimed-verification',
|
|
98
|
+
fires: 'targeted',
|
|
99
|
+
check: 're-run the claimed verification and require verbatim captured output — a claimed pass with no output is not evidence',
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
id: 'impl-detail-tests-only',
|
|
103
|
+
fires: 'targeted',
|
|
104
|
+
check: 'run one concrete behavioral input→output case per public entry point — tests asserting internals cannot catch user-visible failure',
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
id: 'tool-error-storm',
|
|
108
|
+
fires: 'review',
|
|
109
|
+
check: 'independent reviewer checks whether the unexplained tool errors mask a real (often environment-dependent) defect',
|
|
110
|
+
},
|
|
111
|
+
{
|
|
112
|
+
id: 'fix-loop-same-fingerprint',
|
|
113
|
+
fires: 'escalate',
|
|
114
|
+
check: 'fire the unic-decision laneDeepening/verificationDepth question → deeper debug lane (runtime-forensics for a live symptom, systematic-debugging depth otherwise); never another same-lane retry',
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
id: 'unconfirmed-assumptions',
|
|
118
|
+
fires: 'review',
|
|
119
|
+
check: 'independent reviewer verifies the unconfirmed assumptions the run carried to done — each must be confirmed or named as a stated limit',
|
|
120
|
+
},
|
|
121
|
+
]);
|
|
122
|
+
|
|
123
|
+
// Catalogue ids callers may assert through `signals` (object-map or array).
|
|
124
|
+
// Derived signals (`claimed-verification`, `fix-loop-same-fingerprint`) share
|
|
125
|
+
// the same catalogue — the caller never has to restate what the inputs already
|
|
126
|
+
// prove.
|
|
127
|
+
const SIGNAL_ID_SET = new Set(SUSPICION_SIGNAL_CATALOGUE.map((s) => s.id));
|
|
128
|
+
|
|
129
|
+
// Round-cap default: one initial check + one re-check round is the ARCH
|
|
130
|
+
// "one round default-sufficient for small tasks" bound. Every check-driven
|
|
131
|
+
// loop must land `inconclusive` at the cap, never retry forever.
|
|
132
|
+
export const DEFAULT_REVIEW_ROUND_CAP = 2;
|
|
133
|
+
|
|
134
|
+
// Same-fingerprint fix-loop threshold — aligned with the route layer's
|
|
135
|
+
// debugLoopThreshold default (runtimeConfig.js:379). fixLoopCount ≥ threshold
|
|
136
|
+
// derives suspicion signal #7 and routes to `escalate`.
|
|
137
|
+
export const DEFAULT_FIX_LOOP_THRESHOLD = 2;
|
|
138
|
+
|
|
139
|
+
// escalationDepth (forward-declared slot for TASK-C85-016's depth matrix):
|
|
140
|
+
// {depth, reasons[]?} or a bare depth string. The depth matrix may only RAISE
|
|
141
|
+
// the action floor — it never suppresses a fired suspicion signal.
|
|
142
|
+
const DEPTH_MIN_ACTION = Object.freeze({
|
|
143
|
+
none: 'finish',
|
|
144
|
+
runnable: 'run-targeted-check',
|
|
145
|
+
'review-round': 'independent-review',
|
|
146
|
+
escalate: 'escalate',
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
function isPlainObject(value) {
|
|
150
|
+
return Boolean(value) && typeof value === 'object' && !Array.isArray(value);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
function clampInt(value, fallback, min = 0, max = 99) {
|
|
154
|
+
const n = Number(value);
|
|
155
|
+
return Number.isFinite(n) ? Math.max(min, Math.min(max, Math.trunc(n))) : fallback;
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
// doneCriteria entry → {name, status}. Statuses:
|
|
159
|
+
// 'verified' — a real-run/surface receipt attests the predicate.
|
|
160
|
+
// 'claimed' — asserted with no captured output (suspicion signal #4).
|
|
161
|
+
// 'missing' — no evidence at all.
|
|
162
|
+
// Bare strings and `{satisfied:true}` shorthands normalize so callers can pass
|
|
163
|
+
// the completion contract's bare class tokens when they have nothing richer.
|
|
164
|
+
function normalizeDoneCriterion(entry) {
|
|
165
|
+
if (typeof entry === 'string' && entry.trim()) {
|
|
166
|
+
return { name: entry.trim(), status: 'missing' };
|
|
167
|
+
}
|
|
168
|
+
if (!isPlainObject(entry)) return null;
|
|
169
|
+
const name = typeof entry.name === 'string' && entry.name.trim()
|
|
170
|
+
? entry.name.trim()
|
|
171
|
+
: (typeof entry.id === 'string' && entry.id.trim() ? entry.id.trim() : null);
|
|
172
|
+
if (!name) return null;
|
|
173
|
+
let status = typeof entry.status === 'string' ? entry.status.trim().toLowerCase() : null;
|
|
174
|
+
if (status !== 'verified' && status !== 'claimed') {
|
|
175
|
+
status = entry.satisfied === true || entry.verified === true ? 'verified' : 'missing';
|
|
176
|
+
}
|
|
177
|
+
return { name, status };
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
// signals may arrive as an object map ({'unwired-component': true, ...}) or an
|
|
181
|
+
// array of catalogue ids. Only catalogue keys with a truthy flag count — a
|
|
182
|
+
// caller cannot inject self-declared confidence or invent signals, and a
|
|
183
|
+
// `confidence: 'sure'` key is silently ignored by design.
|
|
184
|
+
function assertedSignalIds(signals) {
|
|
185
|
+
const ids = new Set();
|
|
186
|
+
if (Array.isArray(signals)) {
|
|
187
|
+
for (const id of signals) {
|
|
188
|
+
if (typeof id === 'string' && SIGNAL_ID_SET.has(id)) ids.add(id);
|
|
189
|
+
}
|
|
190
|
+
} else if (isPlainObject(signals)) {
|
|
191
|
+
for (const [id, flag] of Object.entries(signals)) {
|
|
192
|
+
if (flag && SIGNAL_ID_SET.has(id)) ids.add(id);
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
return ids;
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
function escalationDepthAction(escalationDepth) {
|
|
199
|
+
const depth = typeof escalationDepth === 'string'
|
|
200
|
+
? escalationDepth
|
|
201
|
+
: escalationDepth?.depth;
|
|
202
|
+
return typeof depth === 'string' && Object.hasOwn(DEPTH_MIN_ACTION, depth)
|
|
203
|
+
? DEPTH_MIN_ACTION[depth]
|
|
204
|
+
: 'finish';
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
// Resolve the route record's crossCheck sub-record (BL-016): a pinned `depth`
|
|
208
|
+
// field counts as an already-resolved value; otherwise the full sub-record
|
|
209
|
+
// (risk/implementerReliability/evidenceQuality/rigor/…) feeds the matrix.
|
|
210
|
+
// Returns null when no usable depth exists — never throws.
|
|
211
|
+
function resolveCrossCheckInput(crossCheck) {
|
|
212
|
+
if (crossCheck == null) return null;
|
|
213
|
+
if (typeof crossCheck === 'string') {
|
|
214
|
+
return Object.hasOwn(DEPTH_MIN_ACTION, crossCheck) ? crossCheck : null;
|
|
215
|
+
}
|
|
216
|
+
if (!isPlainObject(crossCheck)) return null;
|
|
217
|
+
if (typeof crossCheck.depth === 'string' && Object.hasOwn(DEPTH_MIN_ACTION, crossCheck.depth)) {
|
|
218
|
+
return crossCheck.depth;
|
|
219
|
+
}
|
|
220
|
+
try {
|
|
221
|
+
return resolveCrossCheckDepth(crossCheck);
|
|
222
|
+
} catch {
|
|
223
|
+
return null; // a malformed sub-record never blocks the evaluation
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
// ─── evaluator ──────────────────────────────────────────────────────────────
|
|
228
|
+
|
|
229
|
+
export function evaluateReviewPolicy({
|
|
230
|
+
signals = {},
|
|
231
|
+
doneCriteria = [],
|
|
232
|
+
evidenceClasses = [],
|
|
233
|
+
fixLoopCount = 0,
|
|
234
|
+
round = 0,
|
|
235
|
+
cap = DEFAULT_REVIEW_ROUND_CAP,
|
|
236
|
+
escalationDepth = null,
|
|
237
|
+
// TASK-C85-016 (BL-016): the route record's crossCheck sub-record resolves
|
|
238
|
+
// its depth through the matrix when the caller didn't pass an explicit
|
|
239
|
+
// escalationDepth. A pinned `depth` field on the sub-record wins as an
|
|
240
|
+
// already-resolved value; explicit escalationDepth wins over both.
|
|
241
|
+
crossCheck = null,
|
|
242
|
+
fixLoopThreshold = DEFAULT_FIX_LOOP_THRESHOLD,
|
|
243
|
+
projectRoot = null,
|
|
244
|
+
appendDecisionReceipt = null,
|
|
245
|
+
} = {}) {
|
|
246
|
+
const criteria = (Array.isArray(doneCriteria) ? doneCriteria : [])
|
|
247
|
+
.map(normalizeDoneCriterion)
|
|
248
|
+
.filter(Boolean);
|
|
249
|
+
const verified = criteria.filter((c) => c.status === 'verified').map((c) => c.name);
|
|
250
|
+
const notVerified = criteria.filter((c) => c.status !== 'verified').map((c) => c.name);
|
|
251
|
+
const claimed = criteria.filter((c) => c.status === 'claimed').map((c) => c.name);
|
|
252
|
+
const requiredEvidence = (Array.isArray(evidenceClasses) ? evidenceClasses : [])
|
|
253
|
+
.filter((e) => typeof e === 'string' && e.trim());
|
|
254
|
+
const cappedRound = clampInt(round, 0);
|
|
255
|
+
const cappedCap = clampInt(cap, DEFAULT_REVIEW_ROUND_CAP, 1);
|
|
256
|
+
|
|
257
|
+
// ── signal detection ────────────────────────────────────────────────────
|
|
258
|
+
// Caller-asserted signals + the two the inputs derive on their own: a
|
|
259
|
+
// 'claimed' criterion IS signal #4 (verification claimed, no output) and
|
|
260
|
+
// fixLoopCount ≥ threshold IS signal #7 (same-fingerprint fix loop).
|
|
261
|
+
const hitIds = assertedSignalIds(signals);
|
|
262
|
+
if (claimed.length > 0) hitIds.add('claimed-verification');
|
|
263
|
+
if (clampInt(fixLoopCount, 0) >= Math.max(1, clampInt(fixLoopThreshold, DEFAULT_FIX_LOOP_THRESHOLD))) {
|
|
264
|
+
hitIds.add('fix-loop-same-fingerprint');
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
const hits = SUSPICION_SIGNAL_CATALOGUE.filter((s) => hitIds.has(s.id));
|
|
268
|
+
|
|
269
|
+
// ── base action ─────────────────────────────────────────────────────────
|
|
270
|
+
let action = 'finish';
|
|
271
|
+
let namedCheck = null;
|
|
272
|
+
let rationale;
|
|
273
|
+
|
|
274
|
+
if (hits.length > 0) {
|
|
275
|
+
const winning = hits.reduce((best, signal) => (
|
|
276
|
+
ACTION_SEVERITY[SIGNAL_ACTION[signal.fires]] > ACTION_SEVERITY[SIGNAL_ACTION[best.fires]]
|
|
277
|
+
? signal
|
|
278
|
+
: best
|
|
279
|
+
));
|
|
280
|
+
action = SIGNAL_ACTION[winning.fires];
|
|
281
|
+
namedCheck = winning.check;
|
|
282
|
+
rationale = `suspicion signal${hits.length > 1 ? 's' : ''} fired: ${hits.map((s) => s.id).join(', ')} — strongest is ${winning.id} (${action})`;
|
|
283
|
+
} else if (notVerified.length > 0) {
|
|
284
|
+
// Unevidenced work with no named suspicion still cannot finish: demand one
|
|
285
|
+
// cheap targeted check per the cheapest defect to catch.
|
|
286
|
+
action = 'run-targeted-check';
|
|
287
|
+
namedCheck = `obtain the named evidence for: ${notVerified.join(', ')} — receipt or verbatim output, not a claim`;
|
|
288
|
+
rationale = `unverified done-criteria with no fired signal: ${notVerified.join(', ')}`;
|
|
289
|
+
} else if (criteria.length === 0 && requiredEvidence.length > 0) {
|
|
290
|
+
// No criteria but required evidence classes → confirm the classes landed.
|
|
291
|
+
action = 'run-targeted-check';
|
|
292
|
+
namedCheck = `confirm receipt classes present: ${requiredEvidence.join(', ')}`;
|
|
293
|
+
rationale = 'no done-criteria; required evidence classes unaccounted';
|
|
294
|
+
} else if (criteria.length === 0) {
|
|
295
|
+
// Nothing checkable exists at all — the honest report, not a free pass.
|
|
296
|
+
action = 'inconclusive';
|
|
297
|
+
rationale = 'no done-criteria and no evidence classes to verify — nothing checkable was provided';
|
|
298
|
+
} else {
|
|
299
|
+
// Every criterion verified — a verified criterion IS the evidence;
|
|
300
|
+
// declared evidence classes add no extra round on their own.
|
|
301
|
+
rationale = 'all done-criteria verified, no suspicion signals — review round suppressed';
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
// ── cross-check depth floor (escalationDepth) ───────────────────────────
|
|
305
|
+
// TASK-C85-016 wires the depth matrix here: a depth may raise the action
|
|
306
|
+
// floor (e.g. 'review-round' forces ≥ independent-review) but never lowers
|
|
307
|
+
// it — a fired suspicion signal always outranks a shallow depth.
|
|
308
|
+
const effectiveDepth = escalationDepth ?? resolveCrossCheckInput(crossCheck);
|
|
309
|
+
const depthFloor = escalationDepthAction(effectiveDepth);
|
|
310
|
+
if (ACTION_SEVERITY[depthFloor] > ACTION_SEVERITY[action]) {
|
|
311
|
+
const depth = typeof effectiveDepth === 'object' ? (effectiveDepth?.depth ?? effectiveDepth) : effectiveDepth;
|
|
312
|
+
action = depthFloor;
|
|
313
|
+
namedCheck = namedCheck
|
|
314
|
+
?? (action === 'independent-review'
|
|
315
|
+
? 'cross-check depth matrix requires one independent review round — reviewer verifies the request is satisfied, wired into flow, and free of convention breaks'
|
|
316
|
+
: (action === 'escalate'
|
|
317
|
+
? SUSPICION_SIGNAL_CATALOGUE.find((s) => s.id === 'fix-loop-same-fingerprint').check
|
|
318
|
+
: 'run the artifact-class runnable checks required by the depth matrix'));
|
|
319
|
+
rationale += `; escalationDepth=${depth} raised the action floor to ${action}`;
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
// ── round cap → inconclusive (never infinite blocking) ──────────────────
|
|
323
|
+
// 'escalate' bypasses the cap deliberately: it is a terminal hand-off to a
|
|
324
|
+
// deeper lane, not another loop iteration.
|
|
325
|
+
if (action !== 'finish' && action !== 'escalate' && cappedRound >= cappedCap) {
|
|
326
|
+
const previous = action;
|
|
327
|
+
action = 'inconclusive';
|
|
328
|
+
namedCheck = null;
|
|
329
|
+
rationale = `review round cap reached (round=${cappedRound} ≥ cap=${cappedCap}) — ${previous} could not be confirmed; reporting verified-vs-not honestly instead of looping`;
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
const missingEvidence = [...new Set([...notVerified, ...requiredEvidence])];
|
|
333
|
+
|
|
334
|
+
const decision = {
|
|
335
|
+
action,
|
|
336
|
+
namedCheck,
|
|
337
|
+
signalsHit: hits.map((s) => s.id),
|
|
338
|
+
rationale,
|
|
339
|
+
reviewRounds: cappedRound + (action === 'finish' || action === 'inconclusive' ? 0 : 1),
|
|
340
|
+
verified,
|
|
341
|
+
notVerified,
|
|
342
|
+
missingEvidence,
|
|
343
|
+
};
|
|
344
|
+
|
|
345
|
+
// ── calibration receipt (consumed signature — never edited) ─────────────
|
|
346
|
+
// decisionKeys carry the receipt fields the ledger schema doesn't model:
|
|
347
|
+
// signalsHit, round, and outcome (unknown at decision time → pending).
|
|
348
|
+
if (typeof appendDecisionReceipt === 'function' && projectRoot) {
|
|
349
|
+
try {
|
|
350
|
+
const appended = appendDecisionReceipt(projectRoot, {
|
|
351
|
+
kind: 'review-policy',
|
|
352
|
+
boundary: 'review',
|
|
353
|
+
stage: 'deterministic',
|
|
354
|
+
outcomeClass: decision.action,
|
|
355
|
+
checkpoint: decision.namedCheck,
|
|
356
|
+
decisionKeys: [
|
|
357
|
+
`signalsHit=${decision.signalsHit.join(',') || 'none'}`,
|
|
358
|
+
`round=${cappedRound}`,
|
|
359
|
+
'outcome=pending',
|
|
360
|
+
],
|
|
361
|
+
agreement: 'n/a',
|
|
362
|
+
});
|
|
363
|
+
if (appended?.catch) appended.catch(() => {});
|
|
364
|
+
} catch { /* calibration telemetry never fails the evaluation */ }
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
return decision;
|
|
368
|
+
}
|