peaks-loop 4.0.42 → 4.0.44
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +59 -0
- package/README-en.md +1 -1
- package/README.md +1 -1
- package/dist/cli/commands/_register.js +4 -0
- package/dist/cli/commands/api-diff-commands.d.ts +16 -0
- package/dist/cli/commands/api-diff-commands.js +55 -0
- package/dist/cli/commands/audit-commands.d.ts +16 -3
- package/dist/cli/commands/audit-commands.js +84 -31
- package/dist/cli/commands/codegraph-commands.js +191 -6
- package/dist/cli/commands/final-review-commands.d.ts +34 -10
- package/dist/cli/commands/final-review-commands.js +130 -34
- package/dist/cli/commands/job-commands.js +4 -2
- package/dist/cli/commands/scan-commands.js +1 -1
- package/dist/cli/commands/share-commands.d.ts +49 -0
- package/dist/cli/commands/share-commands.js +114 -14
- package/dist/cli/commands/test-commands.d.ts +60 -3
- package/dist/cli/commands/test-commands.js +125 -7
- package/dist/services/audit/audit-goal-service.js +38 -3
- package/dist/services/codegraph/codegraph-autorefresh.js +12 -0
- package/dist/services/codegraph/codegraph-exclude-integrity.d.ts +61 -0
- package/dist/services/codegraph/codegraph-exclude-integrity.js +98 -0
- package/dist/services/codegraph/codegraph-exclude-reconciler.d.ts +26 -0
- package/dist/services/codegraph/codegraph-exclude-reconciler.js +217 -0
- package/dist/services/codegraph/codegraph-exclude-repair.d.ts +102 -0
- package/dist/services/codegraph/codegraph-exclude-repair.js +266 -0
- package/dist/services/codegraph/codegraph-preflight-service.js +12 -0
- package/dist/services/codegraph/codegraph-service.d.ts +0 -1
- package/dist/services/codegraph/codegraph-service.js +5 -4
- package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.d.ts +29 -0
- package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.js +88 -0
- package/dist/services/doctor/doctor-service/checks/ecc-hooks-schema-drift.d.ts +65 -0
- package/dist/services/doctor/doctor-service/checks/ecc-hooks-schema-drift.js +186 -0
- package/dist/services/doctor/doctor-service/plugin-registry.js +4 -0
- package/dist/services/doctor/doctor-service/types.d.ts +47 -0
- package/dist/services/final-review/final-review-service.d.ts +154 -0
- package/dist/services/final-review/final-review-service.js +621 -7
- package/dist/services/final-review/index.d.ts +1 -1
- package/dist/services/final-review/index.js +1 -1
- package/dist/services/llm/anthropic-runner.d.ts +87 -0
- package/dist/services/llm/anthropic-runner.js +171 -0
- package/dist/services/llm/stub-runner.d.ts +11 -0
- package/dist/services/llm/stub-runner.js +33 -0
- package/dist/services/prd/handoff-auto-regen.js +0 -1
- package/dist/services/prd/handoff-service.d.ts +9 -1
- package/dist/services/prd/handoff-service.js +48 -6
- package/dist/services/prd/project-scan-bootstrap-service.js +7 -7
- package/dist/services/scan/api-diff-openapi.d.ts +32 -0
- package/dist/services/scan/api-diff-openapi.js +359 -0
- package/dist/services/scan/api-diff-recorded.d.ts +96 -0
- package/dist/services/scan/api-diff-recorded.js +577 -0
- package/dist/services/scan/api-diff-service.d.ts +34 -0
- package/dist/services/scan/api-diff-service.js +407 -0
- package/dist/services/scan/api-diff-types.d.ts +116 -0
- package/dist/services/scan/api-diff-types.js +46 -0
- package/dist/services/scan/archetype-service.js +27 -1
- package/dist/services/scan/existing-system-service.js +17 -4
- package/dist/services/scan/hook-convention-service.d.ts +26 -0
- package/dist/services/scan/hook-convention-service.js +562 -0
- package/dist/services/scan/scan-types.d.ts +47 -0
- package/dist/services/session/caller-binding-service.d.ts +28 -0
- package/dist/services/session/caller-binding-service.js +10 -2
- package/dist/services/session/caller-id-types.d.ts +12 -2
- package/dist/services/session/index.d.ts +2 -2
- package/dist/services/session/index.js +2 -2
- package/dist/services/session/session-binding-bridge.js +11 -6
- package/dist/services/session/session-manager.d.ts +33 -1
- package/dist/services/session/session-manager.js +84 -25
- package/dist/services/skills/skill-presence-service.d.ts +17 -3
- package/dist/services/skills/skill-presence-service.js +23 -3
- package/package.json +7 -5
- package/skills/bee/peaks-rd/SKILL.md +11 -3
- package/skills/peaks-code/references/existing-system-extraction.md +5 -1
- package/skills/peaks-code/references/frontend-only-mode.md +48 -6
- package/skills/peaks-code/references/project-scan-checklist.md +20 -1
- package/skills/peaks-doctor/references/doctor-check-catalog.md +1 -0
- package/skills/peaks-final-review/SKILL.md +43 -32
|
@@ -23,6 +23,579 @@ export class IncompleteFinalReviewError extends Error {
|
|
|
23
23
|
this.name = 'IncompleteFinalReviewError';
|
|
24
24
|
}
|
|
25
25
|
}
|
|
26
|
+
/**
|
|
27
|
+
* N4 — the reply carried no text block at all.
|
|
28
|
+
*
|
|
29
|
+
* Measured 2/3 on this repo's own machine, and it is NOT truncation: the
|
|
30
|
+
* provider answered with a response whose `content` has no `text` block (a
|
|
31
|
+
* reasoning-only turn, a refusal, or a content filter), so there is no JSON to
|
|
32
|
+
* parse and no budget to raise — an operator sent to "raise the budget" for
|
|
33
|
+
* this failure would be sent the wrong way. It gets its own class, its own
|
|
34
|
+
* `code`, and a message that says so, so it is diagnosable instead of being
|
|
35
|
+
* flattened into "not valid JSON".
|
|
36
|
+
*/
|
|
37
|
+
export class EmptyReviewReplyError extends Error {
|
|
38
|
+
code = 'EMPTY_FINAL_REVIEW_REPLY';
|
|
39
|
+
constructor(message) {
|
|
40
|
+
super(message);
|
|
41
|
+
this.name = 'EmptyReviewReplyError';
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* How many times an empty reply is retried before it is reported. The failure
|
|
46
|
+
* was 2/3 on the observed machine — intermittent, not systematic — so a small
|
|
47
|
+
* bounded retry converts most of it into a completed review, while 3 attempts
|
|
48
|
+
* keeps a genuinely broken provider from being hammered.
|
|
49
|
+
*/
|
|
50
|
+
export const MAX_EMPTY_REPLY_ATTEMPTS = 3;
|
|
51
|
+
/** The runner's own wording for "the response had no text block to return". */
|
|
52
|
+
function isEmptyReplyError(error) {
|
|
53
|
+
return error instanceof Error && /no text block/i.test(error.message);
|
|
54
|
+
}
|
|
55
|
+
function errorText(error) {
|
|
56
|
+
return error instanceof Error ? error.message : String(error);
|
|
57
|
+
}
|
|
58
|
+
/* ------------------------------------------------------------------ *
|
|
59
|
+
* D1 — on-disk evidence collection.
|
|
60
|
+
*
|
|
61
|
+
* The reviewer LLM has NO tools and NO filesystem access: whatever the
|
|
62
|
+
* service does not inline into the prompt does not exist for it. Feeding it
|
|
63
|
+
* only the success criteria forced a guess — the observed failure mode was a
|
|
64
|
+
* 4/4 `inconclusive` verdict, and the dangerous one is an invented `pass`
|
|
65
|
+
* that SKILL.md would read as a clean handoff. Everything below exists so the
|
|
66
|
+
* verdicts rest on what is actually on disk, and so a missing artifact is
|
|
67
|
+
* reported as MISSING instead of being silently skipped.
|
|
68
|
+
* ------------------------------------------------------------------ */
|
|
69
|
+
/**
|
|
70
|
+
* Per-file evidence cap. Real evidence artifacts in this repo run 11–18 KB
|
|
71
|
+
* (`rd/tech-doc.md`, `rd/code-review.md`, `qa/*-findings-*.md`); 8 KB keeps the
|
|
72
|
+
* head of every file (header + verdict + first tables) without letting one
|
|
73
|
+
* verbose artifact crowd out the other sources. Enforced in BYTES against the
|
|
74
|
+
* raw buffer, so multi-byte (CJK) content cannot slip past the cap.
|
|
75
|
+
*/
|
|
76
|
+
export const MAX_EVIDENCE_BYTES_PER_FILE = 8 * 1024;
|
|
77
|
+
/**
|
|
78
|
+
* Total evidence budget across all sources. 32 KB ≈ 8k tokens of input, which
|
|
79
|
+
* keeps the prompt far inside any modern context window. Sources that do not
|
|
80
|
+
* fit are reported as OMITTED — never dropped silently.
|
|
81
|
+
*
|
|
82
|
+
* This is an INPUT cap and stays fixed. The output ceiling that has to sit
|
|
83
|
+
* opposite it is derived per call by `outputBudgetForEvidence()` below — the
|
|
84
|
+
* two used to drift apart, and that drift was the defect.
|
|
85
|
+
*/
|
|
86
|
+
export const MAX_EVIDENCE_BYTES_TOTAL = 32 * 1024;
|
|
87
|
+
/**
|
|
88
|
+
* Reserved floor per dimension — the anti-starvation guarantee.
|
|
89
|
+
*
|
|
90
|
+
* The allocator below used to be strictly first-come-first-served: each source
|
|
91
|
+
* took `min(perFileCap, budgetLeft)` in source order. With this repo's own
|
|
92
|
+
* 9-source evidence set (`2026-09-12-session-e37ef0`, measured) sources 1-4
|
|
93
|
+
* consumed the whole 32 KiB — 4 x 8,192 = 32,768, the cap to the byte — before
|
|
94
|
+
* source 5 was even opened. `existing-functionality-intact` is supplied ONLY by
|
|
95
|
+
* `rd/tech-doc.md` (6th) and `prd/handoff.md` (9th), so that one dimension
|
|
96
|
+
* reached the reviewer with zero evidence on every run and its verdict was
|
|
97
|
+
* structurally locked to `inconclusive` no matter how good the work was. A gate
|
|
98
|
+
* that is always red is noise, and an operator trained to ignore noise has no
|
|
99
|
+
* gate at all — the same harm as a gate that never fires, only quieter.
|
|
100
|
+
*
|
|
101
|
+
* So each dimension with at least one readable source on disk gets one floor
|
|
102
|
+
* reserved for the FIRST such source, and no source that is not that holder may
|
|
103
|
+
* spend it. The reservation is a floor, never a quota: it is released the
|
|
104
|
+
* instant its holder is served, and whatever the holder does not use flows back
|
|
105
|
+
* into the sequential allocation unchanged.
|
|
106
|
+
*
|
|
107
|
+
* Why 4 KiB: the per-file cap exists to keep the "header + verdict + first
|
|
108
|
+
* tables" — the part a reviewer actually cites. Measured on the same run's
|
|
109
|
+
* artifacts (9 files, 8,164-20,543 bytes each): every one of them states its
|
|
110
|
+
* verdict inside the first ~700 bytes. 4 KiB is ~5x that, so a floor holder is
|
|
111
|
+
* not there for depth — it is there so its dimension is not blind. Four
|
|
112
|
+
* dimensions x 4 KiB = 16 KiB of the 32 KiB cap, so at least half the budget
|
|
113
|
+
* still flows through the sequential path below.
|
|
114
|
+
*/
|
|
115
|
+
export const MIN_EVIDENCE_BYTES_PER_DIMENSION = 4 * 1024;
|
|
116
|
+
/* ------------------------------------------------------------------ *
|
|
117
|
+
* D1 layer 3 — the OUTPUT budget.
|
|
118
|
+
*
|
|
119
|
+
* The input side above was raised to 32 KiB of inlined evidence, but the
|
|
120
|
+
* output ceiling stayed hard-coded at 3000 tokens. Measured on this repo's own
|
|
121
|
+
* run (rid `2026-09-12-codegraph-exclude-integrity`, 9 sources, 32 KiB
|
|
122
|
+
* inlined, real anthropic provider): 3/3 attempts failed — 2x
|
|
123
|
+
* INCOMPLETE_FINAL_REVIEW with the JSON cut off mid-string, 1x a reply with no
|
|
124
|
+
* text block. A 4-dimension envelope (4 x `summary` + `evidence[]` +
|
|
125
|
+
* `confidence`, plus `overallSummary`) does not fit in 3000 tokens once the
|
|
126
|
+
* model has ~8k tokens of evidence it is required to cite.
|
|
127
|
+
*
|
|
128
|
+
* N4 — the first fix of this layer derived the ceiling as "3000 + bytes/8" and
|
|
129
|
+
* called 8192 "a backstop only: it does not bind today", citing a 4775-token
|
|
130
|
+
* measurement. Both claims were falsified by re-measurement on the SAME
|
|
131
|
+
* machine the gate ships on (`deepseek-flash[1M]` via
|
|
132
|
+
* `api.deepseek.com/anthropic`, 2026-09-12, QA run 3/3 red + orchestrator
|
|
133
|
+
* re-run 3/3 red, rid `2026-09-12-codegraph-exclude-integrity`, byte-identical
|
|
134
|
+
* prompt):
|
|
135
|
+
*
|
|
136
|
+
* max_tokens=7096 (what the old formula produced for the 32 KiB pack)
|
|
137
|
+
* -> TRUNCATED. `output_tokens=7096`, 14078 characters.
|
|
138
|
+
* max_tokens=8192 -> TRUNCATED, `output_tokens=8192`, 574 characters.
|
|
139
|
+
* max_tokens=16000 -> COMPLETE, `output_tokens=10108`.
|
|
140
|
+
* max_tokens=32000 -> COMPLETE, `output_tokens=8883`.
|
|
141
|
+
*
|
|
142
|
+
* So 8192 WAS the binding constraint and was BELOW the requirement: a gate
|
|
143
|
+
* whose budget is short is worse than a red gate, because it releases the
|
|
144
|
+
* envelope only when the model happens to be terse.
|
|
145
|
+
*
|
|
146
|
+
* The 8:1 bytes-per-token term prices the VISIBLE envelope against the evidence
|
|
147
|
+
* the model must cite, but it was never the whole cost. On a reasoning model
|
|
148
|
+
* `max_tokens` also caps the hidden reasoning that precedes the first character
|
|
149
|
+
* of output, and that cost is invisible to a bytes-per-token formula (the
|
|
150
|
+
* 574-character run above is exactly that: the whole budget consumed before the
|
|
151
|
+
* envelope began). `REASONING_HEADROOM_TOKENS` below is that missing term.
|
|
152
|
+
*
|
|
153
|
+
* A 16384 ceiling with 6144 of headroom (derived 13240) was then measured on
|
|
154
|
+
* the SAME machine and shown to be too thin too: 10 real-machine runs, 2
|
|
155
|
+
* failures, BOTH genuine truncation — and the second one reported BOTH signals
|
|
156
|
+
* at once ("reply ends mid-structure AND provider-reported output reached the
|
|
157
|
+
* ceiling") with `maxTokens=13240`. The requirement is therefore not "derived
|
|
158
|
+
* >= the one 10108 sample" but "derived comfortably above every observed
|
|
159
|
+
* truncation point", which is what the constants below are now sized for.
|
|
160
|
+
* ------------------------------------------------------------------ */
|
|
161
|
+
/**
|
|
162
|
+
* Floor — also the value that shipped before this fix, so no evidence set can
|
|
163
|
+
* end up with a smaller budget than it had. ~3000 tokens is enough for the
|
|
164
|
+
* envelope skeleton plus a short paragraph per dimension.
|
|
165
|
+
*/
|
|
166
|
+
export const MIN_OUTPUT_TOKENS = 3_000;
|
|
167
|
+
/**
|
|
168
|
+
* Headroom for the part of the reply that is not the envelope.
|
|
169
|
+
*
|
|
170
|
+
* A Messages-API-compatible endpoint applies `max_tokens` to the WHOLE
|
|
171
|
+
* response, and a reasoning model spends it on hidden reasoning before it
|
|
172
|
+
* emits a single character of the 4-dim envelope. Measured on this repo's own
|
|
173
|
+
* machine (2026-09-12, rid `2026-09-12-codegraph-exclude-integrity`,
|
|
174
|
+
* `deepseek-flash[1M]` via `api.deepseek.com/anthropic`): `max_tokens=8192`
|
|
175
|
+
* came back with `output_tokens=8192` and only **574** visible characters —
|
|
176
|
+
* the entire budget went to reasoning. A bytes-per-token estimate of the
|
|
177
|
+
* visible output cannot see that cost, which is why the previous formula
|
|
178
|
+
* budgeted 7096 for a reply that needs 10108 — and why a 13240 budget still
|
|
179
|
+
* truncated on 2 of 10 real runs.
|
|
180
|
+
*
|
|
181
|
+
* 12288 (12 KiB) is sized so the largest evidence pack the input caps allow
|
|
182
|
+
* lands at 23480 (see the formula below) — about 1.8x the largest value ever
|
|
183
|
+
* OBSERVED to truncate (13240), which is the margin the observed variance
|
|
184
|
+
* asks for. The numbers are in the block comment above.
|
|
185
|
+
*/
|
|
186
|
+
export const REASONING_HEADROOM_TOKENS = 12 * 1024;
|
|
187
|
+
/**
|
|
188
|
+
* Ceiling, 32_000: the value a real run on this machine was forced to in order
|
|
189
|
+
* to complete the envelope at all, and the largest this endpoint was observed
|
|
190
|
+
* to accept. 16384 was tried first and truncated 2/10 — a ceiling that is
|
|
191
|
+
* merely "above the last successful measurement" is not above the requirement,
|
|
192
|
+
* because the requirement moves with the model's reasoning spend.
|
|
193
|
+
*
|
|
194
|
+
* A model that caps output at 8192 will refuse this. That is still strictly
|
|
195
|
+
* better than shipping a budget measured to be too small, and the env lever
|
|
196
|
+
* below lets an operator pull it down without a code change.
|
|
197
|
+
*/
|
|
198
|
+
export const MAX_OUTPUT_TOKENS = 32_000;
|
|
199
|
+
/**
|
|
200
|
+
* Environment lever. The old failure message told the operator to "raise the
|
|
201
|
+
* budget" while the budget was a module constant with no CLI flag and no env
|
|
202
|
+
* var — an instruction that could not be carried out from any surface the
|
|
203
|
+
* operator has. This is that lever.
|
|
204
|
+
*
|
|
205
|
+
* The value is the output ceiling in tokens; it OVERRIDES the derivation below
|
|
206
|
+
* (it is not a bonus added to it). Unset/invalid/out-of-range handling is in
|
|
207
|
+
* `resolveOutputBudget`.
|
|
208
|
+
*/
|
|
209
|
+
export const MAX_OUTPUT_TOKENS_ENV = 'PEAKS_FINAL_REVIEW_MAX_OUTPUT_TOKENS';
|
|
210
|
+
/**
|
|
211
|
+
* Absolute upper bound the env lever may reach. An endpoint that accepts
|
|
212
|
+
* `max_tokens` at all accepts this; anything above it is a typo (a stray extra
|
|
213
|
+
* digit), not an intent, and clamping is safer than sending it.
|
|
214
|
+
*/
|
|
215
|
+
export const HARD_MAX_OUTPUT_TOKENS = 64_000;
|
|
216
|
+
/**
|
|
217
|
+
* Inlined bytes that buy one extra output token — 4:1.
|
|
218
|
+
*
|
|
219
|
+
* This term prices the visible envelope (4 x `summary` + `evidence[]` +
|
|
220
|
+
* `confidence` + `overallSummary`) against the evidence the model is required
|
|
221
|
+
* to cite. It was 8:1, which put the 32 KiB pack at 4096 tokens of visible
|
|
222
|
+
* output; the same pack has been observed to complete at 10108 and to truncate
|
|
223
|
+
* at 13240, so 8:1 was pricing the visible side BELOW its own measurement.
|
|
224
|
+
* 4:1 doubles it to 8192. It is still not treated as the whole budget — see
|
|
225
|
+
* `REASONING_HEADROOM_TOKENS`.
|
|
226
|
+
*/
|
|
227
|
+
export const EVIDENCE_BYTES_PER_OUTPUT_TOKEN = 4;
|
|
228
|
+
/**
|
|
229
|
+
* Output ceiling for a call whose prompt carries `includedEvidenceBytes` bytes
|
|
230
|
+
* of inlined evidence. Pure, total, and clamped on both ends — the same
|
|
231
|
+
* evidence pack always yields the same budget.
|
|
232
|
+
*/
|
|
233
|
+
export function outputBudgetForEvidence(includedEvidenceBytes) {
|
|
234
|
+
const scaled = MIN_OUTPUT_TOKENS +
|
|
235
|
+
REASONING_HEADROOM_TOKENS +
|
|
236
|
+
Math.ceil(Math.max(0, includedEvidenceBytes) / EVIDENCE_BYTES_PER_OUTPUT_TOKEN);
|
|
237
|
+
return Math.min(MAX_OUTPUT_TOKENS, Math.max(MIN_OUTPUT_TOKENS, scaled));
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* The budget the call actually uses: the derived one, unless
|
|
241
|
+
* `PEAKS_FINAL_REVIEW_MAX_OUTPUT_TOKENS` overrides it.
|
|
242
|
+
*
|
|
243
|
+
* An override that is not a positive integer THROWS rather than being ignored:
|
|
244
|
+
* a silent fallback would leave an operator who passed a bad value with the
|
|
245
|
+
* exact experience this lever exists to remove — a budget they cannot move.
|
|
246
|
+
* Out-of-range values are clamped, not rejected, so a model needing more than
|
|
247
|
+
* `HARD_MAX_OUTPUT_TOKENS` (or a model needing less than `MIN_OUTPUT_TOKENS`)
|
|
248
|
+
* still gets a call made.
|
|
249
|
+
*/
|
|
250
|
+
export function resolveOutputBudget(includedEvidenceBytes, env = process.env) {
|
|
251
|
+
const derived = outputBudgetForEvidence(includedEvidenceBytes);
|
|
252
|
+
const raw = env[MAX_OUTPUT_TOKENS_ENV];
|
|
253
|
+
if (raw === undefined || raw.trim() === '')
|
|
254
|
+
return derived;
|
|
255
|
+
const parsed = Number.parseInt(raw.trim(), 10);
|
|
256
|
+
if (!Number.isInteger(parsed) || parsed <= 0 || String(parsed) !== raw.trim()) {
|
|
257
|
+
throw new Error(`${MAX_OUTPUT_TOKENS_ENV} must be a positive integer number of output tokens (got "${raw}"); ` +
|
|
258
|
+
`unset it to use the derived budget of ${String(derived)} tokens.`);
|
|
259
|
+
}
|
|
260
|
+
return Math.min(HARD_MAX_OUTPUT_TOKENS, Math.max(MIN_OUTPUT_TOKENS, parsed));
|
|
261
|
+
}
|
|
262
|
+
function evidenceSourcesFor(rid) {
|
|
263
|
+
return [
|
|
264
|
+
{
|
|
265
|
+
key: 'qa-test-report',
|
|
266
|
+
label: 'QA execution report (per-command pass/fail counts)',
|
|
267
|
+
segments: ['qa', 'test-reports', `${rid}.md`],
|
|
268
|
+
supports: ['functional-completeness', 'problem-resolution', 'no-new-bugs']
|
|
269
|
+
},
|
|
270
|
+
{
|
|
271
|
+
key: 'qa-test-cases',
|
|
272
|
+
label: 'QA test cases (acceptance-criterion to test mapping)',
|
|
273
|
+
segments: ['qa', 'test-cases', `${rid}.md`],
|
|
274
|
+
supports: ['functional-completeness', 'problem-resolution']
|
|
275
|
+
},
|
|
276
|
+
{
|
|
277
|
+
key: 'qa-security-findings',
|
|
278
|
+
label: 'QA security findings',
|
|
279
|
+
segments: ['qa', `security-findings-${rid}.md`],
|
|
280
|
+
supports: ['no-new-bugs']
|
|
281
|
+
},
|
|
282
|
+
{
|
|
283
|
+
key: 'qa-performance-findings',
|
|
284
|
+
label: 'QA performance findings',
|
|
285
|
+
segments: ['qa', `performance-findings-${rid}.md`],
|
|
286
|
+
supports: ['no-new-bugs']
|
|
287
|
+
},
|
|
288
|
+
{
|
|
289
|
+
key: 'rd-code-review',
|
|
290
|
+
label: 'RD code review',
|
|
291
|
+
segments: ['rd', 'code-review.md'],
|
|
292
|
+
supports: ['no-new-bugs']
|
|
293
|
+
},
|
|
294
|
+
{
|
|
295
|
+
key: 'rd-security-review',
|
|
296
|
+
label: 'RD security review',
|
|
297
|
+
segments: ['rd', 'security-review.md'],
|
|
298
|
+
supports: ['no-new-bugs']
|
|
299
|
+
},
|
|
300
|
+
{
|
|
301
|
+
key: 'rd-tech-doc',
|
|
302
|
+
label: 'RD tech doc (public surface / design intent)',
|
|
303
|
+
segments: ['rd', 'tech-doc.md'],
|
|
304
|
+
supports: ['existing-functionality-intact']
|
|
305
|
+
},
|
|
306
|
+
{
|
|
307
|
+
key: 'rd-bug-analysis',
|
|
308
|
+
label: 'RD bug analysis (original problem statement)',
|
|
309
|
+
segments: ['rd', 'bug-analysis.md'],
|
|
310
|
+
supports: ['problem-resolution']
|
|
311
|
+
},
|
|
312
|
+
{
|
|
313
|
+
key: 'prd-handoff',
|
|
314
|
+
label: 'PRD handoff (approved scope + non-goals)',
|
|
315
|
+
segments: ['prd', 'handoff.md'],
|
|
316
|
+
supports: ['functional-completeness', 'existing-functionality-intact']
|
|
317
|
+
}
|
|
318
|
+
];
|
|
319
|
+
}
|
|
320
|
+
/** Pure read: no commands, no git, no test execution — files only. */
|
|
321
|
+
function readEvidence(projectRoot, sessionId, rid) {
|
|
322
|
+
const runtimeRoot = join(projectRoot, '.peaks', '_runtime', sessionId);
|
|
323
|
+
return evidenceSourcesFor(rid).map(source => {
|
|
324
|
+
const relativePath = ['.peaks', '_runtime', sessionId, ...source.segments].join('/');
|
|
325
|
+
const absolutePath = join(runtimeRoot, ...source.segments);
|
|
326
|
+
let raw = null;
|
|
327
|
+
let error = '';
|
|
328
|
+
try {
|
|
329
|
+
raw = readFileSync(absolutePath);
|
|
330
|
+
}
|
|
331
|
+
catch (err) {
|
|
332
|
+
error = err instanceof Error ? err.message : String(err);
|
|
333
|
+
}
|
|
334
|
+
return {
|
|
335
|
+
source,
|
|
336
|
+
relativePath,
|
|
337
|
+
absolutePath,
|
|
338
|
+
raw,
|
|
339
|
+
error,
|
|
340
|
+
blank: raw === null || raw.toString('utf8').trim().length === 0
|
|
341
|
+
};
|
|
342
|
+
});
|
|
343
|
+
}
|
|
344
|
+
/**
|
|
345
|
+
* The one source per dimension that the budget promises to reach: the first
|
|
346
|
+
* non-blank candidate that can supply that dimension. A dimension with several
|
|
347
|
+
* candidates needs exactly one of them to survive, and the earliest is the one
|
|
348
|
+
* the sequential order would have reached anyway — so naming it costs the other
|
|
349
|
+
* sources nothing they were not already losing.
|
|
350
|
+
*
|
|
351
|
+
* A blank source is deliberately not a holder — see `MIN_EVIDENCE_BYTES_PER_DIMENSION`.
|
|
352
|
+
*/
|
|
353
|
+
function floorHolders(candidates) {
|
|
354
|
+
const holders = new Map();
|
|
355
|
+
const covered = new Set();
|
|
356
|
+
candidates.forEach((candidate, index) => {
|
|
357
|
+
if (candidate.blank)
|
|
358
|
+
return;
|
|
359
|
+
for (const dimension of candidate.source.supports) {
|
|
360
|
+
if (covered.has(dimension))
|
|
361
|
+
continue;
|
|
362
|
+
covered.add(dimension);
|
|
363
|
+
if (!holders.has(index))
|
|
364
|
+
holders.set(index, dimension);
|
|
365
|
+
}
|
|
366
|
+
});
|
|
367
|
+
return holders;
|
|
368
|
+
}
|
|
369
|
+
function collectEvidence(projectRoot, sessionId, rid) {
|
|
370
|
+
const candidates = readEvidence(projectRoot, sessionId, rid);
|
|
371
|
+
const holders = floorHolders(candidates);
|
|
372
|
+
/** Holders that have not been served yet — the floors still owed. */
|
|
373
|
+
const pending = new Set(holders.keys());
|
|
374
|
+
const collected = [];
|
|
375
|
+
let budgetLeft = MAX_EVIDENCE_BYTES_TOTAL;
|
|
376
|
+
for (const [index, candidate] of candidates.entries()) {
|
|
377
|
+
const { source, relativePath, absolutePath, raw } = candidate;
|
|
378
|
+
const base = { source, relativePath, absolutePath };
|
|
379
|
+
if (raw === null) {
|
|
380
|
+
collected.push({
|
|
381
|
+
...base,
|
|
382
|
+
status: 'missing',
|
|
383
|
+
totalBytes: 0,
|
|
384
|
+
includedBytes: 0,
|
|
385
|
+
content: '',
|
|
386
|
+
reason: candidate.error
|
|
387
|
+
});
|
|
388
|
+
continue;
|
|
389
|
+
}
|
|
390
|
+
if (candidate.blank) {
|
|
391
|
+
collected.push({
|
|
392
|
+
...base,
|
|
393
|
+
status: 'empty',
|
|
394
|
+
totalBytes: raw.byteLength,
|
|
395
|
+
includedBytes: 0,
|
|
396
|
+
content: '',
|
|
397
|
+
reason: `file exists but contains no reviewable content (${raw.byteLength} bytes)`
|
|
398
|
+
});
|
|
399
|
+
continue;
|
|
400
|
+
}
|
|
401
|
+
// The floor owed to every dimension still waiting on its holder is spent
|
|
402
|
+
// only on that holder. This is the whole fix: a source that needs no help
|
|
403
|
+
// can no longer eat the last dimension's only chance at being reviewed.
|
|
404
|
+
const holdsFloor = pending.has(index);
|
|
405
|
+
const reservedElsewhere = (pending.size - (holdsFloor ? 1 : 0)) * MIN_EVIDENCE_BYTES_PER_DIMENSION;
|
|
406
|
+
const allowance = budgetLeft - reservedElsewhere;
|
|
407
|
+
if (allowance <= 0) {
|
|
408
|
+
const waiting = [...pending]
|
|
409
|
+
.filter(holder => holder !== index)
|
|
410
|
+
.map(holder => `${holders.get(holder)} (source ${candidates[holder]?.source.key ?? '?'})`);
|
|
411
|
+
collected.push({
|
|
412
|
+
...base,
|
|
413
|
+
status: 'omitted',
|
|
414
|
+
totalBytes: raw.byteLength,
|
|
415
|
+
includedBytes: 0,
|
|
416
|
+
content: '',
|
|
417
|
+
reason: budgetLeft <= 0
|
|
418
|
+
? `total evidence budget (${MAX_EVIDENCE_BYTES_TOTAL} bytes) exhausted before this source`
|
|
419
|
+
: `${reservedElsewhere} of the ${budgetLeft} bytes left are reserved for dimension(s) ${waiting.join(', ')} — their only remaining evidence comes later in the source order`
|
|
420
|
+
});
|
|
421
|
+
continue;
|
|
422
|
+
}
|
|
423
|
+
const includedBytes = Math.min(raw.byteLength, MAX_EVIDENCE_BYTES_PER_FILE, allowance);
|
|
424
|
+
const content = raw.subarray(0, includedBytes).toString('utf8');
|
|
425
|
+
budgetLeft -= includedBytes;
|
|
426
|
+
pending.delete(index);
|
|
427
|
+
collected.push({
|
|
428
|
+
...base,
|
|
429
|
+
status: 'found',
|
|
430
|
+
totalBytes: raw.byteLength,
|
|
431
|
+
includedBytes,
|
|
432
|
+
content,
|
|
433
|
+
reason: ''
|
|
434
|
+
});
|
|
435
|
+
}
|
|
436
|
+
return collected;
|
|
437
|
+
}
|
|
438
|
+
function renderEvidenceSection(collected) {
|
|
439
|
+
return collected
|
|
440
|
+
.map((item, index) => {
|
|
441
|
+
const heading = `### [${index + 1}] ${item.source.key} — ${item.source.label}`;
|
|
442
|
+
const supports = `SUPPORTS: ${item.source.supports.join(', ')}`;
|
|
443
|
+
if (item.status === 'found') {
|
|
444
|
+
const status = item.includedBytes < item.totalBytes
|
|
445
|
+
? `FOUND at ${item.relativePath} — TRUNCATED, showing the first ${item.includedBytes} of ${item.totalBytes} bytes`
|
|
446
|
+
: `FOUND at ${item.relativePath} — ${item.totalBytes} bytes`;
|
|
447
|
+
return `${heading}\n${supports}\nSTATUS: ${status}\n<<<EVIDENCE\n${item.content}\n>>>EVIDENCE`;
|
|
448
|
+
}
|
|
449
|
+
return `${heading}\n${supports}\nSTATUS: MISSING (${item.status}) — no evidence available from ${item.relativePath}: ${item.reason}`;
|
|
450
|
+
})
|
|
451
|
+
.join('\n\n');
|
|
452
|
+
}
|
|
453
|
+
const EVIDENCE_RULES = `## Binding rules for the four verdicts
|
|
454
|
+
1. A dimension may be "pass" ONLY if at least one source in its SUPPORTS list has STATUS: FOUND above, and that source's content actually supports the verdict. The service re-checks this: a "pass" whose supporting sources are all missing/empty/omitted is downgraded to "inconclusive" before any human sees it.
|
|
455
|
+
2. If the evidence a dimension needs is MISSING, EMPTY, or OMITTED, return "inconclusive" with confidence "low". Do not guess "pass".
|
|
456
|
+
3. Absence of evidence is not evidence of absence: "no problem found in what I was given" is "inconclusive", never "pass".
|
|
457
|
+
4. Cite the bracketed source numbers (e.g. "[1]", "[5]") you relied on in each dimension's "evidence[].description"; use an empty list when the verdict is "inconclusive".
|
|
458
|
+
5. "allPass" may be true only when all four verdicts are "pass", and every non-"pass" dimension must be listed in "needsAttention".`;
|
|
459
|
+
/**
|
|
460
|
+
* Which dimensions had at least one FOUND source. A dimension absent from this
|
|
461
|
+
* set has no evidence at all behind it.
|
|
462
|
+
*/
|
|
463
|
+
function dimensionsWithEvidence(collected) {
|
|
464
|
+
const available = new Set();
|
|
465
|
+
for (const item of collected) {
|
|
466
|
+
if (item.status !== 'found')
|
|
467
|
+
continue;
|
|
468
|
+
for (const dimension of item.source.supports)
|
|
469
|
+
available.add(dimension);
|
|
470
|
+
}
|
|
471
|
+
return available;
|
|
472
|
+
}
|
|
473
|
+
/**
|
|
474
|
+
* The honesty guarantee (D1), enforced after parsing rather than merely asked
|
|
475
|
+
* for in the prompt. Prompt instructions are advisory — a model can still
|
|
476
|
+
* answer `pass` — so this pass makes the property structural: a dimension with
|
|
477
|
+
* no supporting evidence on disk is rewritten to `inconclusive` / `low`
|
|
478
|
+
* regardless of what the model returned.
|
|
479
|
+
*
|
|
480
|
+
* A `fail` is never softened: it is already stricter than `inconclusive`.
|
|
481
|
+
*/
|
|
482
|
+
function enforceEvidenceBackedVerdicts(dimensions, evidenceAvailableFor) {
|
|
483
|
+
return dimensions.map(dimension => {
|
|
484
|
+
if (dimension.verdict !== 'pass')
|
|
485
|
+
return dimension;
|
|
486
|
+
if (evidenceAvailableFor.has(dimension.dimension))
|
|
487
|
+
return dimension;
|
|
488
|
+
const downgraded = {
|
|
489
|
+
...dimension,
|
|
490
|
+
verdict: 'inconclusive',
|
|
491
|
+
confidence: 'low',
|
|
492
|
+
summary: `${dimension.summary} [evidence-gate: verdict downgraded from "pass" to "inconclusive" — no on-disk evidence source supporting "${dimension.dimension}" was available to the reviewer.]`
|
|
493
|
+
};
|
|
494
|
+
return downgraded;
|
|
495
|
+
});
|
|
496
|
+
}
|
|
497
|
+
/**
|
|
498
|
+
* True when the reply ends INSIDE a JSON string or with brackets still open —
|
|
499
|
+
* i.e. it was cut off mid-structure rather than being malformed. Distinguishing
|
|
500
|
+
* the two is the whole point: "the reply is not JSON" sends an operator looking
|
|
501
|
+
* for a schema bug, when the real cause is that nobody raised the output
|
|
502
|
+
* budget after the prompt grew.
|
|
503
|
+
*
|
|
504
|
+
* A minimal scanner is enough — braces and quotes inside string literals are
|
|
505
|
+
* skipped, escapes are honoured, and the text is never parsed.
|
|
506
|
+
*/
|
|
507
|
+
function looksTruncated(text) {
|
|
508
|
+
let inString = false;
|
|
509
|
+
let escaped = false;
|
|
510
|
+
let depth = 0;
|
|
511
|
+
for (const char of text) {
|
|
512
|
+
if (inString) {
|
|
513
|
+
if (escaped)
|
|
514
|
+
escaped = false;
|
|
515
|
+
else if (char === '\\')
|
|
516
|
+
escaped = true;
|
|
517
|
+
else if (char === '"')
|
|
518
|
+
inString = false;
|
|
519
|
+
continue;
|
|
520
|
+
}
|
|
521
|
+
if (char === '"')
|
|
522
|
+
inString = true;
|
|
523
|
+
else if (char === '{' || char === '[')
|
|
524
|
+
depth += 1;
|
|
525
|
+
else if (char === '}' || char === ']')
|
|
526
|
+
depth -= 1;
|
|
527
|
+
}
|
|
528
|
+
return inString || depth > 0;
|
|
529
|
+
}
|
|
530
|
+
/** The facts an operator needs to tell a budget problem from a format problem. */
|
|
531
|
+
function describeOutputBudget(maxTokens, outputTokens, characters) {
|
|
532
|
+
return `output budget: maxTokens=${maxTokens}, provider-reported output tokens=${outputTokens}, characters returned=${characters}`;
|
|
533
|
+
}
|
|
534
|
+
/**
|
|
535
|
+
* N4 — say WHICH signal diagnosed the truncation, and flag the provider's usage
|
|
536
|
+
* numbers when they are the only thing pointing at the ceiling. Measured on
|
|
537
|
+
* this repo's machine: `input_tokens: 150` for a ~32 KiB prompt, so a reporter
|
|
538
|
+
* that cannot count the input should not be trusted to count the output.
|
|
539
|
+
*/
|
|
540
|
+
function describeTruncationSignal(structurallyCut, ceilingReached) {
|
|
541
|
+
if (structurallyCut && ceilingReached) {
|
|
542
|
+
return 'reply ends mid-structure AND provider-reported output reached the ceiling';
|
|
543
|
+
}
|
|
544
|
+
if (structurallyCut) {
|
|
545
|
+
return 'reply ends mid-structure (structural)';
|
|
546
|
+
}
|
|
547
|
+
return 'provider-reported output reached the ceiling only — the provider’s usage reporting is not trustworthy on its own, so verify before raising anything';
|
|
548
|
+
}
|
|
549
|
+
/**
|
|
550
|
+
* N4 — call the reviewer, retrying an EMPTY reply a bounded number of times.
|
|
551
|
+
*
|
|
552
|
+
* The empty reply is a separate failure mode from truncation and was measured
|
|
553
|
+
* at 2/3 on this repo's machine. It is intermittent, so a bounded retry turns
|
|
554
|
+
* most occurrences back into a completed review; when it does not, the caller
|
|
555
|
+
* gets `EmptyReviewReplyError` — classified and diagnosable — instead of a
|
|
556
|
+
* truncation message that sends the operator to raise a budget that was never
|
|
557
|
+
* the problem.
|
|
558
|
+
*
|
|
559
|
+
* The classification reads the runner's message because `LlmRunner` is a
|
|
560
|
+
* structural interface here (this module deliberately does not depend on the
|
|
561
|
+
* concrete provider module); "no text block" is the runner's own fixed wording
|
|
562
|
+
* for a response with no text content.
|
|
563
|
+
*/
|
|
564
|
+
async function callReviewer(runner, userPrompt, budget) {
|
|
565
|
+
let lastError;
|
|
566
|
+
for (let attempt = 1; attempt <= MAX_EMPTY_REPLY_ATTEMPTS; attempt += 1) {
|
|
567
|
+
try {
|
|
568
|
+
return await runner.call(SYSTEM_PROMPT, userPrompt, { maxTokens: budget.maxTokens });
|
|
569
|
+
}
|
|
570
|
+
catch (error) {
|
|
571
|
+
if (!isEmptyReplyError(error))
|
|
572
|
+
throw error;
|
|
573
|
+
lastError = error;
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
throw new EmptyReviewReplyError(`The provider returned NO TEXT BLOCK on ${String(MAX_EMPTY_REPLY_ATTEMPTS)}/${String(MAX_EMPTY_REPLY_ATTEMPTS)} attempts — this is an EMPTY-REPLY failure, NOT an output-budget truncation: the response carried no text content (a reasoning-only turn, a refusal, or a content filter), so there was no JSON to parse and raising the budget would not have helped. ${describeOutputBudget(budget.maxTokens, 0, 0)}. Last provider error: ${errorText(lastError)}`);
|
|
577
|
+
}
|
|
578
|
+
/**
|
|
579
|
+
* `allPass` / `needsAttention` are DERIVED from the verdicts, never copied
|
|
580
|
+
* verbatim from the model's own summary fields: a model that writes a fabricated
|
|
581
|
+
* `pass` line plus a matching `allPass: true` would otherwise produce exactly
|
|
582
|
+
* the forged clean handoff this primitive exists to prevent, and once the gate
|
|
583
|
+
* above rewrites a verdict the two would silently disagree.
|
|
584
|
+
*
|
|
585
|
+
* Both fields are only ever NARROWED, never widened — `allPass` cannot become
|
|
586
|
+
* true unless the model also said true and no dimension is non-`pass`, and a
|
|
587
|
+
* dimension the model itself flagged is never dropped from `needsAttention`.
|
|
588
|
+
*/
|
|
589
|
+
function summarizeVerdicts(dimensions, modelFlags) {
|
|
590
|
+
const nonPass = dimensions.filter(d => d.verdict !== 'pass').map(d => d.dimension);
|
|
591
|
+
const flaggedByModel = Array.isArray(modelFlags.needsAttention)
|
|
592
|
+
? modelFlags.needsAttention
|
|
593
|
+
: [];
|
|
594
|
+
return {
|
|
595
|
+
allPass: modelFlags.allPass !== false && dimensions.length > 0 && nonPass.length === 0,
|
|
596
|
+
needsAttention: [...new Set([...flaggedByModel, ...nonPass])]
|
|
597
|
+
};
|
|
598
|
+
}
|
|
26
599
|
export async function prepareFinalReview(rid, opts) {
|
|
27
600
|
const auditGoalPath = join(opts.projectRoot, '.peaks', '_runtime', opts.sessionId, 'audit-goal', `${rid}.json`);
|
|
28
601
|
let approvedGoal;
|
|
@@ -32,24 +605,65 @@ export async function prepareFinalReview(rid, opts) {
|
|
|
32
605
|
catch (err) {
|
|
33
606
|
throw new Error(`Cannot read approved goal from ${auditGoalPath}: ${err.message}`);
|
|
34
607
|
}
|
|
35
|
-
const
|
|
36
|
-
const
|
|
37
|
-
|
|
38
|
-
|
|
608
|
+
const evidence = collectEvidence(opts.projectRoot, opts.sessionId, rid);
|
|
609
|
+
const userPrompt = [
|
|
610
|
+
`Approved goal's success criteria: ${JSON.stringify(approvedGoal.successCriteria)}`,
|
|
611
|
+
'',
|
|
612
|
+
'## On-disk evidence',
|
|
613
|
+
'You have NO tools and NO filesystem access — the blocks below are ALL the evidence that exists for this review. They were collected read-only by the service; nothing was executed.',
|
|
614
|
+
'',
|
|
615
|
+
renderEvidenceSection(evidence),
|
|
616
|
+
'',
|
|
617
|
+
EVIDENCE_RULES,
|
|
618
|
+
'',
|
|
619
|
+
'Prepare the 4-dim review evidence.'
|
|
620
|
+
].join('\n');
|
|
621
|
+
// D1 layer 3: the ceiling follows the evidence actually inlined, so the two
|
|
622
|
+
// sides of the call cannot drift apart again. N4 adds the env lever on top.
|
|
623
|
+
const includedEvidenceBytes = evidence.reduce((sum, item) => sum + item.includedBytes, 0);
|
|
624
|
+
const derivedMaxTokens = outputBudgetForEvidence(includedEvidenceBytes);
|
|
625
|
+
const maxTokens = resolveOutputBudget(includedEvidenceBytes);
|
|
626
|
+
const response = await callReviewer(opts.llmRunner, userPrompt, { maxTokens, derivedMaxTokens });
|
|
39
627
|
let parsed;
|
|
40
628
|
try {
|
|
41
629
|
parsed = JSON.parse(response.output);
|
|
42
630
|
}
|
|
43
631
|
catch (err) {
|
|
44
|
-
|
|
632
|
+
const budget = describeOutputBudget(maxTokens, response.tokens.output, response.output.length);
|
|
633
|
+
// N4 — the STRUCTURAL judgement is primary: `looksTruncated()` reads the
|
|
634
|
+
// reply itself and needs no cooperation from the provider. The
|
|
635
|
+
// `output_tokens >= maxTokens` comparison is kept as a corroborating
|
|
636
|
+
// signal, but it cannot be the only one: this endpoint reported
|
|
637
|
+
// `input_tokens: 150` for a ~32 KiB prompt, so its usage numbers are not
|
|
638
|
+
// trustworthy on their own, and a provider that under-reports a truncated
|
|
639
|
+
// reply would otherwise have it filed below as "not valid JSON" — sending
|
|
640
|
+
// an operator to look for a schema bug that does not exist. Which signal
|
|
641
|
+
// fired is reported, so the diagnosis is auditable rather than inferred.
|
|
642
|
+
const structurallyCut = looksTruncated(response.output);
|
|
643
|
+
const ceilingReached = response.tokens.output >= maxTokens;
|
|
644
|
+
if (structurallyCut || ceilingReached) {
|
|
645
|
+
throw new IncompleteFinalReviewError(`LLM output was TRUNCATED by the output budget before the 4-dim envelope was complete — this is an OUTPUT-BUDGET failure, not a malformed reply. Raise the budget: it scales with inlined evidence bytes, and ${MAX_OUTPUT_TOKENS_ENV} overrides it outright for this run. Signal: ${describeTruncationSignal(structurallyCut, ceilingReached)}. ${budget}. Parser said: ${err.message}`);
|
|
646
|
+
}
|
|
647
|
+
throw new IncompleteFinalReviewError(`LLM output is not valid JSON: ${err.message} (${budget})`);
|
|
45
648
|
}
|
|
46
649
|
const output = parsed;
|
|
47
650
|
const presentDimensions = new Set(output.dimensions.map((d) => d.dimension));
|
|
48
651
|
const missing = REQUIRED_DIMENSIONS.filter(d => !presentDimensions.has(d));
|
|
49
652
|
if (missing.length > 0) {
|
|
50
|
-
|
|
653
|
+
// A reply that stopped at the ceiling and still parsed is still a budget
|
|
654
|
+
// problem, so the same diagnosis is attached here. N4: the structural
|
|
655
|
+
// reading counts too — a reply cut off mid-array parses only because the
|
|
656
|
+
// envelope it produced happened to be closed early.
|
|
657
|
+
const budgetExhausted = response.tokens.output >= maxTokens || looksTruncated(response.output);
|
|
658
|
+
throw new IncompleteFinalReviewError(`Missing required dimensions: ${missing.join(', ')}${budgetExhausted
|
|
659
|
+
? ` — the provider hit the output budget and the reply was cut short (${describeOutputBudget(maxTokens, response.tokens.output, response.output.length)}); raise the budget instead of retrying blindly.`
|
|
660
|
+
: ''}`);
|
|
51
661
|
}
|
|
52
|
-
|
|
662
|
+
// D1: the verdicts must rest on evidence that actually existed, and the
|
|
663
|
+
// derived summary flags must match the verdicts — see the two helpers above.
|
|
664
|
+
const dimensions = enforceEvidenceBackedVerdicts(output.dimensions, dimensionsWithEvidence(evidence));
|
|
665
|
+
const { allPass, needsAttention } = summarizeVerdicts(dimensions, output);
|
|
666
|
+
return { ...output, dimensions, allPass, needsAttention };
|
|
53
667
|
}
|
|
54
668
|
export function decideFifthDimension(input) {
|
|
55
669
|
if (input.audit === null)
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
export { prepareFinalReview, type PrepareFinalReviewOptions, type LlmRunner, IncompleteFinalReviewError, } from './final-review-service.js';
|
|
1
|
+
export { prepareFinalReview, type PrepareFinalReviewOptions, type LlmRunner, IncompleteFinalReviewError, MAX_EVIDENCE_BYTES_PER_FILE, MAX_EVIDENCE_BYTES_TOTAL, } from './final-review-service.js';
|
|
2
2
|
export type { DimensionKind, DimensionVerdict, EvidenceKind, DimensionConfidence, EvidenceItem, DimensionEvidence, FinalReviewOutput, } from './final-review-types.js';
|
|
@@ -1 +1 @@
|
|
|
1
|
-
export { prepareFinalReview, IncompleteFinalReviewError, } from './final-review-service.js';
|
|
1
|
+
export { prepareFinalReview, IncompleteFinalReviewError, MAX_EVIDENCE_BYTES_PER_FILE, MAX_EVIDENCE_BYTES_TOTAL, } from './final-review-service.js';
|