peaks-loop 4.0.42 → 4.0.44

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/CHANGELOG.md +59 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/dist/cli/commands/_register.js +4 -0
  5. package/dist/cli/commands/api-diff-commands.d.ts +16 -0
  6. package/dist/cli/commands/api-diff-commands.js +55 -0
  7. package/dist/cli/commands/audit-commands.d.ts +16 -3
  8. package/dist/cli/commands/audit-commands.js +84 -31
  9. package/dist/cli/commands/codegraph-commands.js +191 -6
  10. package/dist/cli/commands/final-review-commands.d.ts +34 -10
  11. package/dist/cli/commands/final-review-commands.js +130 -34
  12. package/dist/cli/commands/job-commands.js +4 -2
  13. package/dist/cli/commands/scan-commands.js +1 -1
  14. package/dist/cli/commands/share-commands.d.ts +49 -0
  15. package/dist/cli/commands/share-commands.js +114 -14
  16. package/dist/cli/commands/test-commands.d.ts +60 -3
  17. package/dist/cli/commands/test-commands.js +125 -7
  18. package/dist/services/audit/audit-goal-service.js +38 -3
  19. package/dist/services/codegraph/codegraph-autorefresh.js +12 -0
  20. package/dist/services/codegraph/codegraph-exclude-integrity.d.ts +61 -0
  21. package/dist/services/codegraph/codegraph-exclude-integrity.js +98 -0
  22. package/dist/services/codegraph/codegraph-exclude-reconciler.d.ts +26 -0
  23. package/dist/services/codegraph/codegraph-exclude-reconciler.js +217 -0
  24. package/dist/services/codegraph/codegraph-exclude-repair.d.ts +102 -0
  25. package/dist/services/codegraph/codegraph-exclude-repair.js +266 -0
  26. package/dist/services/codegraph/codegraph-preflight-service.js +12 -0
  27. package/dist/services/codegraph/codegraph-service.d.ts +0 -1
  28. package/dist/services/codegraph/codegraph-service.js +5 -4
  29. package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.d.ts +29 -0
  30. package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.js +88 -0
  31. package/dist/services/doctor/doctor-service/checks/ecc-hooks-schema-drift.d.ts +65 -0
  32. package/dist/services/doctor/doctor-service/checks/ecc-hooks-schema-drift.js +186 -0
  33. package/dist/services/doctor/doctor-service/plugin-registry.js +4 -0
  34. package/dist/services/doctor/doctor-service/types.d.ts +47 -0
  35. package/dist/services/final-review/final-review-service.d.ts +154 -0
  36. package/dist/services/final-review/final-review-service.js +621 -7
  37. package/dist/services/final-review/index.d.ts +1 -1
  38. package/dist/services/final-review/index.js +1 -1
  39. package/dist/services/llm/anthropic-runner.d.ts +87 -0
  40. package/dist/services/llm/anthropic-runner.js +171 -0
  41. package/dist/services/llm/stub-runner.d.ts +11 -0
  42. package/dist/services/llm/stub-runner.js +33 -0
  43. package/dist/services/prd/handoff-auto-regen.js +0 -1
  44. package/dist/services/prd/handoff-service.d.ts +9 -1
  45. package/dist/services/prd/handoff-service.js +48 -6
  46. package/dist/services/prd/project-scan-bootstrap-service.js +7 -7
  47. package/dist/services/scan/api-diff-openapi.d.ts +32 -0
  48. package/dist/services/scan/api-diff-openapi.js +359 -0
  49. package/dist/services/scan/api-diff-recorded.d.ts +96 -0
  50. package/dist/services/scan/api-diff-recorded.js +577 -0
  51. package/dist/services/scan/api-diff-service.d.ts +34 -0
  52. package/dist/services/scan/api-diff-service.js +407 -0
  53. package/dist/services/scan/api-diff-types.d.ts +116 -0
  54. package/dist/services/scan/api-diff-types.js +46 -0
  55. package/dist/services/scan/archetype-service.js +27 -1
  56. package/dist/services/scan/existing-system-service.js +17 -4
  57. package/dist/services/scan/hook-convention-service.d.ts +26 -0
  58. package/dist/services/scan/hook-convention-service.js +562 -0
  59. package/dist/services/scan/scan-types.d.ts +47 -0
  60. package/dist/services/session/caller-binding-service.d.ts +28 -0
  61. package/dist/services/session/caller-binding-service.js +10 -2
  62. package/dist/services/session/caller-id-types.d.ts +12 -2
  63. package/dist/services/session/index.d.ts +2 -2
  64. package/dist/services/session/index.js +2 -2
  65. package/dist/services/session/session-binding-bridge.js +11 -6
  66. package/dist/services/session/session-manager.d.ts +33 -1
  67. package/dist/services/session/session-manager.js +84 -25
  68. package/dist/services/skills/skill-presence-service.d.ts +17 -3
  69. package/dist/services/skills/skill-presence-service.js +23 -3
  70. package/package.json +7 -5
  71. package/skills/bee/peaks-rd/SKILL.md +11 -3
  72. package/skills/peaks-code/references/existing-system-extraction.md +5 -1
  73. package/skills/peaks-code/references/frontend-only-mode.md +48 -6
  74. package/skills/peaks-code/references/project-scan-checklist.md +20 -1
  75. package/skills/peaks-doctor/references/doctor-check-catalog.md +1 -0
  76. package/skills/peaks-final-review/SKILL.md +43 -32
@@ -23,6 +23,579 @@ export class IncompleteFinalReviewError extends Error {
23
23
  this.name = 'IncompleteFinalReviewError';
24
24
  }
25
25
  }
26
+ /**
27
+ * N4 — the reply carried no text block at all.
28
+ *
29
+ * Measured 2/3 on this repo's own machine, and it is NOT truncation: the
30
+ * provider answered with a response whose `content` has no `text` block (a
31
+ * reasoning-only turn, a refusal, or a content filter), so there is no JSON to
32
+ * parse and no budget to raise — an operator sent to "raise the budget" for
33
+ * this failure would be sent the wrong way. It gets its own class, its own
34
+ * `code`, and a message that says so, so it is diagnosable instead of being
35
+ * flattened into "not valid JSON".
36
+ */
37
+ export class EmptyReviewReplyError extends Error {
38
+ code = 'EMPTY_FINAL_REVIEW_REPLY';
39
+ constructor(message) {
40
+ super(message);
41
+ this.name = 'EmptyReviewReplyError';
42
+ }
43
+ }
44
+ /**
45
+ * How many times an empty reply is retried before it is reported. The failure
46
+ * was 2/3 on the observed machine — intermittent, not systematic — so a small
47
+ * bounded retry converts most of it into a completed review, while 3 attempts
48
+ * keeps a genuinely broken provider from being hammered.
49
+ */
50
+ export const MAX_EMPTY_REPLY_ATTEMPTS = 3;
51
+ /** The runner's own wording for "the response had no text block to return". */
52
+ function isEmptyReplyError(error) {
53
+ return error instanceof Error && /no text block/i.test(error.message);
54
+ }
55
+ function errorText(error) {
56
+ return error instanceof Error ? error.message : String(error);
57
+ }
58
+ /* ------------------------------------------------------------------ *
59
+ * D1 — on-disk evidence collection.
60
+ *
61
+ * The reviewer LLM has NO tools and NO filesystem access: whatever the
62
+ * service does not inline into the prompt does not exist for it. Feeding it
63
+ * only the success criteria forced a guess — the observed failure mode was a
64
+ * 4/4 `inconclusive` verdict, and the dangerous one is an invented `pass`
65
+ * that SKILL.md would read as a clean handoff. Everything below exists so the
66
+ * verdicts rest on what is actually on disk, and so a missing artifact is
67
+ * reported as MISSING instead of being silently skipped.
68
+ * ------------------------------------------------------------------ */
69
+ /**
70
+ * Per-file evidence cap. Real evidence artifacts in this repo run 11–18 KB
71
+ * (`rd/tech-doc.md`, `rd/code-review.md`, `qa/*-findings-*.md`); 8 KB keeps the
72
+ * head of every file (header + verdict + first tables) without letting one
73
+ * verbose artifact crowd out the other sources. Enforced in BYTES against the
74
+ * raw buffer, so multi-byte (CJK) content cannot slip past the cap.
75
+ */
76
+ export const MAX_EVIDENCE_BYTES_PER_FILE = 8 * 1024;
77
+ /**
78
+ * Total evidence budget across all sources. 32 KB ≈ 8k tokens of input, which
79
+ * keeps the prompt far inside any modern context window. Sources that do not
80
+ * fit are reported as OMITTED — never dropped silently.
81
+ *
82
+ * This is an INPUT cap and stays fixed. The output ceiling that has to sit
83
+ * opposite it is derived per call by `outputBudgetForEvidence()` below — the
84
+ * two used to drift apart, and that drift was the defect.
85
+ */
86
+ export const MAX_EVIDENCE_BYTES_TOTAL = 32 * 1024;
87
+ /**
88
+ * Reserved floor per dimension — the anti-starvation guarantee.
89
+ *
90
+ * The allocator below used to be strictly first-come-first-served: each source
91
+ * took `min(perFileCap, budgetLeft)` in source order. With this repo's own
92
+ * 9-source evidence set (`2026-09-12-session-e37ef0`, measured) sources 1-4
93
+ * consumed the whole 32 KiB — 4 x 8,192 = 32,768, the cap to the byte — before
94
+ * source 5 was even opened. `existing-functionality-intact` is supplied ONLY by
95
+ * `rd/tech-doc.md` (6th) and `prd/handoff.md` (9th), so that one dimension
96
+ * reached the reviewer with zero evidence on every run and its verdict was
97
+ * structurally locked to `inconclusive` no matter how good the work was. A gate
98
+ * that is always red is noise, and an operator trained to ignore noise has no
99
+ * gate at all — the same harm as a gate that never fires, only quieter.
100
+ *
101
+ * So each dimension with at least one readable source on disk gets one floor
102
+ * reserved for the FIRST such source, and no source that is not that holder may
103
+ * spend it. The reservation is a floor, never a quota: it is released the
104
+ * instant its holder is served, and whatever the holder does not use flows back
105
+ * into the sequential allocation unchanged.
106
+ *
107
+ * Why 4 KiB: the per-file cap exists to keep the "header + verdict + first
108
+ * tables" — the part a reviewer actually cites. Measured on the same run's
109
+ * artifacts (9 files, 8,164-20,543 bytes each): every one of them states its
110
+ * verdict inside the first ~700 bytes. 4 KiB is ~5x that, so a floor holder is
111
+ * not there for depth — it is there so its dimension is not blind. Four
112
+ * dimensions x 4 KiB = 16 KiB of the 32 KiB cap, so at least half the budget
113
+ * still flows through the sequential path below.
114
+ */
115
+ export const MIN_EVIDENCE_BYTES_PER_DIMENSION = 4 * 1024;
116
+ /* ------------------------------------------------------------------ *
117
+ * D1 layer 3 — the OUTPUT budget.
118
+ *
119
+ * The input side above was raised to 32 KiB of inlined evidence, but the
120
+ * output ceiling stayed hard-coded at 3000 tokens. Measured on this repo's own
121
+ * run (rid `2026-09-12-codegraph-exclude-integrity`, 9 sources, 32 KiB
122
+ * inlined, real anthropic provider): 3/3 attempts failed — 2x
123
+ * INCOMPLETE_FINAL_REVIEW with the JSON cut off mid-string, 1x a reply with no
124
+ * text block. A 4-dimension envelope (4 x `summary` + `evidence[]` +
125
+ * `confidence`, plus `overallSummary`) does not fit in 3000 tokens once the
126
+ * model has ~8k tokens of evidence it is required to cite.
127
+ *
128
+ * N4 — the first fix of this layer derived the ceiling as "3000 + bytes/8" and
129
+ * called 8192 "a backstop only: it does not bind today", citing a 4775-token
130
+ * measurement. Both claims were falsified by re-measurement on the SAME
131
+ * machine the gate ships on (`deepseek-flash[1M]` via
132
+ * `api.deepseek.com/anthropic`, 2026-09-12, QA run 3/3 red + orchestrator
133
+ * re-run 3/3 red, rid `2026-09-12-codegraph-exclude-integrity`, byte-identical
134
+ * prompt):
135
+ *
136
+ * max_tokens=7096 (what the old formula produced for the 32 KiB pack)
137
+ * -> TRUNCATED. `output_tokens=7096`, 14078 characters.
138
+ * max_tokens=8192 -> TRUNCATED, `output_tokens=8192`, 574 characters.
139
+ * max_tokens=16000 -> COMPLETE, `output_tokens=10108`.
140
+ * max_tokens=32000 -> COMPLETE, `output_tokens=8883`.
141
+ *
142
+ * So 8192 WAS the binding constraint and was BELOW the requirement: a gate
143
+ * whose budget is short is worse than a red gate, because it releases the
144
+ * envelope only when the model happens to be terse.
145
+ *
146
+ * The 8:1 bytes-per-token term prices the VISIBLE envelope against the evidence
147
+ * the model must cite, but it was never the whole cost. On a reasoning model
148
+ * `max_tokens` also caps the hidden reasoning that precedes the first character
149
+ * of output, and that cost is invisible to a bytes-per-token formula (the
150
+ * 574-character run above is exactly that: the whole budget consumed before the
151
+ * envelope began). `REASONING_HEADROOM_TOKENS` below is that missing term.
152
+ *
153
+ * A 16384 ceiling with 6144 of headroom (derived 13240) was then measured on
154
+ * the SAME machine and shown to be too thin too: 10 real-machine runs, 2
155
+ * failures, BOTH genuine truncation — and the second one reported BOTH signals
156
+ * at once ("reply ends mid-structure AND provider-reported output reached the
157
+ * ceiling") with `maxTokens=13240`. The requirement is therefore not "derived
158
+ * >= the one 10108 sample" but "derived comfortably above every observed
159
+ * truncation point", which is what the constants below are now sized for.
160
+ * ------------------------------------------------------------------ */
161
+ /**
162
+ * Floor — also the value that shipped before this fix, so no evidence set can
163
+ * end up with a smaller budget than it had. ~3000 tokens is enough for the
164
+ * envelope skeleton plus a short paragraph per dimension.
165
+ */
166
+ export const MIN_OUTPUT_TOKENS = 3_000;
167
+ /**
168
+ * Headroom for the part of the reply that is not the envelope.
169
+ *
170
+ * A Messages-API-compatible endpoint applies `max_tokens` to the WHOLE
171
+ * response, and a reasoning model spends it on hidden reasoning before it
172
+ * emits a single character of the 4-dim envelope. Measured on this repo's own
173
+ * machine (2026-09-12, rid `2026-09-12-codegraph-exclude-integrity`,
174
+ * `deepseek-flash[1M]` via `api.deepseek.com/anthropic`): `max_tokens=8192`
175
+ * came back with `output_tokens=8192` and only **574** visible characters —
176
+ * the entire budget went to reasoning. A bytes-per-token estimate of the
177
+ * visible output cannot see that cost, which is why the previous formula
178
+ * budgeted 7096 for a reply that needs 10108 — and why a 13240 budget still
179
+ * truncated on 2 of 10 real runs.
180
+ *
181
+ * 12288 (12 KiB) is sized so the largest evidence pack the input caps allow
182
+ * lands at 23480 (see the formula below) — about 1.8x the largest value ever
183
+ * OBSERVED to truncate (13240), which is the margin the observed variance
184
+ * asks for. The numbers are in the block comment above.
185
+ */
186
+ export const REASONING_HEADROOM_TOKENS = 12 * 1024;
187
+ /**
188
+ * Ceiling, 32_000: the value a real run on this machine was forced to in order
189
+ * to complete the envelope at all, and the largest this endpoint was observed
190
+ * to accept. 16384 was tried first and truncated 2/10 — a ceiling that is
191
+ * merely "above the last successful measurement" is not above the requirement,
192
+ * because the requirement moves with the model's reasoning spend.
193
+ *
194
+ * A model that caps output at 8192 will refuse this. That is still strictly
195
+ * better than shipping a budget measured to be too small, and the env lever
196
+ * below lets an operator pull it down without a code change.
197
+ */
198
+ export const MAX_OUTPUT_TOKENS = 32_000;
199
+ /**
200
+ * Environment lever. The old failure message told the operator to "raise the
201
+ * budget" while the budget was a module constant with no CLI flag and no env
202
+ * var — an instruction that could not be carried out from any surface the
203
+ * operator has. This is that lever.
204
+ *
205
+ * The value is the output ceiling in tokens; it OVERRIDES the derivation below
206
+ * (it is not a bonus added to it). Unset/invalid/out-of-range handling is in
207
+ * `resolveOutputBudget`.
208
+ */
209
+ export const MAX_OUTPUT_TOKENS_ENV = 'PEAKS_FINAL_REVIEW_MAX_OUTPUT_TOKENS';
210
+ /**
211
+ * Absolute upper bound the env lever may reach. An endpoint that accepts
212
+ * `max_tokens` at all accepts this; anything above it is a typo (a stray extra
213
+ * digit), not an intent, and clamping is safer than sending it.
214
+ */
215
+ export const HARD_MAX_OUTPUT_TOKENS = 64_000;
216
+ /**
217
+ * Inlined bytes that buy one extra output token — 4:1.
218
+ *
219
+ * This term prices the visible envelope (4 x `summary` + `evidence[]` +
220
+ * `confidence` + `overallSummary`) against the evidence the model is required
221
+ * to cite. It was 8:1, which put the 32 KiB pack at 4096 tokens of visible
222
+ * output; the same pack has been observed to complete at 10108 and to truncate
223
+ * at 13240, so 8:1 was pricing the visible side BELOW its own measurement.
224
+ * 4:1 doubles it to 8192. It is still not treated as the whole budget — see
225
+ * `REASONING_HEADROOM_TOKENS`.
226
+ */
227
+ export const EVIDENCE_BYTES_PER_OUTPUT_TOKEN = 4;
228
+ /**
229
+ * Output ceiling for a call whose prompt carries `includedEvidenceBytes` bytes
230
+ * of inlined evidence. Pure, total, and clamped on both ends — the same
231
+ * evidence pack always yields the same budget.
232
+ */
233
+ export function outputBudgetForEvidence(includedEvidenceBytes) {
234
+ const scaled = MIN_OUTPUT_TOKENS +
235
+ REASONING_HEADROOM_TOKENS +
236
+ Math.ceil(Math.max(0, includedEvidenceBytes) / EVIDENCE_BYTES_PER_OUTPUT_TOKEN);
237
+ return Math.min(MAX_OUTPUT_TOKENS, Math.max(MIN_OUTPUT_TOKENS, scaled));
238
+ }
239
+ /**
240
+ * The budget the call actually uses: the derived one, unless
241
+ * `PEAKS_FINAL_REVIEW_MAX_OUTPUT_TOKENS` overrides it.
242
+ *
243
+ * An override that is not a positive integer THROWS rather than being ignored:
244
+ * a silent fallback would leave an operator who passed a bad value with the
245
+ * exact experience this lever exists to remove — a budget they cannot move.
246
+ * Out-of-range values are clamped, not rejected, so a model needing more than
247
+ * `HARD_MAX_OUTPUT_TOKENS` (or a model needing less than `MIN_OUTPUT_TOKENS`)
248
+ * still gets a call made.
249
+ */
250
+ export function resolveOutputBudget(includedEvidenceBytes, env = process.env) {
251
+ const derived = outputBudgetForEvidence(includedEvidenceBytes);
252
+ const raw = env[MAX_OUTPUT_TOKENS_ENV];
253
+ if (raw === undefined || raw.trim() === '')
254
+ return derived;
255
+ const parsed = Number.parseInt(raw.trim(), 10);
256
+ if (!Number.isInteger(parsed) || parsed <= 0 || String(parsed) !== raw.trim()) {
257
+ throw new Error(`${MAX_OUTPUT_TOKENS_ENV} must be a positive integer number of output tokens (got "${raw}"); ` +
258
+ `unset it to use the derived budget of ${String(derived)} tokens.`);
259
+ }
260
+ return Math.min(HARD_MAX_OUTPUT_TOKENS, Math.max(MIN_OUTPUT_TOKENS, parsed));
261
+ }
262
+ function evidenceSourcesFor(rid) {
263
+ return [
264
+ {
265
+ key: 'qa-test-report',
266
+ label: 'QA execution report (per-command pass/fail counts)',
267
+ segments: ['qa', 'test-reports', `${rid}.md`],
268
+ supports: ['functional-completeness', 'problem-resolution', 'no-new-bugs']
269
+ },
270
+ {
271
+ key: 'qa-test-cases',
272
+ label: 'QA test cases (acceptance-criterion to test mapping)',
273
+ segments: ['qa', 'test-cases', `${rid}.md`],
274
+ supports: ['functional-completeness', 'problem-resolution']
275
+ },
276
+ {
277
+ key: 'qa-security-findings',
278
+ label: 'QA security findings',
279
+ segments: ['qa', `security-findings-${rid}.md`],
280
+ supports: ['no-new-bugs']
281
+ },
282
+ {
283
+ key: 'qa-performance-findings',
284
+ label: 'QA performance findings',
285
+ segments: ['qa', `performance-findings-${rid}.md`],
286
+ supports: ['no-new-bugs']
287
+ },
288
+ {
289
+ key: 'rd-code-review',
290
+ label: 'RD code review',
291
+ segments: ['rd', 'code-review.md'],
292
+ supports: ['no-new-bugs']
293
+ },
294
+ {
295
+ key: 'rd-security-review',
296
+ label: 'RD security review',
297
+ segments: ['rd', 'security-review.md'],
298
+ supports: ['no-new-bugs']
299
+ },
300
+ {
301
+ key: 'rd-tech-doc',
302
+ label: 'RD tech doc (public surface / design intent)',
303
+ segments: ['rd', 'tech-doc.md'],
304
+ supports: ['existing-functionality-intact']
305
+ },
306
+ {
307
+ key: 'rd-bug-analysis',
308
+ label: 'RD bug analysis (original problem statement)',
309
+ segments: ['rd', 'bug-analysis.md'],
310
+ supports: ['problem-resolution']
311
+ },
312
+ {
313
+ key: 'prd-handoff',
314
+ label: 'PRD handoff (approved scope + non-goals)',
315
+ segments: ['prd', 'handoff.md'],
316
+ supports: ['functional-completeness', 'existing-functionality-intact']
317
+ }
318
+ ];
319
+ }
320
+ /** Pure read: no commands, no git, no test execution — files only. */
321
+ function readEvidence(projectRoot, sessionId, rid) {
322
+ const runtimeRoot = join(projectRoot, '.peaks', '_runtime', sessionId);
323
+ return evidenceSourcesFor(rid).map(source => {
324
+ const relativePath = ['.peaks', '_runtime', sessionId, ...source.segments].join('/');
325
+ const absolutePath = join(runtimeRoot, ...source.segments);
326
+ let raw = null;
327
+ let error = '';
328
+ try {
329
+ raw = readFileSync(absolutePath);
330
+ }
331
+ catch (err) {
332
+ error = err instanceof Error ? err.message : String(err);
333
+ }
334
+ return {
335
+ source,
336
+ relativePath,
337
+ absolutePath,
338
+ raw,
339
+ error,
340
+ blank: raw === null || raw.toString('utf8').trim().length === 0
341
+ };
342
+ });
343
+ }
344
+ /**
345
+ * The one source per dimension that the budget promises to reach: the first
346
+ * non-blank candidate that can supply that dimension. A dimension with several
347
+ * candidates needs exactly one of them to survive, and the earliest is the one
348
+ * the sequential order would have reached anyway — so naming it costs the other
349
+ * sources nothing they were not already losing.
350
+ *
351
+ * A blank source is deliberately not a holder — see `MIN_EVIDENCE_BYTES_PER_DIMENSION`.
352
+ */
353
+ function floorHolders(candidates) {
354
+ const holders = new Map();
355
+ const covered = new Set();
356
+ candidates.forEach((candidate, index) => {
357
+ if (candidate.blank)
358
+ return;
359
+ for (const dimension of candidate.source.supports) {
360
+ if (covered.has(dimension))
361
+ continue;
362
+ covered.add(dimension);
363
+ if (!holders.has(index))
364
+ holders.set(index, dimension);
365
+ }
366
+ });
367
+ return holders;
368
+ }
369
+ function collectEvidence(projectRoot, sessionId, rid) {
370
+ const candidates = readEvidence(projectRoot, sessionId, rid);
371
+ const holders = floorHolders(candidates);
372
+ /** Holders that have not been served yet — the floors still owed. */
373
+ const pending = new Set(holders.keys());
374
+ const collected = [];
375
+ let budgetLeft = MAX_EVIDENCE_BYTES_TOTAL;
376
+ for (const [index, candidate] of candidates.entries()) {
377
+ const { source, relativePath, absolutePath, raw } = candidate;
378
+ const base = { source, relativePath, absolutePath };
379
+ if (raw === null) {
380
+ collected.push({
381
+ ...base,
382
+ status: 'missing',
383
+ totalBytes: 0,
384
+ includedBytes: 0,
385
+ content: '',
386
+ reason: candidate.error
387
+ });
388
+ continue;
389
+ }
390
+ if (candidate.blank) {
391
+ collected.push({
392
+ ...base,
393
+ status: 'empty',
394
+ totalBytes: raw.byteLength,
395
+ includedBytes: 0,
396
+ content: '',
397
+ reason: `file exists but contains no reviewable content (${raw.byteLength} bytes)`
398
+ });
399
+ continue;
400
+ }
401
+ // The floor owed to every dimension still waiting on its holder is spent
402
+ // only on that holder. This is the whole fix: a source that needs no help
403
+ // can no longer eat the last dimension's only chance at being reviewed.
404
+ const holdsFloor = pending.has(index);
405
+ const reservedElsewhere = (pending.size - (holdsFloor ? 1 : 0)) * MIN_EVIDENCE_BYTES_PER_DIMENSION;
406
+ const allowance = budgetLeft - reservedElsewhere;
407
+ if (allowance <= 0) {
408
+ const waiting = [...pending]
409
+ .filter(holder => holder !== index)
410
+ .map(holder => `${holders.get(holder)} (source ${candidates[holder]?.source.key ?? '?'})`);
411
+ collected.push({
412
+ ...base,
413
+ status: 'omitted',
414
+ totalBytes: raw.byteLength,
415
+ includedBytes: 0,
416
+ content: '',
417
+ reason: budgetLeft <= 0
418
+ ? `total evidence budget (${MAX_EVIDENCE_BYTES_TOTAL} bytes) exhausted before this source`
419
+ : `${reservedElsewhere} of the ${budgetLeft} bytes left are reserved for dimension(s) ${waiting.join(', ')} — their only remaining evidence comes later in the source order`
420
+ });
421
+ continue;
422
+ }
423
+ const includedBytes = Math.min(raw.byteLength, MAX_EVIDENCE_BYTES_PER_FILE, allowance);
424
+ const content = raw.subarray(0, includedBytes).toString('utf8');
425
+ budgetLeft -= includedBytes;
426
+ pending.delete(index);
427
+ collected.push({
428
+ ...base,
429
+ status: 'found',
430
+ totalBytes: raw.byteLength,
431
+ includedBytes,
432
+ content,
433
+ reason: ''
434
+ });
435
+ }
436
+ return collected;
437
+ }
438
+ function renderEvidenceSection(collected) {
439
+ return collected
440
+ .map((item, index) => {
441
+ const heading = `### [${index + 1}] ${item.source.key} — ${item.source.label}`;
442
+ const supports = `SUPPORTS: ${item.source.supports.join(', ')}`;
443
+ if (item.status === 'found') {
444
+ const status = item.includedBytes < item.totalBytes
445
+ ? `FOUND at ${item.relativePath} — TRUNCATED, showing the first ${item.includedBytes} of ${item.totalBytes} bytes`
446
+ : `FOUND at ${item.relativePath} — ${item.totalBytes} bytes`;
447
+ return `${heading}\n${supports}\nSTATUS: ${status}\n<<<EVIDENCE\n${item.content}\n>>>EVIDENCE`;
448
+ }
449
+ return `${heading}\n${supports}\nSTATUS: MISSING (${item.status}) — no evidence available from ${item.relativePath}: ${item.reason}`;
450
+ })
451
+ .join('\n\n');
452
+ }
453
+ const EVIDENCE_RULES = `## Binding rules for the four verdicts
454
+ 1. A dimension may be "pass" ONLY if at least one source in its SUPPORTS list has STATUS: FOUND above, and that source's content actually supports the verdict. The service re-checks this: a "pass" whose supporting sources are all missing/empty/omitted is downgraded to "inconclusive" before any human sees it.
455
+ 2. If the evidence a dimension needs is MISSING, EMPTY, or OMITTED, return "inconclusive" with confidence "low". Do not guess "pass".
456
+ 3. Absence of evidence is not evidence of absence: "no problem found in what I was given" is "inconclusive", never "pass".
457
+ 4. Cite the bracketed source numbers (e.g. "[1]", "[5]") you relied on in each dimension's "evidence[].description"; use an empty list when the verdict is "inconclusive".
458
+ 5. "allPass" may be true only when all four verdicts are "pass", and every non-"pass" dimension must be listed in "needsAttention".`;
459
+ /**
460
+ * Which dimensions had at least one FOUND source. A dimension absent from this
461
+ * set has no evidence at all behind it.
462
+ */
463
+ function dimensionsWithEvidence(collected) {
464
+ const available = new Set();
465
+ for (const item of collected) {
466
+ if (item.status !== 'found')
467
+ continue;
468
+ for (const dimension of item.source.supports)
469
+ available.add(dimension);
470
+ }
471
+ return available;
472
+ }
473
+ /**
474
+ * The honesty guarantee (D1), enforced after parsing rather than merely asked
475
+ * for in the prompt. Prompt instructions are advisory — a model can still
476
+ * answer `pass` — so this pass makes the property structural: a dimension with
477
+ * no supporting evidence on disk is rewritten to `inconclusive` / `low`
478
+ * regardless of what the model returned.
479
+ *
480
+ * A `fail` is never softened: it is already stricter than `inconclusive`.
481
+ */
482
+ function enforceEvidenceBackedVerdicts(dimensions, evidenceAvailableFor) {
483
+ return dimensions.map(dimension => {
484
+ if (dimension.verdict !== 'pass')
485
+ return dimension;
486
+ if (evidenceAvailableFor.has(dimension.dimension))
487
+ return dimension;
488
+ const downgraded = {
489
+ ...dimension,
490
+ verdict: 'inconclusive',
491
+ confidence: 'low',
492
+ summary: `${dimension.summary} [evidence-gate: verdict downgraded from "pass" to "inconclusive" — no on-disk evidence source supporting "${dimension.dimension}" was available to the reviewer.]`
493
+ };
494
+ return downgraded;
495
+ });
496
+ }
497
+ /**
498
+ * True when the reply ends INSIDE a JSON string or with brackets still open —
499
+ * i.e. it was cut off mid-structure rather than being malformed. Distinguishing
500
+ * the two is the whole point: "the reply is not JSON" sends an operator looking
501
+ * for a schema bug, when the real cause is that nobody raised the output
502
+ * budget after the prompt grew.
503
+ *
504
+ * A minimal scanner is enough — braces and quotes inside string literals are
505
+ * skipped, escapes are honoured, and the text is never parsed.
506
+ */
507
+ function looksTruncated(text) {
508
+ let inString = false;
509
+ let escaped = false;
510
+ let depth = 0;
511
+ for (const char of text) {
512
+ if (inString) {
513
+ if (escaped)
514
+ escaped = false;
515
+ else if (char === '\\')
516
+ escaped = true;
517
+ else if (char === '"')
518
+ inString = false;
519
+ continue;
520
+ }
521
+ if (char === '"')
522
+ inString = true;
523
+ else if (char === '{' || char === '[')
524
+ depth += 1;
525
+ else if (char === '}' || char === ']')
526
+ depth -= 1;
527
+ }
528
+ return inString || depth > 0;
529
+ }
530
+ /** The facts an operator needs to tell a budget problem from a format problem. */
531
+ function describeOutputBudget(maxTokens, outputTokens, characters) {
532
+ return `output budget: maxTokens=${maxTokens}, provider-reported output tokens=${outputTokens}, characters returned=${characters}`;
533
+ }
534
+ /**
535
+ * N4 — say WHICH signal diagnosed the truncation, and flag the provider's usage
536
+ * numbers when they are the only thing pointing at the ceiling. Measured on
537
+ * this repo's machine: `input_tokens: 150` for a ~32 KiB prompt, so a reporter
538
+ * that cannot count the input should not be trusted to count the output.
539
+ */
540
+ function describeTruncationSignal(structurallyCut, ceilingReached) {
541
+ if (structurallyCut && ceilingReached) {
542
+ return 'reply ends mid-structure AND provider-reported output reached the ceiling';
543
+ }
544
+ if (structurallyCut) {
545
+ return 'reply ends mid-structure (structural)';
546
+ }
547
+ return 'provider-reported output reached the ceiling only — the provider’s usage reporting is not trustworthy on its own, so verify before raising anything';
548
+ }
549
+ /**
550
+ * N4 — call the reviewer, retrying an EMPTY reply a bounded number of times.
551
+ *
552
+ * The empty reply is a separate failure mode from truncation and was measured
553
+ * at 2/3 on this repo's machine. It is intermittent, so a bounded retry turns
554
+ * most occurrences back into a completed review; when it does not, the caller
555
+ * gets `EmptyReviewReplyError` — classified and diagnosable — instead of a
556
+ * truncation message that sends the operator to raise a budget that was never
557
+ * the problem.
558
+ *
559
+ * The classification reads the runner's message because `LlmRunner` is a
560
+ * structural interface here (this module deliberately does not depend on the
561
+ * concrete provider module); "no text block" is the runner's own fixed wording
562
+ * for a response with no text content.
563
+ */
564
+ async function callReviewer(runner, userPrompt, budget) {
565
+ let lastError;
566
+ for (let attempt = 1; attempt <= MAX_EMPTY_REPLY_ATTEMPTS; attempt += 1) {
567
+ try {
568
+ return await runner.call(SYSTEM_PROMPT, userPrompt, { maxTokens: budget.maxTokens });
569
+ }
570
+ catch (error) {
571
+ if (!isEmptyReplyError(error))
572
+ throw error;
573
+ lastError = error;
574
+ }
575
+ }
576
+ throw new EmptyReviewReplyError(`The provider returned NO TEXT BLOCK on ${String(MAX_EMPTY_REPLY_ATTEMPTS)}/${String(MAX_EMPTY_REPLY_ATTEMPTS)} attempts — this is an EMPTY-REPLY failure, NOT an output-budget truncation: the response carried no text content (a reasoning-only turn, a refusal, or a content filter), so there was no JSON to parse and raising the budget would not have helped. ${describeOutputBudget(budget.maxTokens, 0, 0)}. Last provider error: ${errorText(lastError)}`);
577
+ }
578
+ /**
579
+ * `allPass` / `needsAttention` are DERIVED from the verdicts, never copied
580
+ * verbatim from the model's own summary fields: a model that writes a fabricated
581
+ * `pass` line plus a matching `allPass: true` would otherwise produce exactly
582
+ * the forged clean handoff this primitive exists to prevent, and once the gate
583
+ * above rewrites a verdict the two would silently disagree.
584
+ *
585
+ * Both fields are only ever NARROWED, never widened — `allPass` cannot become
586
+ * true unless the model also said true and no dimension is non-`pass`, and a
587
+ * dimension the model itself flagged is never dropped from `needsAttention`.
588
+ */
589
+ function summarizeVerdicts(dimensions, modelFlags) {
590
+ const nonPass = dimensions.filter(d => d.verdict !== 'pass').map(d => d.dimension);
591
+ const flaggedByModel = Array.isArray(modelFlags.needsAttention)
592
+ ? modelFlags.needsAttention
593
+ : [];
594
+ return {
595
+ allPass: modelFlags.allPass !== false && dimensions.length > 0 && nonPass.length === 0,
596
+ needsAttention: [...new Set([...flaggedByModel, ...nonPass])]
597
+ };
598
+ }
26
599
  export async function prepareFinalReview(rid, opts) {
27
600
  const auditGoalPath = join(opts.projectRoot, '.peaks', '_runtime', opts.sessionId, 'audit-goal', `${rid}.json`);
28
601
  let approvedGoal;
@@ -32,24 +605,65 @@ export async function prepareFinalReview(rid, opts) {
32
605
  catch (err) {
33
606
  throw new Error(`Cannot read approved goal from ${auditGoalPath}: ${err.message}`);
34
607
  }
35
- const userPrompt = `Approved goal's success criteria: ${JSON.stringify(approvedGoal.successCriteria)}\n\nPrepare the 4-dim review evidence.`;
36
- const response = await opts.llmRunner.call(SYSTEM_PROMPT, userPrompt, {
37
- maxTokens: 3000
38
- });
608
+ const evidence = collectEvidence(opts.projectRoot, opts.sessionId, rid);
609
+ const userPrompt = [
610
+ `Approved goal's success criteria: ${JSON.stringify(approvedGoal.successCriteria)}`,
611
+ '',
612
+ '## On-disk evidence',
613
+ 'You have NO tools and NO filesystem access — the blocks below are ALL the evidence that exists for this review. They were collected read-only by the service; nothing was executed.',
614
+ '',
615
+ renderEvidenceSection(evidence),
616
+ '',
617
+ EVIDENCE_RULES,
618
+ '',
619
+ 'Prepare the 4-dim review evidence.'
620
+ ].join('\n');
621
+ // D1 layer 3: the ceiling follows the evidence actually inlined, so the two
622
+ // sides of the call cannot drift apart again. N4 adds the env lever on top.
623
+ const includedEvidenceBytes = evidence.reduce((sum, item) => sum + item.includedBytes, 0);
624
+ const derivedMaxTokens = outputBudgetForEvidence(includedEvidenceBytes);
625
+ const maxTokens = resolveOutputBudget(includedEvidenceBytes);
626
+ const response = await callReviewer(opts.llmRunner, userPrompt, { maxTokens, derivedMaxTokens });
39
627
  let parsed;
40
628
  try {
41
629
  parsed = JSON.parse(response.output);
42
630
  }
43
631
  catch (err) {
44
- throw new IncompleteFinalReviewError(`LLM output is not valid JSON: ${err.message}`);
632
+ const budget = describeOutputBudget(maxTokens, response.tokens.output, response.output.length);
633
+ // N4 — the STRUCTURAL judgement is primary: `looksTruncated()` reads the
634
+ // reply itself and needs no cooperation from the provider. The
635
+ // `output_tokens >= maxTokens` comparison is kept as a corroborating
636
+ // signal, but it cannot be the only one: this endpoint reported
637
+ // `input_tokens: 150` for a ~32 KiB prompt, so its usage numbers are not
638
+ // trustworthy on their own, and a provider that under-reports a truncated
639
+ // reply would otherwise have it filed below as "not valid JSON" — sending
640
+ // an operator to look for a schema bug that does not exist. Which signal
641
+ // fired is reported, so the diagnosis is auditable rather than inferred.
642
+ const structurallyCut = looksTruncated(response.output);
643
+ const ceilingReached = response.tokens.output >= maxTokens;
644
+ if (structurallyCut || ceilingReached) {
645
+ throw new IncompleteFinalReviewError(`LLM output was TRUNCATED by the output budget before the 4-dim envelope was complete — this is an OUTPUT-BUDGET failure, not a malformed reply. Raise the budget: it scales with inlined evidence bytes, and ${MAX_OUTPUT_TOKENS_ENV} overrides it outright for this run. Signal: ${describeTruncationSignal(structurallyCut, ceilingReached)}. ${budget}. Parser said: ${err.message}`);
646
+ }
647
+ throw new IncompleteFinalReviewError(`LLM output is not valid JSON: ${err.message} (${budget})`);
45
648
  }
46
649
  const output = parsed;
47
650
  const presentDimensions = new Set(output.dimensions.map((d) => d.dimension));
48
651
  const missing = REQUIRED_DIMENSIONS.filter(d => !presentDimensions.has(d));
49
652
  if (missing.length > 0) {
50
- throw new IncompleteFinalReviewError(`Missing required dimensions: ${missing.join(', ')}`);
653
+ // A reply that stopped at the ceiling and still parsed is still a budget
654
+ // problem, so the same diagnosis is attached here. N4: the structural
655
+ // reading counts too — a reply cut off mid-array parses only because the
656
+ // envelope it produced happened to be closed early.
657
+ const budgetExhausted = response.tokens.output >= maxTokens || looksTruncated(response.output);
658
+ throw new IncompleteFinalReviewError(`Missing required dimensions: ${missing.join(', ')}${budgetExhausted
659
+ ? ` — the provider hit the output budget and the reply was cut short (${describeOutputBudget(maxTokens, response.tokens.output, response.output.length)}); raise the budget instead of retrying blindly.`
660
+ : ''}`);
51
661
  }
52
- return output;
662
+ // D1: the verdicts must rest on evidence that actually existed, and the
663
+ // derived summary flags must match the verdicts — see the two helpers above.
664
+ const dimensions = enforceEvidenceBackedVerdicts(output.dimensions, dimensionsWithEvidence(evidence));
665
+ const { allPass, needsAttention } = summarizeVerdicts(dimensions, output);
666
+ return { ...output, dimensions, allPass, needsAttention };
53
667
  }
54
668
  export function decideFifthDimension(input) {
55
669
  if (input.audit === null)
@@ -1,2 +1,2 @@
1
- export { prepareFinalReview, type PrepareFinalReviewOptions, type LlmRunner, IncompleteFinalReviewError, } from './final-review-service.js';
1
+ export { prepareFinalReview, type PrepareFinalReviewOptions, type LlmRunner, IncompleteFinalReviewError, MAX_EVIDENCE_BYTES_PER_FILE, MAX_EVIDENCE_BYTES_TOTAL, } from './final-review-service.js';
2
2
  export type { DimensionKind, DimensionVerdict, EvidenceKind, DimensionConfidence, EvidenceItem, DimensionEvidence, FinalReviewOutput, } from './final-review-types.js';
@@ -1 +1 @@
1
- export { prepareFinalReview, IncompleteFinalReviewError, } from './final-review-service.js';
1
+ export { prepareFinalReview, IncompleteFinalReviewError, MAX_EVIDENCE_BYTES_PER_FILE, MAX_EVIDENCE_BYTES_TOTAL, } from './final-review-service.js';