peaks-loop 4.0.44 → 4.0.46

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. package/CHANGELOG.md +59 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/dist/cli/commands/code-runtime-commands.js +30 -19
  5. package/dist/cli/commands/codegraph-commands.d.ts +1 -0
  6. package/dist/cli/commands/codegraph-commands.js +49 -3
  7. package/dist/cli/commands/final-review-commands.js +3 -1
  8. package/dist/services/code/auto-compact-lifecycle.d.ts +39 -3
  9. package/dist/services/code/auto-compact-lifecycle.js +41 -6
  10. package/dist/services/code/auto-compact-orchestrator.js +20 -11
  11. package/dist/services/compact-statusline/compact-lifecycle-store.d.ts +11 -2
  12. package/dist/services/compact-statusline/compact-lifecycle-store.js +19 -1
  13. package/dist/services/compact-statusline/compact-statusline-service.d.ts +1 -1
  14. package/dist/services/compact-statusline/compact-statusline-service.js +22 -0
  15. package/dist/services/final-review/final-review-service.d.ts +224 -44
  16. package/dist/services/final-review/final-review-service.js +934 -97
  17. package/dist/services/final-review/index.d.ts +2 -1
  18. package/dist/services/final-review/index.js +2 -1
  19. package/dist/services/final-review/pre-post-diff.d.ts +137 -0
  20. package/dist/services/final-review/pre-post-diff.js +657 -0
  21. package/dist/services/skills/skill-statusline-renderer.js +30 -5
  22. package/dist/services/skills/statusline-palette.d.ts +6 -0
  23. package/dist/services/skills/statusline-palette.js +4 -1
  24. package/package.json +5 -5
  25. package/skills/peaks-code/SKILL.md +2 -2
  26. package/skills/peaks-final-review/SKILL.md +51 -18
  27. package/skills/peaks-final-review/references/4-dimensions.md +42 -5
@@ -1,6 +1,7 @@
1
1
  import { readFileSync } from 'node:fs';
2
2
  import { join } from 'node:path';
3
3
  import { isStale } from '../capability-audit-service/staleness.js';
4
+ import { API_DIFF_ARTIFACT_SEGMENTS, PRE_POST_DIFF_VERDICT_MARKER, classifyPrePostDiffVerdict, producePrePostDiff } from './pre-post-diff.js';
4
5
  const REQUIRED_DIMENSIONS = [
5
6
  'functional-completeness',
6
7
  'problem-resolution',
@@ -67,56 +68,149 @@ function errorText(error) {
67
68
  * reported as MISSING instead of being silently skipped.
68
69
  * ------------------------------------------------------------------ */
69
70
  /**
70
- * Per-file evidence cap. Real evidence artifacts in this repo run 11–18 KB
71
- * (`rd/tech-doc.md`, `rd/code-review.md`, `qa/*-findings-*.md`); 8 KB keeps the
72
- * head of every file (header + verdict + first tables) without letting one
73
- * verbose artifact crowd out the other sources. Enforced in BYTES against the
74
- * raw buffer, so multi-byte (CJK) content cannot slip past the cap.
75
- */
76
- export const MAX_EVIDENCE_BYTES_PER_FILE = 8 * 1024;
77
- /**
78
- * Total evidence budget across all sources. 32 KB ≈ 8k tokens of input, which
71
+ * Total evidence budget across all sources. 40 KB ≈ 10k tokens of input, which
79
72
  * keeps the prompt far inside any modern context window. Sources that do not
80
73
  * fit are reported as OMITTED — never dropped silently.
81
74
  *
82
- * This is an INPUT cap and stays fixed. The output ceiling that has to sit
83
- * opposite it is derived per call by `outputBudgetForEvidence()` below — the
84
- * two used to drift apart, and that drift was the defect.
85
- */
86
- export const MAX_EVIDENCE_BYTES_TOTAL = 32 * 1024;
87
- /**
88
- * Reserved floor per dimension — the anti-starvation guarantee.
89
- *
90
- * The allocator below used to be strictly first-come-first-served: each source
91
- * took `min(perFileCap, budgetLeft)` in source order. With this repo's own
92
- * 9-source evidence set (`2026-09-12-session-e37ef0`, measured) sources 1-4
93
- * consumed the whole 32 KiB — 4 x 8,192 = 32,768, the cap to the byte — before
94
- * source 5 was even opened. `existing-functionality-intact` is supplied ONLY by
95
- * `rd/tech-doc.md` (6th) and `prd/handoff.md` (9th), so that one dimension
96
- * reached the reviewer with zero evidence on every run and its verdict was
97
- * structurally locked to `inconclusive` no matter how good the work was. A gate
98
- * that is always red is noise, and an operator trained to ignore noise has no
99
- * gate at all — the same harm as a gate that never fires, only quieter.
100
- *
101
- * So each dimension with at least one readable source on disk gets one floor
102
- * reserved for the FIRST such source, and no source that is not that holder may
103
- * spend it. The reservation is a floor, never a quota: it is released the
104
- * instant its holder is served, and whatever the holder does not use flows back
105
- * into the sequential allocation unchanged.
106
- *
107
- * Why 4 KiB: the per-file cap exists to keep the "header + verdict + first
108
- * tables" — the part a reviewer actually cites. Measured on the same run's
109
- * artifacts (9 files, 8,164-20,543 bytes each): every one of them states its
110
- * verdict inside the first ~700 bytes. 4 KiB is ~5x that, so a floor holder is
111
- * not there for depth — it is there so its dimension is not blind. Four
112
- * dimensions x 4 KiB = 16 KiB of the 32 KiB cap, so at least half the budget
113
- * still flows through the sequential path below.
114
- */
115
- export const MIN_EVIDENCE_BYTES_PER_DIMENSION = 4 * 1024;
75
+ * This is an INPUT cap and stays fixed per run. The output ceiling that has to
76
+ * sit opposite it is derived per call by `outputBudgetForEvidence()` below —
77
+ * the two used to drift apart, and that drift was the defect.
78
+ *
79
+ * RE-EVALUATED 2026-09-12 (F-BLOCK) — 32 KiB did NOT hold once the tenth
80
+ * source (`final-review-pre-post-diff`) was appended. Measured on this repo's
81
+ * own run (`2026-09-12-session-e37ef0`, rid
82
+ * `2026-09-12-codegraph-exclude-integrity`), the ten sources are 2,621 / 8,164
83
+ * / 9,492 / 10,839 / 11,422 / 13,050 / 13,462 / 13,852 / 17,745 / 20,543
84
+ * bytes — 121,190 bytes on disk, 76,321 bytes once the 8 KiB per-file cap is
85
+ * applied. Four sources at the cap spent the old 32,768 to the byte, so the
86
+ * appended tenth (added LAST, by design) was the first block the budget
87
+ * dropped — on every run, and on the compact set QA measured too (9 x 3.8 KB
88
+ * ≈ 34 KB, which the old cap already could not hold).
89
+ *
90
+ * 40 KiB is a budget that fits the measured ten-source set at the allocator's
91
+ * per-source unit (5 x 8 KiB units + the smaller ones), and it is the largest
92
+ * this budget may grow to today: the derived output ceiling is
93
+ * `3000 + 12288 + bytes/4`, so it reaches MAX_OUTPUT_TOKENS (32,000) at exactly
94
+ * 66,848 bytes of inlined evidence and is clamped from there on — a cap at or
95
+ * above that point buys the reviewer no more output room at all.
96
+ *
97
+ * F-CAP — an earlier version of this comment claimed 40 KiB (40,960 =
98
+ * 10 x 4,096) was "the smallest cap that affords one floor-sized slice to EVERY
99
+ * source in the ten-source set". That was arithmetic about a budget, not a
100
+ * statement about the allocator: sources are capped at
101
+ * MAX_EVIDENCE_BYTES_PER_FILE (10 KiB each, derived — see below) while a floor
102
+ * is per DIMENSION (four of them), so 10 x 4,096 was never what any source
103
+ * received. Re-measured on
104
+ * the same two saturated fixtures (QA round 4: the 9 x 40 KB fixture and this
105
+ * repo's own file sizes), raising the cap from 32 KiB to 40 KiB left **4 of the
106
+ * 10 sources inlined with zero bytes** — `rd/security-review`,
107
+ * `rd/bug-analysis`, `prd/handoff` and the appended
108
+ * `final-review-pre-post-diff` — and it did not deliver the baseline either;
109
+ * all it did was change WHICH source was starved (at 32 KiB that source was
110
+ * `rd/code-review`).
111
+ *
112
+ * So read this constant as "how much evidence fits", never as "which sources
113
+ * arrive". Arrival is decided by the allocator below (one whole unit per source,
114
+ * with a dimension's floor reserving its holder's unit) and, for the two sources
115
+ * a dimension's verdict may not outlive, by the delivery gates on top of it.
116
+ * And note the honest consequence, stated rather than papered over: on a
117
+ * saturated run the allocator still omits sources, so `prd/handoff.md` and the
118
+ * pre/post diff can be dropped — which is exactly why
119
+ * `enforceScopeContractDelivery()` and `enforcePrePostDiffAvailability()` exist
120
+ * and why a `pass` on those two dimensions is keyed on delivery, not on
121
+ * existence.
122
+ */
123
+ export const MAX_EVIDENCE_BYTES_TOTAL = 40 * 1024;
124
+ /**
125
+ * Per-file evidence cap — and the allocator's UNIT. DERIVED, never typed.
126
+ *
127
+ * F5 — this used to be the literal `8 * 1024`, and the whole floor guarantee
128
+ * below was silently conditioned on `4 x perFile <= total`: the reservation
129
+ * promises every pending holder its own unit, and that promise holds only while
130
+ * the four units fit the budget together. At `8 * 1024` the inequality held by
131
+ * coincidence (4 x 8,192 = 32,768 <= 40,960) and nothing in the code said so —
132
+ * raising the cap past 10,240 would have starved all four dimensions in the
133
+ * same run, which reads as four independent red gates rather than one broken
134
+ * constant. The cap is therefore DIVIDED OUT OF the total: the relation is
135
+ * definitional instead of remembered, and `assertFloorReservationAffordable()`
136
+ * still checks it at the point of use (the division must also be exact).
137
+ *
138
+ * The value is 10,240 because that is `40,960 / 4` — the largest unit the
139
+ * four-dimension reservation can afford. It is also the smallest cap at which
140
+ * the decisive evidence source of this repo's own run can be delivered WHOLE:
141
+ * `qa/test-reports/<rid>.md` measured 9,492 bytes there, and `whole` is the
142
+ * delivery rule for every source whose producer publishes no conclusion literal
143
+ * (see `isDelivered`), so a cap under 9,492 makes `problem-resolution` and
144
+ * `no-new-bugs` undeliverable on every run — the always-red gate this module
145
+ * refuses to ship. Enforced in BYTES against the raw buffer, so multi-byte
146
+ * (CJK) content cannot slip past the cap.
147
+ *
148
+ * A source is inlined as `min(bytes, this cap)` — its whole unit — or not at
149
+ * all. So a `TRUNCATED` source block can only ever mean "the FILE is bigger
150
+ * than this cap"; it can no longer mean "the budget ran out while this file was
151
+ * being copied in". That distinction is the point of the all-or-nothing
152
+ * allocator below: the second reading is what let one byte of a 4,226-byte
153
+ * artifact be counted as delivered evidence (F-BLOCK-1BYTE).
154
+ */
155
+ export const MAX_EVIDENCE_BYTES_PER_FILE = MAX_EVIDENCE_BYTES_TOTAL / REQUIRED_DIMENSIONS.length;
156
+ /* ------------------------------------------------------------------ *
157
+ * The anti-starvation reservation (the "floor").
158
+ *
159
+ * The allocator would otherwise be strictly first-come-first-served, and with
160
+ * this repo's own evidence set (`2026-09-12-session-e37ef0`, measured) the
161
+ * first four sources consumed the entire 32 KiB cap — 4 x 8,192 = 32,768, to
162
+ * the byte — before source 5 was even opened. `existing-functionality-intact`
163
+ * is supplied ONLY by `rd/tech-doc.md` (6th), `prd/handoff.md` (9th) and the
164
+ * appended pre/post diff (10th), so that dimension reached the reviewer with
165
+ * zero evidence on every run and its verdict was structurally locked to
166
+ * `inconclusive` no matter how good the work was. A gate that is always red is
167
+ * noise, and an operator trained to ignore noise has no gate at all — the same
168
+ * harm as a gate that never fires, only quieter.
169
+ *
170
+ * So each dimension with at least one readable source on disk gets one
171
+ * reservation, held for the FIRST such source; no source that is not that
172
+ * holder may spend it, and it is released the instant its holder is served.
173
+ * `floorHolders()` below decides who holds what and `collectEvidence()` is the
174
+ * only thing that spends it.
175
+ *
176
+ * WHAT A RESERVATION IS SIZED AT (all-or-nothing): the holder's own UNIT — its
177
+ * whole `min(bytes, perFileCap)` slice — not a fixed number of bytes. Under the
178
+ * all-or-nothing allocator a holder needs its entire unit to be served at all,
179
+ * so reserving less than that would promise something the allocator cannot
180
+ * keep. The promise stays affordable because a holder is at most one per
181
+ * dimension and there are four dimensions:
182
+ *
183
+ * REQUIRED_DIMENSIONS.length x MAX_EVIDENCE_BYTES_PER_FILE
184
+ * = 4 x 10,240 = 40,960 == MAX_EVIDENCE_BYTES_TOTAL = 40,960
185
+ *
186
+ * L4 — those are the DERIVED numbers and the relation is an EQUALITY, not an
187
+ * inequality with room to spare: the reservation spends the entire budget and
188
+ * leaves ZERO slack. This block used to read `= 4 x 8,192 = 32,768 <= 40,960`,
189
+ * which was true of the typed-literal cap and stopped being true the moment the
190
+ * cap was divided out of the total; left as written it read as 8 KiB of
191
+ * headroom that does not exist and would have invited "just raise the cap a
192
+ * little". One byte more per file breaks the invariant outright, which is why
193
+ * the cap is derived rather than typed and why
194
+ * `assertFloorReservationAffordable()` re-checks the derivation (exact integer
195
+ * quotient, product <= total) at the point of use instead of trusting this
196
+ * comment (pinned by a test).
197
+ *
198
+ * That equality is the whole invariant: at every step `budgetLeft` covers the
199
+ * units still owed to the pending holders, so a holder is always served once it
200
+ * is reached, and the invariant cannot silently rot while the three constants
201
+ * keep their published relationship.
202
+ *
203
+ * A reservation is held for a DIMENSION, not for any particular SOURCE:
204
+ * reserving one for `final-review-pre-post-diff` on top of its dimension's
205
+ * would make "a computed baseline the reviewer never saw" UNREACHABLE, and an
206
+ * unreachable branch inside a gate is one nobody can verify. The delivery gates
207
+ * below are the guarantee; see `enforcePrePostDiffAvailability` and
208
+ * `enforceScopeContractDelivery`.
209
+ * ------------------------------------------------------------------ */
116
210
  /* ------------------------------------------------------------------ *
117
211
  * D1 layer 3 — the OUTPUT budget.
118
212
  *
119
- * The input side above was raised to 32 KiB of inlined evidence, but the
213
+ * The input side above was raised (32 KiB then, 40 KiB today) but the
120
214
  * output ceiling stayed hard-coded at 3000 tokens. Measured on this repo's own
121
215
  * run (rid `2026-09-12-codegraph-exclude-integrity`, 9 sources, 32 KiB
122
216
  * inlined, real anthropic provider): 3/3 attempts failed — 2x
@@ -179,9 +273,12 @@ export const MIN_OUTPUT_TOKENS = 3_000;
179
273
  * truncated on 2 of 10 real runs.
180
274
  *
181
275
  * 12288 (12 KiB) is sized so the largest evidence pack the input caps allow
182
- * lands at 23480 (see the formula below) — about 1.8x the largest value ever
183
- * OBSERVED to truncate (13240), which is the margin the observed variance
184
- * asks for. The numbers are in the block comment above.
276
+ * lands at 25528 (see the formula below; the input cap is 40 KiB as of the
277
+ * F-BLOCK re-evaluation, and this floor was checked against the raised cap, not
278
+ * against the 32 KiB the measurements above were taken at — a 12 KiB headroom
279
+ * under a bigger cap is the conservative direction) — about 1.9x the largest
280
+ * value ever OBSERVED to truncate (13240), which is the margin the observed
281
+ * variance asks for. The numbers are in the block comment above.
185
282
  */
186
283
  export const REASONING_HEADROOM_TOKENS = 12 * 1024;
187
284
  /**
@@ -259,76 +356,168 @@ export function resolveOutputBudget(includedEvidenceBytes, env = process.env) {
259
356
  }
260
357
  return Math.min(HARD_MAX_OUTPUT_TOKENS, Math.max(MIN_OUTPUT_TOKENS, parsed));
261
358
  }
262
- function evidenceSourcesFor(rid) {
263
- return [
359
+ /**
360
+ * The one source whose ABSENCE from the delivered prompt is not merely a
361
+ * missing file: it is the only source in the set that is a before/after
362
+ * comparison, and `existing-functionality-intact` is defined by it. Named once
363
+ * so the source builder and the delivery check cannot drift apart.
364
+ */
365
+ const PRE_POST_DIFF_SOURCE_KEY = 'final-review-pre-post-diff';
366
+ /**
367
+ * The approved-scope contract. Also named once: it is the source
368
+ * `functional-completeness` is defined against and the one its delivery gate
369
+ * keys on.
370
+ */
371
+ const SCOPE_CONTRACT_SOURCE_KEY = 'prd-handoff';
372
+ /**
373
+ * The source a DIMENSION'S DELIVERY GATE depends on — i.e. the source the
374
+ * dimension's `pass` may not outlive. `floorHolders` reserves that source's
375
+ * unit, and this table is why the reservation protects the RIGHT source.
376
+ *
377
+ * F1 — the floor used to go to the FIRST source that merely *mentioned* the
378
+ * dimension (`qa-test-report`, index 0, and `rd/tech-doc`, index 6), while both
379
+ * delivery gates keyed on sources at the very END of the order
380
+ * (`prd-handoff`, index 8, and the appended pre/post diff, index 9) that no
381
+ * reservation covered. On this repo's own run the budget was exhausted at
382
+ * source 4, so BOTH gate sources were omitted on every run — necessarily, not
383
+ * accidentally — while 8,192 bytes of floor were spent on `rd/tech-doc`, the
384
+ * one source prompt rule 6 declares insufficient for that same dimension.
385
+ *
386
+ * measured floor reservation after the fix:
387
+ * prd/handoff.md 8,164 + api-diff.txt 4,713 + qa/test-reports 9,492
388
+ * = 22,369 <= MAX_EVIDENCE_BYTES_TOTAL 40,960
389
+ *
390
+ * so all three holders are served WHOLE and the two gate-backed dimensions
391
+ * become deliverable again. A dimension absent from this table has no delivery
392
+ * gate, and its floor stays with the earliest source that supports it.
393
+ */
394
+ const GATE_SOURCE_FOR_DIMENSION = {
395
+ 'functional-completeness': SCOPE_CONTRACT_SOURCE_KEY,
396
+ 'existing-functionality-intact': PRE_POST_DIFF_SOURCE_KEY
397
+ };
398
+ /**
399
+ * The on-disk sources, in allocation order.
400
+ *
401
+ * `prePostDiffAvailable` is the ONE conditional member, and it is opt-in rather
402
+ * than always-present: the `pre-post-diff` producer writes
403
+ * `final-review/api-diff.txt` only when it actually computed a baseline (see
404
+ * `pre-post-diff.ts`), so when the artifact is absent there is nothing behind
405
+ * that source at all. Rendering a permanently-MISSING tenth block would add
406
+ * prompt bytes to every run and hide nothing — the producer's own status block
407
+ * (`renderPrePostDiffStatus`) states the absence and the reason to the reviewer
408
+ * in as many words. When the artifact IS there, it is appended LAST so it
409
+ * cannot displace the byte-allocation order the other nine sources rely on.
410
+ *
411
+ * Being appended last is also why it was the FIRST source the budget dropped —
412
+ * see MAX_EVIDENCE_BYTES_TOTAL for the measurement. Appending it last stays
413
+ * correct for the other nine sources; what changed is that a `pass` on this
414
+ * dimension now requires the block to have been DELIVERED, not merely computed.
415
+ */
416
+ function evidenceSourcesFor(rid, prePostDiffAvailable) {
417
+ const sources = [
264
418
  {
265
419
  key: 'qa-test-report',
266
420
  label: 'QA execution report (per-command pass/fail counts)',
267
421
  segments: ['qa', 'test-reports', `${rid}.md`],
268
- supports: ['functional-completeness', 'problem-resolution', 'no-new-bugs']
422
+ supports: ['functional-completeness', 'problem-resolution', 'no-new-bugs'],
423
+ delivery: { kind: 'whole' }
269
424
  },
270
425
  {
271
426
  key: 'qa-test-cases',
272
427
  label: 'QA test cases (acceptance-criterion to test mapping)',
273
428
  segments: ['qa', 'test-cases', `${rid}.md`],
274
- supports: ['functional-completeness', 'problem-resolution']
429
+ supports: ['functional-completeness', 'problem-resolution'],
430
+ delivery: { kind: 'whole' }
275
431
  },
276
432
  {
277
433
  key: 'qa-security-findings',
278
434
  label: 'QA security findings',
279
435
  segments: ['qa', `security-findings-${rid}.md`],
280
- supports: ['no-new-bugs']
436
+ supports: ['no-new-bugs'],
437
+ delivery: { kind: 'whole' }
281
438
  },
282
439
  {
283
440
  key: 'qa-performance-findings',
284
441
  label: 'QA performance findings',
285
442
  segments: ['qa', `performance-findings-${rid}.md`],
286
- supports: ['no-new-bugs']
443
+ supports: ['no-new-bugs'],
444
+ delivery: { kind: 'whole' }
287
445
  },
288
446
  {
289
447
  key: 'rd-code-review',
290
448
  label: 'RD code review',
291
449
  segments: ['rd', 'code-review.md'],
292
- supports: ['no-new-bugs']
450
+ supports: ['no-new-bugs'],
451
+ delivery: { kind: 'whole' }
293
452
  },
294
453
  {
295
454
  key: 'rd-security-review',
296
455
  label: 'RD security review',
297
456
  segments: ['rd', 'security-review.md'],
298
- supports: ['no-new-bugs']
457
+ supports: ['no-new-bugs'],
458
+ delivery: { kind: 'whole' }
299
459
  },
300
460
  {
301
461
  key: 'rd-tech-doc',
302
462
  label: 'RD tech doc (public surface / design intent)',
303
463
  segments: ['rd', 'tech-doc.md'],
304
- supports: ['existing-functionality-intact']
464
+ supports: ['existing-functionality-intact'],
465
+ delivery: { kind: 'whole' }
305
466
  },
306
467
  {
307
468
  key: 'rd-bug-analysis',
308
469
  label: 'RD bug analysis (original problem statement)',
309
470
  segments: ['rd', 'bug-analysis.md'],
310
- supports: ['problem-resolution']
471
+ supports: ['problem-resolution'],
472
+ delivery: { kind: 'whole' }
311
473
  },
312
474
  {
313
- key: 'prd-handoff',
475
+ key: SCOPE_CONTRACT_SOURCE_KEY,
314
476
  label: 'PRD handoff (approved scope + non-goals)',
315
477
  segments: ['prd', 'handoff.md'],
316
- supports: ['functional-completeness', 'existing-functionality-intact']
478
+ supports: ['functional-completeness', 'existing-functionality-intact'],
479
+ delivery: { kind: 'whole' }
317
480
  }
318
481
  ];
482
+ if (prePostDiffAvailable) {
483
+ sources.push({
484
+ key: PRE_POST_DIFF_SOURCE_KEY,
485
+ label: 'Pre/post structural baseline diff (test surface + public API surface)',
486
+ segments: [...API_DIFF_ARTIFACT_SEGMENTS],
487
+ // The dimension's own contract asks for a `pre-post-diff`; this is the
488
+ // only source in the set that is one.
489
+ supports: ['existing-functionality-intact'],
490
+ // The one source whose producer publishes its conclusion as a literal:
491
+ // the artifact opens with its `VERDICT:` line, so delivery is checkable
492
+ // against the BYTES the reviewer received rather than against the size of
493
+ // the slice it happened to get.
494
+ delivery: { kind: 'conclusion', marker: PRE_POST_DIFF_VERDICT_MARKER }
495
+ });
496
+ }
497
+ return sources;
498
+ }
499
+ /**
500
+ * ENOENT is "no such file"; every other errno is "the file is there and this
501
+ * read failed". Only the first means the source does not exist for this run.
502
+ */
503
+ function classifyReadFailure(error) {
504
+ const code = error?.code;
505
+ return code === 'ENOENT' ? 'missing' : 'unreadable';
319
506
  }
320
507
  /** Pure read: no commands, no git, no test execution — files only. */
321
- function readEvidence(projectRoot, sessionId, rid) {
508
+ function readEvidence(projectRoot, sessionId, rid, prePostDiffAvailable) {
322
509
  const runtimeRoot = join(projectRoot, '.peaks', '_runtime', sessionId);
323
- return evidenceSourcesFor(rid).map(source => {
510
+ return evidenceSourcesFor(rid, prePostDiffAvailable).map(source => {
324
511
  const relativePath = ['.peaks', '_runtime', sessionId, ...source.segments].join('/');
325
512
  const absolutePath = join(runtimeRoot, ...source.segments);
326
513
  let raw = null;
327
514
  let error = '';
515
+ let read = 'ok';
328
516
  try {
329
517
  raw = readFileSync(absolutePath);
330
518
  }
331
519
  catch (err) {
520
+ read = classifyReadFailure(err);
332
521
  error = err instanceof Error ? err.message : String(err);
333
522
  }
334
523
  return {
@@ -336,23 +525,54 @@ function readEvidence(projectRoot, sessionId, rid) {
336
525
  relativePath,
337
526
  absolutePath,
338
527
  raw,
528
+ read,
339
529
  error,
340
530
  blank: raw === null || raw.toString('utf8').trim().length === 0
341
531
  };
342
532
  });
343
533
  }
344
534
  /**
345
- * The one source per dimension that the budget promises to reach: the first
346
- * non-blank candidate that can supply that dimension. A dimension with several
347
- * candidates needs exactly one of them to survive, and the earliest is the one
348
- * the sequential order would have reached anyway — so naming it costs the other
535
+ * The one source per dimension that the budget promises to reach — the source
536
+ * that dimension's DELIVERY GATE depends on where there is one, and otherwise
537
+ * the first non-blank candidate that can supply it. A dimension with several
538
+ * candidates needs exactly one of them to survive, so naming it costs the other
349
539
  * sources nothing they were not already losing.
350
540
  *
351
- * A blank source is deliberately not a holder — see `MIN_EVIDENCE_BYTES_PER_DIMENSION`.
541
+ * F1 — the gate sources used to be eligible for no reservation at all, because
542
+ * they are last in the order and a later source can never be a "first mention".
543
+ * Reserving for them is what makes the two gates' promises reachable rather
544
+ * than merely correct: a gate that always fires because its source never fits
545
+ * is noise, not a gate. A dimension with no gate entry keeps the old rule.
546
+ *
547
+ * A blank source is deliberately not a holder: a floor reserved for a file that
548
+ * carries no evidence is a floor spent on nothing.
352
549
  */
353
550
  function floorHolders(candidates) {
354
551
  const holders = new Map();
355
552
  const covered = new Set();
553
+ const indexByKey = new Map();
554
+ candidates.forEach((candidate, index) => {
555
+ if (candidate.blank)
556
+ return;
557
+ if (!indexByKey.has(candidate.source.key))
558
+ indexByKey.set(candidate.source.key, index);
559
+ });
560
+ // Pass 1 — the source each delivery gate depends on, whether or not it is the
561
+ // first to mention its dimension.
562
+ for (const dimension of REQUIRED_DIMENSIONS) {
563
+ const gateKey = GATE_SOURCE_FOR_DIMENSION[dimension];
564
+ if (gateKey === undefined)
565
+ continue;
566
+ const index = indexByKey.get(gateKey);
567
+ // No such source this run (e.g. no baseline was computed): the dimension
568
+ // falls back to the first-mention rule below.
569
+ if (index === undefined)
570
+ continue;
571
+ covered.add(dimension);
572
+ if (!holders.has(index))
573
+ holders.set(index, dimension);
574
+ }
575
+ // Pass 2 — every dimension without a gate: the first source that mentions it.
356
576
  candidates.forEach((candidate, index) => {
357
577
  if (candidate.blank)
358
578
  return;
@@ -366,9 +586,60 @@ function floorHolders(candidates) {
366
586
  });
367
587
  return holders;
368
588
  }
369
- function collectEvidence(projectRoot, sessionId, rid) {
370
- const candidates = readEvidence(projectRoot, sessionId, rid);
589
+ /**
590
+ * The unit the allocator hands out: one source's WHOLE slice — its bytes up to
591
+ * the per-file cap — or nothing at all.
592
+ *
593
+ * F-BLOCK-1BYTE — the allocator used to take `min(perFileCap, budgetLeft)`, so
594
+ * the last source the budget reached was cut wherever the budget ran out.
595
+ * `status: 'found'` then meant "at least one byte arrived", and every gate that
596
+ * asked "did the reviewer see this source?" was answering a question with a
597
+ * continuum of possible answers. The observed output (QA round 4, 9 x 4,551 B
598
+ * sources, `api-diff.txt` at 4,226 B): the pre/post diff was inlined as **1
599
+ * byte** — the character `#` — while the producer block still said
600
+ * `STATUS: COMPUTED … cite it`, the dimension kept its `pass`/`high`, the
601
+ * artifact was attached as evidence, and the envelope returned `allPass: true`.
602
+ * Each earlier fix had moved the threshold that separated "some bytes" from
603
+ * "the evidence"; this removes the continuum instead, so no predicate can be
604
+ * re-tuned into the same hole a fifth time.
605
+ */
606
+ function sourceUnit(candidate) {
607
+ return candidate.raw === null
608
+ ? 0
609
+ : Math.min(candidate.raw.byteLength, MAX_EVIDENCE_BYTES_PER_FILE);
610
+ }
611
+ /**
612
+ * F5 — the floor promise, checked where it is relied on instead of remembered.
613
+ *
614
+ * The allocator's loop invariant is `budgetLeft >= sum(units of pending
615
+ * holders)`, and its INITIAL condition is
616
+ * `REQUIRED_DIMENSIONS.length x MAX_EVIDENCE_BYTES_PER_FILE <=
617
+ * MAX_EVIDENCE_BYTES_TOTAL`. While that holds, every holder is served when it
618
+ * is reached and the reservation is a guarantee; the moment it fails, all four
619
+ * dimensions are starved in the same run — which surfaces as four independent
620
+ * red gates rather than as one broken constant, and is therefore the kind of
621
+ * breakage nobody diagnoses correctly.
622
+ *
623
+ * The cap is derived from the total so the inequality cannot be typed wrong,
624
+ * and this check covers the two ways a derivation can still go bad: a
625
+ * non-integer quotient (a fifth dimension, say) and a future re-typing of
626
+ * either constant. It is exported because a test asserts it directly.
627
+ */
628
+ export function assertFloorReservationAffordable() {
629
+ if (!Number.isInteger(MAX_EVIDENCE_BYTES_PER_FILE)) {
630
+ throw new Error(`MAX_EVIDENCE_BYTES_PER_FILE is not an integer (${String(MAX_EVIDENCE_BYTES_PER_FILE)} = ${String(MAX_EVIDENCE_BYTES_TOTAL)} / ${String(REQUIRED_DIMENSIONS.length)}): the per-dimension floor reservation is not affordable at a fractional unit.`);
631
+ }
632
+ const owed = REQUIRED_DIMENSIONS.length * MAX_EVIDENCE_BYTES_PER_FILE;
633
+ if (owed > MAX_EVIDENCE_BYTES_TOTAL) {
634
+ throw new Error(`the floor reservation is not affordable: ${String(REQUIRED_DIMENSIONS.length)} dimensions x ${String(MAX_EVIDENCE_BYTES_PER_FILE)} bytes per file = ${String(owed)} exceeds MAX_EVIDENCE_BYTES_TOTAL (${String(MAX_EVIDENCE_BYTES_TOTAL)}). Every dimension would be starved in the same run.`);
635
+ }
636
+ }
637
+ function collectEvidence(projectRoot, sessionId, rid, prePostDiffAvailable) {
638
+ assertFloorReservationAffordable();
639
+ const candidates = readEvidence(projectRoot, sessionId, rid, prePostDiffAvailable);
371
640
  const holders = floorHolders(candidates);
641
+ /** Every source's all-or-nothing unit, computed before any byte is spent. */
642
+ const units = candidates.map(candidate => sourceUnit(candidate));
372
643
  /** Holders that have not been served yet — the floors still owed. */
373
644
  const pending = new Set(holders.keys());
374
645
  const collected = [];
@@ -377,13 +648,19 @@ function collectEvidence(projectRoot, sessionId, rid) {
377
648
  const { source, relativePath, absolutePath, raw } = candidate;
378
649
  const base = { source, relativePath, absolutePath };
379
650
  if (raw === null) {
651
+ // `read` is never `'ok'` here (it is set exactly when the read failed),
652
+ // and `'ok'` is not a status — the branch is what says which of the two
653
+ // failure facts this is.
654
+ const status = candidate.read === 'unreadable' ? 'unreadable' : 'missing';
380
655
  collected.push({
381
656
  ...base,
382
- status: 'missing',
657
+ status,
383
658
  totalBytes: 0,
384
659
  includedBytes: 0,
385
660
  content: '',
386
- reason: candidate.error
661
+ reason: candidate.read === 'missing'
662
+ ? candidate.error
663
+ : `the file exists but could not be read (${candidate.error})`
387
664
  });
388
665
  continue;
389
666
  }
@@ -398,13 +675,19 @@ function collectEvidence(projectRoot, sessionId, rid) {
398
675
  });
399
676
  continue;
400
677
  }
401
- // The floor owed to every dimension still waiting on its holder is spent
402
- // only on that holder. This is the whole fix: a source that needs no help
403
- // can no longer eat the last dimension's only chance at being reviewed.
404
- const holdsFloor = pending.has(index);
405
- const reservedElsewhere = (pending.size - (holdsFloor ? 1 : 0)) * MIN_EVIDENCE_BYTES_PER_DIMENSION;
678
+ const unit = units[index] ?? 0;
679
+ // The floors still owed to dimensions whose holder has not been served are
680
+ // spent only on those holders: a source that needs no help cannot eat the
681
+ // last dimension's only chance at being reviewed. Reserved at the holder's
682
+ // own unit, because under all-or-nothing a partial slice serves nothing.
683
+ const reservedElsewhere = [...pending]
684
+ .filter(holder => holder !== index)
685
+ .reduce((sum, holder) => sum + (units[holder] ?? 0), 0);
406
686
  const allowance = budgetLeft - reservedElsewhere;
407
- if (allowance <= 0) {
687
+ // ALL-OR-NOTHING: this source is inlined as its whole unit or it is
688
+ // reported omitted. There is deliberately no third outcome — a partial
689
+ // slice would re-create the continuum F-BLOCK-1BYTE was found in.
690
+ if (allowance < unit) {
408
691
  const waiting = [...pending]
409
692
  .filter(holder => holder !== index)
410
693
  .map(holder => `${holders.get(holder)} (source ${candidates[holder]?.source.key ?? '?'})`);
@@ -414,21 +697,20 @@ function collectEvidence(projectRoot, sessionId, rid) {
414
697
  totalBytes: raw.byteLength,
415
698
  includedBytes: 0,
416
699
  content: '',
417
- reason: budgetLeft <= 0
418
- ? `total evidence budget (${MAX_EVIDENCE_BYTES_TOTAL} bytes) exhausted before this source`
419
- : `${reservedElsewhere} of the ${budgetLeft} bytes left are reserved for dimension(s) ${waiting.join(', ')} — their only remaining evidence comes later in the source order`
700
+ reason: reservedElsewhere > 0
701
+ ? `${reservedElsewhere} bytes of the budget are reserved for dimension(s) ${waiting.join(', ')} — their only remaining evidence comes later in the source order; this source is inlined WHOLE or not at all and needs ${unit} bytes, while ${budgetLeft} were left (${allowance} after the reservation)`
702
+ : `total evidence budget (${MAX_EVIDENCE_BYTES_TOTAL} bytes) is exhausted — this source is inlined WHOLE or not at all and needs ${unit} bytes, ${budgetLeft} were left`
420
703
  });
421
704
  continue;
422
705
  }
423
- const includedBytes = Math.min(raw.byteLength, MAX_EVIDENCE_BYTES_PER_FILE, allowance);
424
- const content = raw.subarray(0, includedBytes).toString('utf8');
425
- budgetLeft -= includedBytes;
706
+ const content = raw.subarray(0, unit).toString('utf8');
707
+ budgetLeft -= unit;
426
708
  pending.delete(index);
427
709
  collected.push({
428
710
  ...base,
429
711
  status: 'found',
430
712
  totalBytes: raw.byteLength,
431
- includedBytes,
713
+ includedBytes: unit,
432
714
  content,
433
715
  reason: ''
434
716
  });
@@ -446,27 +728,97 @@ function renderEvidenceSection(collected) {
446
728
  : `FOUND at ${item.relativePath} — ${item.totalBytes} bytes`;
447
729
  return `${heading}\n${supports}\nSTATUS: ${status}\n<<<EVIDENCE\n${item.content}\n>>>EVIDENCE`;
448
730
  }
731
+ if (item.status === 'unreadable') {
732
+ // F4 — NOT "missing". The reviewer is told the file is there and the
733
+ // read failed, which is a different instruction to a human than "this
734
+ // run had no PRD phase": one is a fact about the run, the other is a
735
+ // fact about the machine, and only the second is fixable.
736
+ return `${heading}\n${supports}\nSTATUS: UNREADABLE — no evidence available from ${item.relativePath}: ${item.reason}`;
737
+ }
449
738
  return `${heading}\n${supports}\nSTATUS: MISSING (${item.status}) — no evidence available from ${item.relativePath}: ${item.reason}`;
450
739
  })
451
740
  .join('\n\n');
452
741
  }
742
+ /**
743
+ * The producer's own status, told to the reviewer in as many words.
744
+ *
745
+ * It exists because the artifact's ABSENCE is not self-explanatory: a reviewer
746
+ * shown nothing about `existing-functionality-intact` beyond two design-intent
747
+ * documents cannot tell "the diff was clean" from "no diff was ever computed",
748
+ * and the pre-fix run resolved that ambiguity by quietly returning
749
+ * `inconclusive` forever. Naming the reason is what makes the unavailable case
750
+ * a stated fact instead of an invisible one.
751
+ *
752
+ * Kept short on purpose: it rides inside the same prompt that is byte-capped at
753
+ * `MAX_EVIDENCE_BYTES_TOTAL` + scaffolding.
754
+ *
755
+ * F-BLOCK: this block may only say COMPUTED when the artifact it names was
756
+ * actually inlined. Saying "computed — cite it" one line above a source block
757
+ * reading `STATUS: MISSING (omitted)` told the reviewer to cite evidence it had
758
+ * not been given, and that contradiction is what let a `pass` survive a
759
+ * baseline nobody saw.
760
+ */
761
+ function renderPrePostDiffStatus(prePostDiff, delivered) {
762
+ const heading = '## Pre/post baseline diff producer (existing-functionality-intact)';
763
+ if (prePostDiff.status === 'computed' && delivered) {
764
+ return [
765
+ heading,
766
+ `STATUS: COMPUTED — ${prePostDiff.summary}`,
767
+ `Artifact: ${prePostDiff.relativePath} (the next source block). Cite it with evidence kind "pre-post-diff"; it is the structural before/after comparison this dimension's definition asks for.`
768
+ ].join('\n');
769
+ }
770
+ if (prePostDiff.status === 'computed') {
771
+ return [
772
+ heading,
773
+ `STATUS: COMPUTED ON DISK, NOT DELIVERED — the baseline was produced at ${prePostDiff.relativePath}, but its source block could not be inlined into this prompt (the evidence budget omitted it), so it is NOT among the evidence above and you have NOT seen it.`,
774
+ 'Report "existing-functionality-intact" as "inconclusive" and name this in its summary. Do NOT report "pass": the service downgrades a "pass" on this dimension whenever the comparison was not delivered, and a comparison you were not shown is not a comparison that came back clean.'
775
+ ].join('\n');
776
+ }
777
+ const consequence = prePostDiff.inGitWorkTree
778
+ ? 'This project IS a git work tree, so a baseline was expected to be computable: the service treats its absence as a tooling failure and downgrades a "pass" on this dimension to "inconclusive" before any human sees it.'
779
+ : 'This project is not a git work tree, so no baseline can be computed at all. The service downgrades a "pass" on this dimension to "inconclusive" here TOO: "this project keeps no baseline" explains why the evidence is absent, and the absence of a comparison is not a comparison that came back clean.';
780
+ return [
781
+ heading,
782
+ `STATUS: UNAVAILABLE — ${prePostDiff.reason}.`,
783
+ `No pre/post baseline diff exists for this run. Report "existing-functionality-intact" as "inconclusive" with confidence "low" and name the reason above in its summary. Do NOT report "pass": a missing baseline is not evidence of no drift. ${consequence}`
784
+ ].join('\n');
785
+ }
453
786
  const EVIDENCE_RULES = `## Binding rules for the four verdicts
454
787
  1. A dimension may be "pass" ONLY if at least one source in its SUPPORTS list has STATUS: FOUND above, and that source's content actually supports the verdict. The service re-checks this: a "pass" whose supporting sources are all missing/empty/omitted is downgraded to "inconclusive" before any human sees it.
455
788
  2. If the evidence a dimension needs is MISSING, EMPTY, or OMITTED, return "inconclusive" with confidence "low". Do not guess "pass".
456
789
  3. Absence of evidence is not evidence of absence: "no problem found in what I was given" is "inconclusive", never "pass".
457
790
  4. Cite the bracketed source numbers (e.g. "[1]", "[5]") you relied on in each dimension's "evidence[].description"; use an empty list when the verdict is "inconclusive".
458
- 5. "allPass" may be true only when all four verdicts are "pass", and every non-"pass" dimension must be listed in "needsAttention".`;
791
+ 5. "allPass" may be true only when all four verdicts are "pass", and every non-"pass" dimension must be listed in "needsAttention".
792
+ 6. "existing-functionality-intact" may be "pass" ONLY when the pre/post baseline diff block above shows STATUS: FOUND. A design-intent document (RD tech doc, PRD handoff) states what was INTENDED; it is not a before/after comparison of the test surface or the public API surface, and a "pass" resting on one is downgraded to "inconclusive" by the service before any human sees it.
793
+ 7. An "inconclusive" verdict cannot be confident: report it with confidence "low" (or "medium" when the reviewer is sure the evidence is merely incomplete). "high" on "inconclusive" is a contradiction and the service clamps it.
794
+ 8. "functional-completeness" may be "pass" ONLY when the approved-scope contract block (prd-handoff, the source carrying the approved scope and non-goals) is present above with its WHOLE byte count — a STATUS line reading "FOUND at ... — N bytes", not a truncated or an omitted one. A passing test report shows that something was built; only the contract shows that what was built IS the approved scope. The service re-checks this: it downgrades a "pass" on this dimension whenever that contract exists for the run but was not delivered in full.`;
459
795
  /**
460
- * Which dimensions had at least one FOUND source. A dimension absent from this
461
- * set has no evidence at all behind it.
796
+ * Which dimensions the reviewer was actually given evidence FOR. A dimension
797
+ * absent from this set has no delivered evidence behind it.
798
+ *
799
+ * F3 / 1.3 — this used to read `status !== 'found'`, i.e. "a source that
800
+ * mentions this dimension was inlined (at least a byte of it)". Two things were
801
+ * wrong with that, and both are the same thing: it asked about the SOURCE (was
802
+ * it included) instead of about the DIMENSION (was its evidence seen), and it
803
+ * answered with whatever the allocator's unit happened to be. Measured on this
804
+ * repo's own run, eight of the ten sources are larger than the per-file cap, so
805
+ * every dimension's evidence set was the cap-sized HEAD of a document — and a
806
+ * head can be a table of contents while the findings live past the cut (QA
807
+ * reproduced exactly that: one 20,545-byte source whose first 10,240 bytes are
808
+ * front matter, four `pass`/`high` verdicts). "Some bytes of a source that
809
+ * lists this dimension" is not evidence; the source's conclusion is.
810
+ *
811
+ * It now delegates, one dimension at a time, to the module's single delivery
812
+ * judgement — so a source that cannot be delivered for a dimension cannot back
813
+ * a pass on it either, here or anywhere else.
462
814
  */
463
815
  function dimensionsWithEvidence(collected) {
464
816
  const available = new Set();
465
817
  for (const item of collected) {
466
- if (item.status !== 'found')
467
- continue;
468
- for (const dimension of item.source.supports)
469
- available.add(dimension);
818
+ for (const dimension of item.source.supports) {
819
+ if (isDelivered(item, dimension))
820
+ available.add(dimension);
821
+ }
470
822
  }
471
823
  return available;
472
824
  }
@@ -494,6 +846,444 @@ function enforceEvidenceBackedVerdicts(dimensions, evidenceAvailableFor) {
494
846
  return downgraded;
495
847
  });
496
848
  }
849
+ /**
850
+ * DELIVERED — the module's ONE delivery judgement.
851
+ *
852
+ * A source is delivered TO A DIMENSION when the bytes the reviewer actually
853
+ * received carry the source's conclusion FOR THAT DIMENSION. Nothing else is
854
+ * consulted: not the file's existence, not `status`, not a byte count. The
855
+ * question it answers is "did the reviewer see this dimension's supporting
856
+ * content?", never "was this source included?".
857
+ *
858
+ * Why one predicate and not four. Every one of the six holes this primitive has
859
+ * shipped was the same shape: a local predicate that was satisfied by a PROXY
860
+ * for delivery — the substring's presence, the stub's usefulness, "this is not
861
+ * a git repo", `status === 'computed'`, `status === 'found'`, and finally
862
+ * `totalBytes === 0` / `status !== 'found'` in two more places. Each fix was
863
+ * correct where it was applied, and the shape reappeared next door because
864
+ * "was it delivered?" had as many answers as there were call sites. There is
865
+ * now one: this function, reading one declaration per source
866
+ * (`EvidenceSource.delivery`).
867
+ *
868
+ * The two rules are the two ways a conclusion can be said to have arrived:
869
+ * - `whole` — the whole document arrived. A truncated document is not a
870
+ * weaker version of itself; the conclusion may be in the part that was cut.
871
+ * This is the rule everywhere the producer is another role's skill and
872
+ * publishes no conclusion literal: the module may not guess where a
873
+ * document's conclusion lives.
874
+ * - `conclusion(marker)` — the literal that IS the conclusion is among the
875
+ * bytes received.
876
+ *
877
+ * F-BLOCK-1BYTE — `found` was read as "delivered", and `found` was defined by a
878
+ * byte COUNT, so 1 byte of a 4,226-byte artifact passed for a delivered
879
+ * comparison. A count can be satisfied by a fragment; a conclusion cannot.
880
+ *
881
+ * A dimension the source does not SUPPORT is never delivered: `supports` is a
882
+ * claim about what a document can evidence, and reading it as "the reviewer saw
883
+ * this dimension's evidence" is exactly the presence-as-substance error. The
884
+ * check lives here so no caller can skip it.
885
+ */
886
+ function isDelivered(item, dimension) {
887
+ if (!item.source.supports.includes(dimension))
888
+ return false;
889
+ // No bytes arrived at all: missing, empty, unreadable, omitted.
890
+ if (item.status !== 'found')
891
+ return false;
892
+ const rule = item.source.delivery;
893
+ switch (rule.kind) {
894
+ case 'whole':
895
+ return item.totalBytes > 0 && item.includedBytes === item.totalBytes;
896
+ case 'conclusion':
897
+ return item.content.includes(rule.marker);
898
+ }
899
+ }
900
+ /**
901
+ * Was the pre/post-diff source delivered to the dimension it exists for?
902
+ *
903
+ * Kept as a name because three call sites share it and the reason they share it
904
+ * is the point: the gate, the artifact attachment and the evidence set must all
905
+ * mean the same thing by "the reviewer saw the comparison". It reads the ONE
906
+ * predicate; it does not re-decide anything.
907
+ */
908
+ function prePostDiffDelivered(collected) {
909
+ const block = collected.find(item => item.source.key === PRE_POST_DIFF_SOURCE_KEY);
910
+ return block !== undefined && isDelivered(block, 'existing-functionality-intact');
911
+ }
912
+ /**
913
+ * Can this source EVER be delivered, whatever the budget?
914
+ *
915
+ * `whole` is the only rule with a structural ceiling. Its test is
916
+ * `includedBytes === totalBytes`, and the allocator can only ever inline
917
+ * `min(bytes, MAX_EVIDENCE_BYTES_PER_FILE)` — so a `whole`-rule source LARGER
918
+ * than the per-file cap can never be delivered: not on this run, not on any
919
+ * run, not with any budget, because the budget is not what cuts it. Raising
920
+ * `MAX_EVIDENCE_BYTES_TOTAL` changes nothing either, since the per-file cap is
921
+ * derived from it and the reservation spends all of it.
922
+ *
923
+ * (`conclusion` sources are deliberately NOT covered. A marker missing from the
924
+ * first cap-sized bytes of an oversize artifact might simply live past the cut,
925
+ * so "undeliverable forever" is not a claim this module can make about them —
926
+ * and an over-claiming check is the failure mode this file keeps closing.)
927
+ */
928
+ function isStructurallyUndeliverable(item) {
929
+ if (item.source.delivery.kind !== 'whole')
930
+ return false;
931
+ return item.totalBytes > MAX_EVIDENCE_BYTES_PER_FILE;
932
+ }
933
+ /**
934
+ * H2 — every required dimension whose verdict is locked to `inconclusive` by
935
+ * BYTE ARITHMETIC rather than by the reviewer's judgement.
936
+ *
937
+ * The defect this exists for: `qa/test-reports/<rid>.md` measured 9,492 bytes
938
+ * on this repo's own run and is the ONLY source on disk for `problem-resolution`
939
+ * and `no-new-bugs` (the other sources that support them are absent), against a
940
+ * per-file cap of 10,240 — a 748-byte margin on a file that is REWRITTEN every
941
+ * round and only grows. The moment it crosses 10,240 both dimensions go
942
+ * permanently red, and NOTHING said so: `assertFloorReservationAffordable()`
943
+ * only checks the constant-level relation (`4 x cap <= total`), never whether
944
+ * any actual source fits the cap it must live under, so the red handoff read as
945
+ * "the reviewer was unsure" instead of "no evidence can ever reach the
946
+ * reviewer". A gate that is always red and never explains itself is noise, and
947
+ * this is the same "always red" harm the floor reservation was built to remove
948
+ * — one layer down.
949
+ *
950
+ * The conditions, all three of which must hold, are chosen so the report cannot
951
+ * be noise:
952
+ * 1. NOTHING on disk can back the dimension (no `isDelivered`), and
953
+ * 2. there IS something on disk to deliver (a source that was never written
954
+ * is not a delivery failure — same reasoning as the scope-contract gate's
955
+ * `missing` exemption: a run with no QA phase has no report to lose), and
956
+ * 3. EVERY one of those on-disk sources is structurally undeliverable — a
957
+ * single source that merely did not fit TODAY (budget-exhausted `omitted`)
958
+ * is a different, self-correcting state and is left to the allocator.
959
+ *
960
+ * The report is consumed by the prompt (stated to the reviewer), by the
961
+ * envelope (a marker on each dimension's summary) and by `needsAttention`
962
+ * (which also clears `allPass`) — see `enforceDeliveryReachability` and
963
+ * `renderDeliveryReachabilityStatus`. It is a LOUD STATEMENT, not a silent
964
+ * downgrade, and it is deliberately not a throw: a crash would destroy the
965
+ * evidence for the three dimensions that ARE deliverable, and the honest fact
966
+ * here is per-dimension, so it is reported per-dimension.
967
+ */
968
+ export function undeliverableDimensions(collected) {
969
+ const report = [];
970
+ for (const dimension of REQUIRED_DIMENSIONS) {
971
+ const supporting = collected.filter(item => item.source.supports.includes(dimension));
972
+ if (supporting.some(item => isDelivered(item, dimension)))
973
+ continue;
974
+ const onDisk = supporting.filter(item => item.totalBytes > 0);
975
+ if (onDisk.length === 0)
976
+ continue;
977
+ if (!onDisk.every(isStructurallyUndeliverable))
978
+ continue;
979
+ report.push({
980
+ dimension,
981
+ sources: onDisk.map(item => ({
982
+ key: item.source.key,
983
+ relativePath: item.relativePath,
984
+ totalBytes: item.totalBytes
985
+ }))
986
+ });
987
+ }
988
+ return report;
989
+ }
990
+ /**
991
+ * H2 — tell the reviewer, in the prompt, that a dimension has no deliverable
992
+ * source at all. Emitted only when there is something to say, so the byte
993
+ * budget is not spent on an empty section every run.
994
+ *
995
+ * It reads like `renderPrePostDiffStatus` on purpose: the reviewer is told the
996
+ * FACT and what to answer with, so a permanently red dimension arrives at the
997
+ * human with its cause attached instead of as an unexplained `inconclusive`.
998
+ */
999
+ function renderDeliveryReachabilityStatus(report) {
1000
+ if (report.length === 0)
1001
+ return '';
1002
+ // Kept to one line per source and one line per dimension: this block rides
1003
+ // inside the same byte-capped prompt as the evidence it describes, and the
1004
+ // evidence blocks above already carry each source's path and status.
1005
+ const lines = report.map(entry => {
1006
+ const sources = entry.sources
1007
+ .map(item => `${item.key} (${String(item.totalBytes)} bytes)`)
1008
+ .join(', ');
1009
+ return ` - ${entry.dimension}: NO deliverable source. ${sources} exceeds the per-file cap of ${String(MAX_EVIDENCE_BYTES_PER_FILE)} bytes, and a source is inlined WHOLE or not at all.`;
1010
+ });
1011
+ return [
1012
+ '## Evidence delivery reachability (structural)',
1013
+ `The allocator inlines at most ${String(MAX_EVIDENCE_BYTES_PER_FILE)} bytes of any one source, WHOLE or not at all, and a source delivered under the "whole" rule is delivered only when the reviewer received ALL of it. For the dimension(s) below, every source on disk that supports it is larger than that cap, so no source CAN be delivered — not on this run and not on any run, whatever the budget.`,
1014
+ ...lines,
1015
+ 'Report each of them as "inconclusive" with confidence "low" and name this reason in its summary. Do NOT report "pass": the service re-checks it, and a "pass" here is downgraded. This is a STRUCTURAL impossibility, not a judgement you are being asked to make — say so rather than reporting an unexplained "inconclusive".'
1016
+ ].join('\n');
1017
+ }
1018
+ /**
1019
+ * H2 — the envelope half of the same fact. A dimension named by
1020
+ * `undeliverableDimensions()` cannot be `pass` (no source on disk can carry its
1021
+ * conclusion), so this is not where the verdict is decided — that is
1022
+ * `enforceEvidenceBackedVerdicts`, and this gate re-applies it so the property
1023
+ * does not depend on that gate's order. What this adds is the REASON: the
1024
+ * dimension's own summary says it is red by construction. Because the verdict
1025
+ * it leaves behind is non-`pass`, the dimension also reaches `needsAttention`
1026
+ * (and clears `allPass`) through the ordinary verdict route, which is the field
1027
+ * the CLI envelope prints — so the human is told WHY instead of being left to
1028
+ * read a permanent red as the reviewer's uncertainty.
1029
+ */
1030
+ function enforceDeliveryReachability(dimensions, report) {
1031
+ if (report.length === 0)
1032
+ return dimensions;
1033
+ const byDimension = new Map(report.map(entry => [entry.dimension, entry]));
1034
+ return dimensions.map(dimension => {
1035
+ const entry = byDimension.get(dimension.dimension);
1036
+ if (entry === undefined)
1037
+ return dimension;
1038
+ // The ENVELOPE is not byte-capped, so it names the path as well as the
1039
+ // size: the path is what a human has to act on to fix it.
1040
+ const detail = entry.sources
1041
+ .map(item => `${item.key} at ${item.relativePath} (${String(item.totalBytes)} bytes)`)
1042
+ .join(', ');
1043
+ const marker = `[delivery-reachability: ${dimension.verdict === 'pass'
1044
+ ? 'verdict downgraded from "pass" to "inconclusive"'
1045
+ : `gate ran; verdict "${dimension.verdict}" is already non-"pass" and is left unchanged`} — EVERY source on disk that supports "${dimension.dimension}" (${detail}) is larger than the per-file delivery cap of ${String(MAX_EVIDENCE_BYTES_PER_FILE)} bytes, and a source is inlined WHOLE or not at all, so no evidence for this dimension can ever reach the reviewer. The dimension is red by byte arithmetic, not by the reviewer's judgement, and it is listed in needsAttention for that reason.]`;
1046
+ const annotated = {
1047
+ ...dimension,
1048
+ summary: `${dimension.summary} ${marker}`
1049
+ };
1050
+ if (dimension.verdict !== 'pass')
1051
+ return annotated;
1052
+ return { ...annotated, verdict: 'inconclusive', confidence: 'low' };
1053
+ });
1054
+ }
1055
+ /**
1056
+ * The pre/post-diff half of the honesty rule: a `pass` on
1057
+ * `existing-functionality-intact` may not outlive the baseline it claims.
1058
+ *
1059
+ * It fires whenever no baseline was DELIVERED to the reviewer, for EVERY reason:
1060
+ * no base ref resolvable, a base that resolves to HEAD itself, a project that is
1061
+ * not a git work tree at all, or a baseline that WAS computed but whose source
1062
+ * block never reached the prompt (the budget omitted it). The first three are
1063
+ * causes of a missing artifact; the last is a missing DELIVERY of an artifact
1064
+ * that exists. None of the four is a licence to trust the claim: they are the
1065
+ * CAUSE of the missing evidence, and a comparison the reviewer never saw is the
1066
+ * absence of an answer, never an answer that came back clean.
1067
+ *
1068
+ * The earlier version exempted the non-git case, reasoning that "a non-git
1069
+ * project would otherwise be permanently red". That reasoning was wrong in the
1070
+ * one way this whole primitive exists to prevent: the permanently red verdict is
1071
+ * the HONEST one (nothing was ever compared), while green-with-no-evidence is a
1072
+ * forged clean handoff. Worse, it pierced the structural gate that
1073
+ * `enforceEvidenceBackedVerdicts` exists to be — a `pass` whose supporting
1074
+ * evidence is entirely absent is downgraded there, and it must not be let
1075
+ * through by supplying a REASON for the absence.
1076
+ *
1077
+ * The version that followed it keyed on `prePostDiff.status === 'computed'` —
1078
+ * "a baseline exists on disk" read as "the reviewer saw one". Those two facts
1079
+ * separate the moment the budget omits the block, and the observed output was
1080
+ * self-contradicting: the producer's own block said `STATUS: COMPUTED` one line
1081
+ * above the source block's `STATUS: MISSING (omitted)`, the model was handed a
1082
+ * `pass` it could not have justified, and the service attached the artifact to
1083
+ * the dimension as evidence on top. `delivered` is that missing term.
1084
+ *
1085
+ * The gate also leaves its marker on the dimension whenever it fires, even if
1086
+ * that dimension is already non-`pass`: with both gates firing, the earlier
1087
+ * version's early return left only the OTHER gate's marker behind, so the
1088
+ * envelope could not tell an operator that this gate had run at all.
1089
+ */
1090
+ function enforcePrePostDiffAvailability(dimensions, prePostDiff, collected) {
1091
+ if (prePostDiff.status === 'computed' && prePostDiffDelivered(collected))
1092
+ return dimensions;
1093
+ const why = prePostDiff.status !== 'computed'
1094
+ ? prePostDiff.inGitWorkTree
1095
+ ? 'this is a git work tree, but no pre/post baseline diff could be produced for this run'
1096
+ : 'this project is not a git work tree, so no pre/post baseline diff can exist for it'
1097
+ : 'a baseline WAS computed for this run, but its source block was not delivered into the reviewer prompt (the evidence budget omitted it), so the reviewer never saw the comparison it would have to rest on';
1098
+ const reason = prePostDiff.status === 'computed'
1099
+ ? 'the artifact exists on disk but did not reach the reviewer'
1100
+ : prePostDiff.reason;
1101
+ return dimensions.map(dimension => {
1102
+ if (dimension.dimension !== 'existing-functionality-intact')
1103
+ return dimension;
1104
+ const marker = `[pre-post-diff-gate: ${dimension.verdict === 'pass'
1105
+ ? 'verdict downgraded from "pass" to "inconclusive"'
1106
+ : `gate ran; verdict "${dimension.verdict}" is already non-"pass" and is left unchanged`} — ${why}. Reason: ${reason}]`;
1107
+ const annotated = {
1108
+ ...dimension,
1109
+ summary: `${dimension.summary} ${marker}`
1110
+ };
1111
+ if (dimension.verdict !== 'pass')
1112
+ return annotated;
1113
+ return { ...annotated, verdict: 'inconclusive', confidence: 'low' };
1114
+ });
1115
+ }
1116
+ /**
1117
+ * F-NIT — `inconclusive` is the one verdict that cannot carry `high`
1118
+ * confidence. "I am highly confident that I could not tell" is the schema's
1119
+ * one self-contradicting combination, and it reads as a strong statement to the
1120
+ * human this envelope is handed to. Both service gates that produce an
1121
+ * `inconclusive` a reviewer did not write already write `low`; this closes the
1122
+ * same hole for the ones the reviewer DID write, clamping to `medium` (the
1123
+ * upper bound `references/4-dimensions.md` documents for this verdict) and
1124
+ * leaving a marker so the clamp is auditable rather than silent.
1125
+ *
1126
+ * `fail` and `pass` are untouched: their confidence says something real.
1127
+ */
1128
+ function clampInconclusiveConfidence(dimensions) {
1129
+ return dimensions.map(dimension => {
1130
+ if (dimension.verdict !== 'inconclusive' || dimension.confidence !== 'high')
1131
+ return dimension;
1132
+ return {
1133
+ ...dimension,
1134
+ confidence: 'medium',
1135
+ summary: `${dimension.summary} [confidence-gate: confidence clamped from "high" to "medium" — an "inconclusive" verdict cannot be highly confident that it could not tell.]`
1136
+ };
1137
+ });
1138
+ }
1139
+ /**
1140
+ * The source whose absence from the delivered prompt is a DELIVERY failure for
1141
+ * `functional-completeness`: the approved-scope contract.
1142
+ *
1143
+ * F-SAME-SHAPE — the hole the pre/post-diff gate closes for
1144
+ * `existing-functionality-intact` was still open one dimension over. That
1145
+ * dimension's rule is "a `pass` may not outlive the source it rests on", and
1146
+ * `functional-completeness` is DEFINED against the approved scope and its
1147
+ * non-goals — `prd/handoff.md` — but its `pass` only requires SOME source in its
1148
+ * SUPPORTS list to be FOUND, and `qa-test-report` (the first source in the
1149
+ * order, so the one the budget can never starve) also supports it. Measured on
1150
+ * both saturated fixtures: `prd/handoff.md` was inlined with zero bytes on every
1151
+ * run while the dimension still came back `pass`/`high` — the contract that
1152
+ * says what "complete" meant was never shown to the reviewer judging
1153
+ * completeness. Exactly the forged-clean-handoff shape, one dimension over.
1154
+ *
1155
+ * The delivery test for this source is the module's `whole` rule (`isDelivered`),
1156
+ * not "some bytes arrived": the contract is prose whose point is the part a
1157
+ * truncation would cut (scope then non-goals), so a partial slice is not a
1158
+ * weaker version of the contract — it is a different document, and the reviewer
1159
+ * would be judging "complete" against a scope list that stops mid-sentence.
1160
+ *
1161
+ * The gate deliberately does NOT fire when the contract is `missing` — ENOENT,
1162
+ * i.e. the file is absent from the project, whether because the workflow has no
1163
+ * PRD phase or because it never wrote one. A workflow with no PRD phase has no
1164
+ * contract to lose, and reddening that dimension forever would be the "gate
1165
+ * that is always red is noise" failure this primitive already names.
1166
+ *
1167
+ * F4 — what the gate fires on is the opposite fact: the contract IS there for
1168
+ * this run and did not reach the reviewer. `totalBytes === 0` used to stand in
1169
+ * for "not there", and it was wider than the comment above it: a 0-byte
1170
+ * `prd/handoff.md` is `empty`, not `missing` — the file exists, the PRD phase
1171
+ * ran, and the reviewer received no contract at all — yet the early return
1172
+ * skipped the gate and let `functional-completeness` come back `pass`/`high`.
1173
+ * The same line collapsed `unreadable` (EACCES / EBUSY: the contract exists and
1174
+ * this process could not open it) into "no PRD phase". The test is now the
1175
+ * STATUS the read phase produced, so "there is nothing to deliver" and "there
1176
+ * is something to deliver and it did not arrive" cannot be the same branch.
1177
+ */
1178
+ function enforceScopeContractDelivery(dimensions, collected) {
1179
+ const block = collected.find(item => item.source.key === SCOPE_CONTRACT_SOURCE_KEY);
1180
+ if (block === undefined || block.status === 'missing')
1181
+ return dimensions;
1182
+ if (isDelivered(block, 'functional-completeness'))
1183
+ return dimensions;
1184
+ return dimensions.map(dimension => {
1185
+ if (dimension.dimension !== 'functional-completeness')
1186
+ return dimension;
1187
+ const marker = `[scope-contract-gate: ${dimension.verdict === 'pass'
1188
+ ? 'verdict downgraded from "pass" to "inconclusive"'
1189
+ : `gate ran; verdict "${dimension.verdict}" is already non-"pass" and is left unchanged`} — the approved-scope contract (${block.relativePath}) exists for this run, but it did not reach the reviewer in full (${block.includedBytes} of ${block.totalBytes} bytes, status "${block.status}"), so "functional-completeness" was judged without the scope and non-goals it is defined against. Reason: ${block.reason || 'inlined only in part'}]`;
1190
+ const annotated = {
1191
+ ...dimension,
1192
+ summary: `${dimension.summary} ${marker}`
1193
+ };
1194
+ if (dimension.verdict !== 'pass')
1195
+ return annotated;
1196
+ return { ...annotated, verdict: 'inconclusive', confidence: 'low' };
1197
+ });
1198
+ }
1199
+ /**
1200
+ * Attach the computed diff to the dimension as a first-class `pre-post-diff`
1201
+ * `EvidenceItem`.
1202
+ *
1203
+ * The service does this rather than trusting the reviewer to cite the source:
1204
+ * the dimension's contract NAMES this evidence kind and this artifact path, and
1205
+ * a machine-produced fact that only appears when an LLM remembers to type it is
1206
+ * not a guarantee. Appending cannot upgrade a verdict — the gate above has
1207
+ * already run, and this only makes the artifact the verdict rests on auditable.
1208
+ *
1209
+ * It follows the same delivery rule as the gate — it asks the SAME function, on
1210
+ * purpose: attaching the artifact to a dimension whose reviewer never received
1211
+ * that block would re-create the very contradiction this layer exists to remove
1212
+ * — an envelope claiming a `pre-post-diff` evidence item next to an
1213
+ * `inconclusive` verdict that says the comparison was never seen.
1214
+ */
1215
+ function attachPrePostDiffEvidence(dimensions, prePostDiff, collected) {
1216
+ if (prePostDiff.status !== 'computed' || !prePostDiffDelivered(collected))
1217
+ return dimensions;
1218
+ const item = {
1219
+ kind: 'pre-post-diff',
1220
+ description: prePostDiff.summary,
1221
+ artifact: prePostDiff.relativePath
1222
+ };
1223
+ return dimensions.map(dimension => {
1224
+ if (dimension.dimension !== 'existing-functionality-intact')
1225
+ return dimension;
1226
+ const alreadyPresent = dimension.evidence.some(existing => existing.kind === 'pre-post-diff' && existing.artifact === item.artifact);
1227
+ if (alreadyPresent)
1228
+ return dimension;
1229
+ return { ...dimension, evidence: [...dimension.evidence, item] };
1230
+ });
1231
+ }
1232
+ /**
1233
+ * F2 — read what the delivered conclusion SAYS, and make a detected drift
1234
+ * impossible to hand over as "nothing needs attention".
1235
+ *
1236
+ * The layer that made delivery a machine fact stopped one step short: it
1237
+ * verified that the reviewer received the `VERDICT:` line and then threw the
1238
+ * line away, keeping only a boolean. Measured (QA round 5, one export removed,
1239
+ * the model replying 4/4 `pass`):
1240
+ *
1241
+ * VERDICT: STRUCTURAL DRIFT DETECTED — ... 1 export name(s).
1242
+ * existing-functionality-intact: pass/high
1243
+ * allPass: true | needsAttention: []
1244
+ *
1245
+ * — an envelope that says "clean handoff" directly above the evidence it
1246
+ * attached itself, whose first line says a removal was detected. That is the
1247
+ * same shape as every other hole here: a check that ran, whose RESULT was never
1248
+ * consumed, so the check's presence was mistaken for its verdict.
1249
+ *
1250
+ * The verdict is deliberately NOT forced to `fail`: a removal can be authorized
1251
+ * by the approved scope, and that judgement belongs to the reviewer and to the
1252
+ * human. What is refused is SILENCE — a drift the service detected is a
1253
+ * dimension a human has to look at, so it is named in `needsAttention` (which
1254
+ * also clears `allPass`: a handoff with an open question is not a clean one)
1255
+ * and its dimension says so in its own summary.
1256
+ *
1257
+ * `indeterminate` counts too. A delivered conclusion this service cannot
1258
+ * classify is not "no drift" — it is a conclusion nobody read, which is the
1259
+ * defect this function exists to close, so it is surfaced rather than dropped.
1260
+ */
1261
+ function enforceStructuralDriftAttention(dimensions, deliveredConclusion) {
1262
+ // `null` is "no conclusion was DELIVERED" — the gate above owns that case. It
1263
+ // must not be folded into `indeterminate`: a project with no baseline has not
1264
+ // delivered an unreadable conclusion, it has delivered none, and saying
1265
+ // otherwise would put a false "the comparison was never read" marker on every
1266
+ // non-git run.
1267
+ if (deliveredConclusion === null)
1268
+ return { dimensions, mustAttend: false };
1269
+ if (deliveredConclusion === 'no-drift' || deliveredConclusion === 'additions-only') {
1270
+ return { dimensions, mustAttend: false };
1271
+ }
1272
+ const what = deliveredConclusion === 'drift-detected'
1273
+ ? 'the delivered pre/post baseline diff reports STRUCTURAL DRIFT DETECTED'
1274
+ : 'the delivered pre/post baseline diff carries a VERDICT line this service cannot classify, so the comparison was never actually read';
1275
+ return {
1276
+ dimensions: dimensions.map(dimension => {
1277
+ if (dimension.dimension !== 'existing-functionality-intact')
1278
+ return dimension;
1279
+ return {
1280
+ ...dimension,
1281
+ summary: `${dimension.summary} [pre-post-diff-drift-gate: ${what}. The removal may well be authorized by the approved scope — that is the reviewer's and the human's call — but a detected drift is never "nothing needs attention": this dimension is listed in needsAttention and allPass is false.]`
1282
+ };
1283
+ }),
1284
+ mustAttend: true
1285
+ };
1286
+ }
497
1287
  /**
498
1288
  * True when the reply ends INSIDE a JSON string or with brackets still open —
499
1289
  * i.e. it was cut off mid-structure rather than being malformed. Distinguishing
@@ -586,14 +1376,24 @@ async function callReviewer(runner, userPrompt, budget) {
586
1376
  * true unless the model also said true and no dimension is non-`pass`, and a
587
1377
  * dimension the model itself flagged is never dropped from `needsAttention`.
588
1378
  */
589
- function summarizeVerdicts(dimensions, modelFlags) {
1379
+ function summarizeVerdicts(dimensions, modelFlags,
1380
+ /**
1381
+ * Dimensions the SERVICE must flag whatever the model said — today, the one
1382
+ * F2 names when the delivered baseline reports structural drift. A handoff
1383
+ * with a machine-detected drift in it is not clean, so this also clears
1384
+ * `allPass`, exactly as a non-`pass` verdict does.
1385
+ */
1386
+ mustAttend = []) {
590
1387
  const nonPass = dimensions.filter(d => d.verdict !== 'pass').map(d => d.dimension);
591
1388
  const flaggedByModel = Array.isArray(modelFlags.needsAttention)
592
1389
  ? modelFlags.needsAttention
593
1390
  : [];
594
1391
  return {
595
- allPass: modelFlags.allPass !== false && dimensions.length > 0 && nonPass.length === 0,
596
- needsAttention: [...new Set([...flaggedByModel, ...nonPass])]
1392
+ allPass: modelFlags.allPass !== false &&
1393
+ dimensions.length > 0 &&
1394
+ nonPass.length === 0 &&
1395
+ mustAttend.length === 0,
1396
+ needsAttention: [...new Set([...flaggedByModel, ...nonPass, ...mustAttend])]
597
1397
  };
598
1398
  }
599
1399
  export async function prepareFinalReview(rid, opts) {
@@ -605,7 +1405,27 @@ export async function prepareFinalReview(rid, opts) {
605
1405
  catch (err) {
606
1406
  throw new Error(`Cannot read approved goal from ${auditGoalPath}: ${err.message}`);
607
1407
  }
608
- const evidence = collectEvidence(opts.projectRoot, opts.sessionId, rid);
1408
+ // The pre/post baseline diff is produced BEFORE the read phase: it is the one
1409
+ // step in this service that runs a command (`git`, read-only) and writes a
1410
+ // file, and it has to finish first because its artifact is also an evidence
1411
+ // source below. Everything after this line is still pure file reading.
1412
+ const prePostDiff = producePrePostDiff({
1413
+ projectRoot: opts.projectRoot,
1414
+ sessionId: opts.sessionId,
1415
+ ...(opts.baseRef === undefined ? {} : { baseRef: opts.baseRef })
1416
+ });
1417
+ const evidence = collectEvidence(opts.projectRoot, opts.sessionId, rid, prePostDiff.status === 'computed');
1418
+ // F-BLOCK: the fourth dimension needs the baseline to have been DELIVERED,
1419
+ // not merely computed, so the delivery fact is measured on the collected
1420
+ // evidence (what the prompt carries) rather than read off the producer's
1421
+ // status (what exists on disk).
1422
+ const ppdDelivered = prePostDiffDelivered(evidence);
1423
+ // H2: a dimension whose every on-disk source is structurally undeliverable is
1424
+ // a fact about the EVIDENCE, so it is measured where the evidence is and
1425
+ // stated in both the prompt and the envelope — an always-red dimension that
1426
+ // explains itself is a finding; one that does not is noise.
1427
+ const undeliverable = undeliverableDimensions(evidence);
1428
+ const reachabilityStatus = renderDeliveryReachabilityStatus(undeliverable);
609
1429
  const userPrompt = [
610
1430
  `Approved goal's success criteria: ${JSON.stringify(approvedGoal.successCriteria)}`,
611
1431
  '',
@@ -614,6 +1434,9 @@ export async function prepareFinalReview(rid, opts) {
614
1434
  '',
615
1435
  renderEvidenceSection(evidence),
616
1436
  '',
1437
+ renderPrePostDiffStatus(prePostDiff, ppdDelivered),
1438
+ '',
1439
+ ...(reachabilityStatus === '' ? [] : [reachabilityStatus, '']),
617
1440
  EVIDENCE_RULES,
618
1441
  '',
619
1442
  'Prepare the 4-dim review evidence.'
@@ -659,10 +1482,24 @@ export async function prepareFinalReview(rid, opts) {
659
1482
  ? ` — the provider hit the output budget and the reply was cut short (${describeOutputBudget(maxTokens, response.tokens.output, response.output.length)}); raise the budget instead of retrying blindly.`
660
1483
  : ''}`);
661
1484
  }
1485
+ // F2: what the DELIVERED conclusion says. Read before the verdicts are
1486
+ // assembled, because a detected drift has to reach `needsAttention` whatever
1487
+ // the reviewer answered. `null` — nothing delivered — is NOT
1488
+ // `indeterminate`: see `enforceStructuralDriftAttention`.
1489
+ const ppdBlock = evidence.find(item => item.source.key === PRE_POST_DIFF_SOURCE_KEY);
1490
+ const ppdConclusion = ppdBlock !== undefined && ppdDelivered ? classifyPrePostDiffVerdict(ppdBlock.content) : null;
662
1491
  // D1: the verdicts must rest on evidence that actually existed, and the
663
- // derived summary flags must match the verdicts — see the two helpers above.
664
- const dimensions = enforceEvidenceBackedVerdicts(output.dimensions, dimensionsWithEvidence(evidence));
665
- const { allPass, needsAttention } = summarizeVerdicts(dimensions, output);
1492
+ // derived summary flags must match the verdicts — see the helpers above.
1493
+ const gated = enforceStructuralDriftAttention(output.dimensions, ppdConclusion);
1494
+ const dimensions = attachPrePostDiffEvidence(enforceDeliveryReachability(enforceScopeContractDelivery(enforcePrePostDiffAvailability(enforceEvidenceBackedVerdicts(clampInconclusiveConfidence(gated.dimensions), dimensionsWithEvidence(evidence)), prePostDiff, evidence), evidence), undeliverable), prePostDiff, evidence);
1495
+ // H2 adds nothing to `mustAttend` on purpose: `enforceDeliveryReachability`
1496
+ // has already made every undeliverable dimension non-`pass`, so it is in
1497
+ // `needsAttention` (and `allPass` is false) through the verdict route. A
1498
+ // second flag for the same dimension would be a check whose result nothing
1499
+ // reads — the shape this primitive exists to catch.
1500
+ const { allPass, needsAttention } = summarizeVerdicts(dimensions, output, [
1501
+ ...(gated.mustAttend ? ['existing-functionality-intact'] : [])
1502
+ ]);
666
1503
  return { ...output, dimensions, allPass, needsAttention };
667
1504
  }
668
1505
  export function decideFifthDimension(input) {