peaks-loop 4.0.44 → 4.0.46

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. package/CHANGELOG.md +59 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/dist/cli/commands/code-runtime-commands.js +30 -19
  5. package/dist/cli/commands/codegraph-commands.d.ts +1 -0
  6. package/dist/cli/commands/codegraph-commands.js +49 -3
  7. package/dist/cli/commands/final-review-commands.js +3 -1
  8. package/dist/services/code/auto-compact-lifecycle.d.ts +39 -3
  9. package/dist/services/code/auto-compact-lifecycle.js +41 -6
  10. package/dist/services/code/auto-compact-orchestrator.js +20 -11
  11. package/dist/services/compact-statusline/compact-lifecycle-store.d.ts +11 -2
  12. package/dist/services/compact-statusline/compact-lifecycle-store.js +19 -1
  13. package/dist/services/compact-statusline/compact-statusline-service.d.ts +1 -1
  14. package/dist/services/compact-statusline/compact-statusline-service.js +22 -0
  15. package/dist/services/final-review/final-review-service.d.ts +224 -44
  16. package/dist/services/final-review/final-review-service.js +934 -97
  17. package/dist/services/final-review/index.d.ts +2 -1
  18. package/dist/services/final-review/index.js +2 -1
  19. package/dist/services/final-review/pre-post-diff.d.ts +137 -0
  20. package/dist/services/final-review/pre-post-diff.js +657 -0
  21. package/dist/services/skills/skill-statusline-renderer.js +30 -5
  22. package/dist/services/skills/statusline-palette.d.ts +6 -0
  23. package/dist/services/skills/statusline-palette.js +4 -1
  24. package/package.json +5 -5
  25. package/skills/peaks-code/SKILL.md +2 -2
  26. package/skills/peaks-final-review/SKILL.md +51 -18
  27. package/skills/peaks-final-review/references/4-dimensions.md +42 -5
@@ -1,4 +1,5 @@
1
1
  import { basename } from 'node:path';
2
+ import { AUTO_COMPACT_RED_LINE_RATIO } from '../context/auto-compact-types.js';
2
3
  import { computeRootSuffix as computeRootSuffixImpl, } from './skill-statusline-sid-suffix.js';
3
4
  // Re-export so existing test imports
4
5
  // (`import { formatShortSid, computeRootSuffix } from '.../skill-statusline-renderer'`)
@@ -181,6 +182,10 @@ function formatRatio(value) {
181
182
  *
182
183
  * <stage-glyph> <bar> <label>[ · <before>%][ → <after>%]
183
184
  *
185
+ * `armed` is the one exception: it renders WITHOUT the bar
186
+ * (`<glyph> armed · <before>% · fires at <redLine>%`) because a bar is
187
+ * a progress claim and a registered-but-unfired trigger has no progress.
188
+ *
184
189
  * Failed states additionally suffix the stage at which the compact failed.
185
190
  * Stalled states keep the active-stage cell count and render a plain
186
191
  * "stalled" label. Invalid states surface the read-reason verbatim as a
@@ -202,6 +207,16 @@ function renderCompact(state, palette) {
202
207
  return `${palette.compact.compacting} ${renderCompactBar(4, palette)} compacting${typeof state.triggerRatio === 'number'
203
208
  ? `${palette.inlineSeparator}${formatRatio(state.triggerRatio)}`
204
209
  : ''}`;
210
+ case 'armed': {
211
+ // Slice 2026-09-12-compact-band-policy: NO bar. A bar is a
212
+ // progress claim, and a registered-but-unfired trigger has no
213
+ // progress to report — it is waiting for the ratio to reach the
214
+ // red line on its own. Say exactly that instead.
215
+ const now = typeof state.triggerRatio === 'number'
216
+ ? `${palette.inlineSeparator}${formatRatio(state.triggerRatio)}`
217
+ : '';
218
+ return `${palette.compact.armed} armed${now}${palette.inlineSeparator}fires at ${formatRatio(AUTO_COMPACT_RED_LINE_RATIO)}`;
219
+ }
205
220
  case 'verifying':
206
221
  return `${palette.compact.verifying} ${renderCompactBar(6, palette)} verifying`;
207
222
  case 'completed':
@@ -518,7 +533,14 @@ export function renderStatusLine(model, options, env) {
518
533
  const compactSegment = renderCompact(model.compact, palette);
519
534
  let line;
520
535
  const hasCompact = compactSegment.length > 0;
521
- if (hasCompact) {
536
+ // Slice 2026-09-12-compact-band-policy: `armed` is a RESTING state, not
537
+ // an in-flight compact. It can hold for the whole band between the
538
+ // auto-fire ratio and the red line, so it is appended to the normal
539
+ // line instead of REPLACING the skill token — hiding which skill is
540
+ // active for many turns would be a regression the user never asked
541
+ // for. Every other (genuinely in-flight) stage keeps replacing it.
542
+ const armedOnly = model.compact.kind === 'armed';
543
+ if (hasCompact && !armedOnly) {
522
544
  // Compact state replaces the active / stale / idle skill content.
523
545
  // `invalid-presence` still surfaces its own diagnostic when compact
524
546
  // is also `invalid` (the compact diagnostic wins, since it's the
@@ -526,21 +548,24 @@ export function renderStatusLine(model, options, env) {
526
548
  line = `${brand} ${compactSegment}${rootSuffix}`;
527
549
  }
528
550
  else {
551
+ let base;
529
552
  switch (model.state) {
530
553
  case 'active':
531
- line = `${brand} ${renderActive(model.presence, palette, nowMs, capability, noColor, model.activeLeaf, model.twentyFourHourState)}${rootSuffix}`;
554
+ base = renderActive(model.presence, palette, nowMs, capability, noColor, model.activeLeaf, model.twentyFourHourState);
532
555
  break;
533
556
  case 'stale':
534
- line = `${brand} ${renderStale(model.presence, model.ageMs, palette, capability, noColor)}${rootSuffix}`;
557
+ base = renderStale(model.presence, model.ageMs, palette, capability, noColor);
535
558
  break;
536
559
  case 'invalid-presence':
537
- line = `${brand} ${renderInvalid(palette)}${rootSuffix}`;
560
+ base = renderInvalid(palette);
538
561
  break;
539
562
  case 'idle':
540
563
  default:
541
- line = `${brand} ${renderIdle(palette)}${rootSuffix}`;
564
+ base = renderIdle(palette);
542
565
  break;
543
566
  }
567
+ const armedSuffix = armedOnly ? `${palette.inlineSeparator}${compactSegment}` : '';
568
+ line = `${brand} ${base}${armedSuffix}${rootSuffix}`;
544
569
  }
545
570
  // Marquee is OFF for idle (and only idle). Compact states always
546
571
  // carry the band because the compact bar IS the headline. NO_COLOR
@@ -35,6 +35,12 @@ interface CompactPalette {
35
35
  readonly queued: string;
36
36
  readonly preparing: string;
37
37
  readonly compacting: string;
38
+ /**
39
+ * Slice 2026-09-12-compact-band-policy: a trigger is registered but no
40
+ * compaction is running. Rendered WITHOUT a bar — see
41
+ * `renderCompact` in skill-statusline-renderer.ts.
42
+ */
43
+ readonly armed: string;
38
44
  readonly verifying: string;
39
45
  readonly completed: string;
40
46
  readonly failed: string;
@@ -117,7 +117,7 @@ function buildPalette(capability, noColor) {
117
117
  idleLabel: 'empty',
118
118
  invalidMessage: 'presence unreadable',
119
119
  compact: {
120
- queued: '[', preparing: '+', compacting: '+', verifying: '+',
120
+ queued: '[', preparing: '+', compacting: '+', armed: '~', verifying: '+',
121
121
  completed: '*', failed,
122
122
  },
123
123
  barFilled: '#',
@@ -142,6 +142,9 @@ function buildPalette(capability, noColor) {
142
142
  queued: brandGlyph('◐'),
143
143
  preparing: brandGlyph('◑'),
144
144
  compacting: brandGlyph('◒'),
145
+ // Distinct from `compacting` on purpose: "waiting for the trigger"
146
+ // must not wear the same face as "a compact is in flight".
147
+ armed: brandGlyph('◔'),
145
148
  verifying: brandGlyph('◓'),
146
149
  completed: brandGlyph('✓'),
147
150
  failed,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "peaks-loop",
3
- "version": "4.0.44",
3
+ "version": "4.0.46",
4
4
  "description": "Loop Engineering CLI — workflow primitive / loop guards / evaluators / slice orchestration",
5
5
  "author": "SquabbyZ",
6
6
  "keywords": [
@@ -102,10 +102,10 @@
102
102
  "picomatch": "4.0.4",
103
103
  "yaml": "^2.9.0",
104
104
  "zod": "^4.4.3",
105
- "peaks-loop-internal-runtime": "0.0.29",
106
- "peaks-loop-mut": "0.1.42",
107
- "peaks-loop-shared-channel": "0.0.46",
108
- "peaks-loop-shared": "0.0.78"
105
+ "peaks-loop-internal-runtime": "0.0.31",
106
+ "peaks-loop-shared": "0.0.80",
107
+ "peaks-loop-shared-channel": "0.0.48",
108
+ "peaks-loop-mut": "0.1.44"
109
109
  },
110
110
  "devDependencies": {
111
111
  "@changesets/cli": "2.31.1",
@@ -185,7 +185,7 @@ Before the first planning action, run `peaks fresh-context preflight --prompt "<
185
185
  1. **PreToolUse hook — `peaks code gate-step-08`.** Installed by `peaks workspace init` on the `Bash` matcher; checks `job-shape.json` presence + fail-closed backup regex. If `job-shape.json` AND `progress.json` exist, surfaces `Next: slice #N of M (<currentSlice>)` so the LLM cannot wake up cold.
186
186
  2. **Size-fear ban — `peaks code emit-handoff`.** Refuses to emit a final handoff while `remaining > 0` under Job mode. Pass `--force-under-job` only with explicit user approval.
187
187
  3. **On-disk slice progress — `peaks job progress`.** `peaks job checkpoint --state done` writes `progress.json`. `peaks job progress --job-id <jid> [--allow-missing]` is the canonical reader.
188
- 4. **Forced auto-compact — `--enforce-job-mode`.** `peaks code context-now --enforce-job-mode` returns `action: 'auto-compact-now'` at ≥ 0.85. **Job mode at ≥ 0.85 is MANDATORY auto-compact** — Code MUST call `peaks code auto-compact` without confirmation.
188
+ 4. **Forced auto-compact — `peaks code context-now`.** It returns `action: 'auto-compact-now'` at ≥ 0.85. **≥ 0.85 is MANDATORY auto-compact in every mode (single-rid included)** — Code MUST call `peaks code auto-compact` without confirmation. `--enforce-job-mode` (v3.1.2) only labels the run `jobMode=true`; the ≥ 0.85 downgrade that used to apply to single-rid sessions was removed 2026-09-12.
189
189
 
190
190
  **Step 0.7 resume rule (read-FIRST):** on resume, `peaks code gate-step-08` reads `progress.json` first and surfaces `Next: slice #N of M (<currentSlice>)` so the orchestrator picks up at the right slice without re-reading the artifact tree.
191
191
 
@@ -209,7 +209,7 @@ Before the first planning action, run `peaks fresh-context preflight --prompt "<
209
209
 
210
210
  **Enforcement layers (defense in depth):**
211
211
  1. `src/services/code/auto-compact-orchestrator.ts` — `evaluateAutoCompactDecision` default-returns `shouldCompact: true` for both `pre-compact` and `red-line`. Only deferral is `inFlightBatch.hasInFlightBatch` (D6.e); no LLM/human approval branch.
212
- 2. `--enforce-job-mode` (v3.1.2) — Job mode elevates ≥0.85 to MANDATORY regardless of in-flight batch.
212
+ 2. `peaks code context-now` — ≥ 0.85 is MANDATORY (`auto-compact-now`) in every mode; `--enforce-job-mode` no longer gates that (2026-09-12). Only an in-flight sub-agent batch defers it.
213
213
  3. `peaks code gate-step-08` (PreToolUse hook) — surfaces `auto-compact-now` on every Bash call when ratio is in the zone, so the LLM cannot wake up cold and forget.
214
214
  4. Karpathy §4 exception: `peaks code auto-compact` is fired *by the orchestrator*, not by the user. If you find yourself about to write "ask the user to compact" / "prompt the user to run `/compact`", STOP — that is the regression.
215
215
 
@@ -137,40 +137,73 @@ Each `DimensionEvidence` carries:
137
137
  - `evidence` — list of `EvidenceItem` (`{ kind, description, artifact?, link? }`) with `EvidenceKind` ∈ `test-result | test-coverage | manual-spot-check | pre-post-diff | regression-suite | ac-mapping`
138
138
  - `confidence` — `high | medium | low`
139
139
 
140
- The service enforces that **all 4 dimensions are present**; a missing dimension throws `IncompleteFinalReviewError` and the call is a gate failure. Treat `allPass === true` + empty `needsAttention` as a clean handoff to the human. Anything else — `allPass === false`, a `fail` verdict, or an `inconclusive` verdict — must come back to the LLM loop with a re-prompt (do not ask the human to interpret raw LLM output).
140
+ The service enforces that **all 4 dimensions are present**; a missing dimension throws `IncompleteFinalReviewError` and the call is a gate failure. Treat `allPass === true` + empty `needsAttention` as a clean handoff to the human **only where it is reachable at all**: if the pre/post baseline for dimension 4 cannot be **computed** at all (see dimension 4 below), that dimension is permanently `inconclusive` and `allPass` is structurally `false` on every run — a `false` there means "no comparison was shown to the reviewer", not "the work regressed". The same delivery rule applies to **dimension 1's approved-scope contract** (`prd/handoff.md`): a contract that is absent, empty, unreadable, or too large to reach the reviewer costs that dimension its `pass` (a `scope-contract-gate` marker says so in the summary). Read both as "the reviewer was not given the document this verdict needs", never as "the work regressed". A computed baseline or a present contract is no longer at the mercy of the byte budget: both sources hold a floor in the allocator, so the budget-starved delivery failures left are the ones with no document behind them. One failure is NOT the budget's to fix, and the service says so in as many words: when **every** source on disk that supports a dimension is larger than the per-file cap (`MAX_EVIDENCE_BYTES_PER_FILE`, 10,240 bytes — derived from the total, so raising the total does not raise it), that dimension can never be delivered — not on this run and not on any run. The reviewer gets a `## Evidence delivery reachability (structural)` block naming the source and its size, and the dimension's summary carries a `delivery-reachability` marker with the same arithmetic. Read that `inconclusive` as "no evidence for this dimension can reach the reviewer", never as "the reviewer was unsure". Two cases worth naming: `needsAttention` is populated by the SERVICE too, not only by the verdicts — a delivered baseline whose own `VERDICT: ` line reports `STRUCTURAL DRIFT DETECTED` puts dimension 4 in `needsAttention` and clears `allPass` even when the reviewer passed it, because a detected removal is not "nothing needs attention"; and every dimension named by the reachability block above lands there too (it can never be `pass`), so the field tells you *why* it is red. And anything else — `allPass === false`, a `fail` verdict, or an `inconclusive` verdict — must come back to the LLM loop with a re-prompt (do not ask the human to interpret raw LLM output).
141
141
 
142
142
  ## The 4 dimensions (one-line summary)
143
143
 
144
144
  Full evidence contract per dimension: `references/4-dimensions.md`.
145
145
 
146
- 1. **functional-completeness** — every AC from the approved audit-goal maps to a passing test (`evidence.kind === 'ac-mapping'` + `test-result`).
146
+ 1. **functional-completeness** — every AC from the approved audit-goal maps to a passing test (`evidence.kind === 'ac-mapping'` + `test-result`), and the approved-scope contract (`prd/handoff.md`) was delivered to the reviewer in full.
147
147
  2. **problem-resolution** — there is a targeted test for the original problem case (`evidence.kind === 'test-result'` against the original repro).
148
148
  3. **no-new-bugs** — the regression suite is green AND the LLM surfaces 0 net-new failures (`evidence.kind === 'regression-suite'` + `manual-spot-check`).
149
149
  4. **existing-functionality-intact** — a pre/post baseline diff (test count, public API surface, key behavior) shows no unintended drift (`evidence.kind === 'pre-post-diff'`).
150
150
 
151
- > **⚠️ This dimension currently cannot pass — read before acting on it (verified 2026-09-12).**
152
- > `pre-post-diff` is a declared `EvidenceKind`, but **nothing in peaks-loop produces that artifact**.
153
- > The evidence actually mapped to this dimension is `rd/tech-doc.md` (design intent) and
154
- > `prd/handoff.md` (scope / non-goals) — neither is a baseline diff, and the model correctly
155
- > reports that ("the only FOUND source… is a design-intent document, not a regression assessment").
156
- > `peaks scan api-diff <doc>` is *not* a producer: it diffs an API **document**, not the code surface.
151
+ > **Status of this dimension — updated 2026-09-12 (the `pre-post-diff` producer now ships).**
152
+ > The producer exists. `peaks prepare-final-review` runs a read-only git comparison of a base ref
153
+ > against the working tree and writes `.peaks/_runtime/<sessionId>/final-review/api-diff.txt`
154
+ > — the test-file and non-test `.ts` / `.tsx` source-file lists, the `it(` / `test(` case counts
155
+ > before/after, and the added/removed top-level `export` names over changed `.ts` / `.tsx` files —
156
+ > and maps that artifact into this dimension's `supports`. The service also attaches it to the
157
+ > dimension as an `EvidenceItem` of kind `pre-post-diff` whose `artifact` points at that file.
158
+ > The dimension can reach `pass` — when a baseline could actually be computed **and was delivered to the
159
+ > reviewer**. Both halves are required. The source holds a floor in the evidence allocation, so a
160
+ > budget that runs out no longer drops it; if it ever is dropped the reviewer's prompt says
161
+ > `STATUS: COMPUTED ON DISK, NOT DELIVERED`
162
+ > instead of claiming a comparison it does not carry, the dimension's `pass` is downgraded to
163
+ > `inconclusive`, and no `pre-post-diff` `EvidenceItem` is attached to it. A baseline the reviewer never
164
+ > saw is not evidence, exactly as a baseline that was never computed is not — the verdict may not
165
+ > outlive the evidence that was actually handed over. Two properties make that judgement structural
166
+ > rather than arithmetic: a source is delivered only when the bytes the reviewer received carry its
167
+ > conclusion (for this artifact, its opening `VERDICT:` line), and the delivered conclusion is then
168
+ > READ — a `STRUCTURAL DRIFT DETECTED` line puts this dimension in `needsAttention` and clears
169
+ > `allPass` even when the reviewer answered `pass`, because a detected removal is a question for a
170
+ > human, not a clean handoff.
157
171
  >
158
- > **Consequences:** `allPass === true` is **unreachable by construction**, for every workflow.
159
- > This dimension will return `inconclusive` with an empty `evidence[]` even when the work is
160
- > perfect. Treat that as a **tooling** state, not as evidence of a regression — and do **not**
161
- > "fix" it by re-mapping `qa/test-reports` into this dimension's `supports`, which would turn the
162
- > gate green without producing the baseline diff the definition above requires.
172
+ > **A baseline that cannot be computed is never invented.** No base ref resolving, a base ref that
173
+ > resolves to HEAD itself (an empty range — what a shallow clone's `merge-base` produces on its
174
+ > default path), or a project that is not a git work tree, all mean the same thing: no artifact is
175
+ > written, the reviewer is told why in the prompt, and a `pass` on this dimension is downgraded to
176
+ > `inconclusive` — for EVERY one of those causes, with no exemptions. "This project keeps no
177
+ > baseline" is the CAUSE of the missing evidence; it is not a reason to trust the claim the
178
+ > evidence was supposed to support, and a permanently-`inconclusive` dimension is the honest
179
+ > reading of a comparison that never happened. The base ref comes from `--base <ref>`; the default
180
+ > is the merge-base with `origin/HEAD`, then `origin/main`, then `origin/master`, then `HEAD~1`,
181
+ > and when none of those resolve the reason says so and names `--base` as the way out.
163
182
  >
164
- > **Real fix (unbuilt):** a producer for the pre/post baseline diff (test-count delta, public-API
165
- > surface snapshot) written to `.peaks/_runtime/<sessionId>/final-review/api-diff.txt`, then mapped
166
- > into this dimension's `supports`. Until that ships, `needsAttention` always contains this
167
- > dimension — a permanently-red gate that reviewers will otherwise learn to ignore.
183
+ > **What that means for `allPass`.** On a project that is not a git work tree — or where no base ref
184
+ > resolves — this dimension is permanently `inconclusive`, so `allPass === true` can never be reached
185
+ > there, however complete the other nine evidence sources are. That is the intended reading and not a
186
+ > bug to work around: nothing was ever compared, so there is no answer to hand over. Likewise, a
187
+ > project whose evidence set outgrows the reviewer's input budget leaves this block OMITTED, and the
188
+ > honest envelope then says `inconclusive` rather than asserting a comparison the reviewer was never
189
+ > shown.
190
+ >
191
+ > **Boundary:** the export comparison is a line-anchored regex, not a type checker — it detects a
192
+ > removed or renamed export and cannot detect a changed signature. Type-only exports count, both
193
+ > spellings (`export type { T }` and `export { type T }`). File-level DELETION is visible, because
194
+ > it is read from the file lists rather than inferred from the counts — a module with no
195
+ > `export ` line and no `it(` is still reported when it is deleted. And the reverse direction is
196
+ > guarded too: a name that appears on both sides of the diff, a `git mv`, and an `it(` ->
197
+ > `test.each(` conversion are reported as changes, not as removals. The artifact carries a
198
+ > wall-clock timestamp, so two runs over the same base differ on that line and on nothing else.
199
+ > And still do **not** "fix" a red dimension by re-mapping `qa/test-reports` into its `supports`:
200
+ > that turns the gate green without producing the baseline diff the definition above requires.
168
201
 
169
202
  ## Human's role
170
203
 
171
204
  The human reviews evidence, **judges business outcomes (NOT code)**. The LLM produces structured evidence; the human's job is to:
172
205
 
173
- - confirm that `allPass === true` corresponds to the business outcome they actually want (not just "tests are green");
206
+ - confirm that `allPass === true` corresponds to the business outcome they actually want (not just "tests are green", and — for dimension 4 — not just "a diff file exists somewhere on disk");
174
207
  - decide what to do with `needsAttention` items — accept the LLM's verdict, override a `pass` to `fail` when the evidence is weak, or send the slice back to RD with a re-prompt;
175
208
  - gate the release / archive action based on `allPass` + their own business review, not just on the LLM signal.
176
209
 
@@ -20,10 +20,26 @@ Every acceptance criterion in the approved audit-goal at `.peaks/_runtime/<sessi
20
20
 
21
21
  ### Verdict semantics
22
22
 
23
- - `pass` — every AC has a passing test and the test suite is green.
23
+ - `pass` — every AC has a passing test, the test suite is green, **and the approved-scope contract was delivered**: the `prd/handoff.md` source block reached the reviewer with its WHOLE byte count (`FOUND at … — N bytes`), because "complete" is defined against the approved scope and its non-goals, and a test report alone shows only that *something* was built.
24
24
  - `fail` — one or more ACs are unmapped, the targeted test is failing, or the suite is red on a non-flaky ground.
25
25
  - `inconclusive` (needs-human) — the mapping is plausible but the human needs to confirm that a passing test truly reflects the business intent of an AC. Example: a test exists for "config-service splits into 3 modules" but the human must decide whether the *seam* the test exercises is the seam they actually wanted.
26
26
 
27
+ ### When the scope contract does not arrive — same rule as dimension 4
28
+
29
+ The service keys this dimension's `pass` on the **delivery** of its scope contract, not on its existence on disk, for the same reason dimension 4 keys its `pass` on the delivered baseline:
30
+
31
+ | Situation | What the reviewer is told | Verdict the service allows |
32
+ |---|---|---|
33
+ | `prd/handoff.md` inlined in full | `STATUS: FOUND at … — N bytes` | `pass` is available |
34
+ | `prd/handoff.md` present but the budget did not reach it, or the read was truncated | `STATUS: MISSING (omitted)`, or `TRUNCATED, showing the first N of M bytes` | `pass` is downgraded to `inconclusive`, with a `scope-contract-gate` marker in the summary |
35
+ | `prd/handoff.md` exists for this run but carries nothing (0 bytes / whitespace) | `STATUS: MISSING (empty)` | Same downgrade: the PRD phase ran and the reviewer was given no contract. An empty contract is a delivery failure, not an absent phase. |
36
+ | `prd/handoff.md` exists but this process could not read it (EACCES / EBUSY / EISDIR) | `STATUS: UNREADABLE` | Same downgrade. Deliberately NOT reported as "no PRD phase": the file is there, the read failed, and only one of those two facts is fixable. |
37
+ | No `prd/handoff.md` in this project at all — ENOENT (no PRD phase) | `STATUS: MISSING (missing)` | Not a delivery failure — the dimension is judged on the evidence that exists. Absence is not the same fact as non-delivery. |
38
+
39
+ The delivery rule is deliberately not "some bytes arrived": `qa-test-report` also supports this dimension, which is exactly how a `pass` used to survive while the contract that defines the dimension was inlined with **zero** bytes. See `enforceScopeContractDelivery()` in `src/services/final-review/final-review-service.ts`.
40
+
41
+ The contract source holds a **floor** in the byte allocator: its whole unit is reserved for this dimension, so a saturated run can no longer drop it as a side effect of the allocation order. The downgrade rows above are therefore about a document that is genuinely absent, empty or unreadable — not about a contract that merely happened to sit last in the order.
42
+
27
43
  ### Example
28
44
 
29
45
  > `dimension: "functional-completeness"`, `verdict: "pass"`, `summary: "All 3 success criteria from the approved goal are covered by passing tests. AC-1 covered by config-service.modules.test.ts; AC-2 covered by config-service.api.test.ts (public API snapshot unchanged); AC-3 covered by coverage report at 100% lines/branches for the changed files."`, `evidence: [...]`, `confidence: "high"`.
@@ -92,10 +108,30 @@ A pre/post baseline diff shows no unintended drift in the test surface, public A
92
108
 
93
109
  ### Verdict semantics
94
110
 
95
- - `pass` — every measured dimension (tests, API, behavior) is unchanged or changed only in ways that the slice was explicitly authorized to change (e.g. AC-2 says "add a new exported helper `resolveWithSchema()`" — that IS the authorized change).
111
+ - `pass` — the baseline was **compared and delivered**: the pre/post diff block reached the reviewer **with its conclusion** (`STATUS: FOUND`, and the `VERDICT:` line is in the delivered bytes — the verdict is the first thing in the artifact, so an over-cap slice still carries it), and every measured dimension (tests, API, behavior) is unchanged or changed only in ways that the slice was explicitly authorized to change (e.g. AC-2 says "add a new exported helper `resolveWithSchema()`" — that IS the authorized change).
96
112
  - `fail` — an unauthorized change slipped in: an exported symbol disappeared, a test was deleted rather than updated, a behavior baseline drifted without a corresponding AC.
97
113
  - `inconclusive` (needs-human) — a change is present that *could* be authorized drift or *could* be an unintentional regression. The human rules.
98
114
 
115
+ ### When the baseline itself is missing — the one case where no verdict can be earned
116
+
117
+ A `pass` here rests on a **comparison**, so two things must both be true: the producer had to *compute* a baseline, and that baseline had to *reach the reviewer's prompt* — conclusion included. Neither substitutes for the other.
118
+
119
+ | Situation | What the reviewer is told | Verdict the service allows |
120
+ |---|---|---|
121
+ | Baseline computed and inlined | `STATUS: COMPUTED` + the diff block | `pass` is available |
122
+ | Baseline computed, dropped by the evidence budget | `STATUS: COMPUTED ON DISK, NOT DELIVERED` + the block marked `MISSING (omitted)` | `pass` is downgraded to `inconclusive` |
123
+ | Baseline inlined but larger than the per-file cap | `STATUS: FOUND … TRUNCATED, showing the first N of M bytes` — the slice still opens with the `VERDICT:` line | `pass` is available (the conclusion is in the delivered bytes) |
124
+ | No base ref resolves / base resolves to HEAD (empty range) | `STATUS: UNAVAILABLE` + the reason | `pass` is downgraded to `inconclusive` |
125
+ | The project is not a git work tree | `STATUS: UNAVAILABLE` + the reason | `pass` is downgraded to `inconclusive` — permanently, on every run |
126
+
127
+ The "dropped by the evidence budget" row is now a defensive branch rather than the expected failure: this source holds a **floor** in the allocator — its whole unit is reserved for this dimension — so when a baseline IS computed it is delivered, and the reviewer is never asked to judge this dimension against a comparison it was not shown. The rows that still fire are the ones where no baseline exists to deliver.
128
+
129
+ Three consequences worth stating plainly, because all three read like defects and are not:
130
+
131
+ - **A non-git project can never reach `allPass === true`.** Dimension 4 is permanently `inconclusive` there: no comparison exists to hand over, so there is no honest way to have it green. The service does not invent one, and a design-intent document (`rd/tech-doc.md`, `prd/handoff.md`) is not a substitute — it states what was intended, not what changed. There is no CLI way out either: `--base <ref>` names a COMMIT to compare against, so it cannot help a project that has no git history to resolve a ref in — the reason the reviewer is given is the project's state, not a missing argument.
132
+ - **A dimension whose evidence was omitted does not get a pass "because the file exists".** If the block is not in the prompt, the reviewer never saw it; the service downgrades the verdict and does **not** attach the artifact as an `EvidenceItem`, because an envelope citing evidence the reviewer never received is the forged clean handoff this gate exists to prevent.
133
+ - **A saturated run can redden more than one dimension, and that is the honest reading.** The evidence budget allocates each source WHOLE or not at all (never a partial slice), so a run whose sources do not all fit drops whole sources — and each dimension whose contract source was dropped loses its `pass`: dimension 4 when the baseline goes, dimension 1 when the approved-scope contract does.
134
+
99
135
  ### Example
100
136
 
101
137
  > `dimension: "existing-functionality-intact"`, `verdict: "pass"`, `summary: "Public API snapshot: 0 symbols removed, 1 symbol added (resolveWithSchema — authorized by AC-2). Test count: +12 (new tests for resolveWithSchema), -3 (deleted tests for the old monolithic resolve that resolveWithSchema supersedes). CLI help text byte-identical to pre-fix golden."`, `evidence: [{ kind: "pre-post-diff", description: "Public API surface diff: +resolveWithSchema, -3 obsolete test files, no other deltas", artifact: ".peaks/_runtime/<sessionId>/final-review/api-diff.txt" }]`, `confidence: "high"`.
@@ -106,12 +142,13 @@ A pre/post baseline diff shows no unintended drift in the test surface, public A
106
142
 
107
143
  The service contract (`src/services/final-review/final-review-service.ts:23-31` and `:88-93`):
108
144
 
109
- - `allPass === true` iff every dimension's `verdict === 'pass'`.
110
- - `needsAttention` is the list of dimension names whose verdict is `fail` or `inconclusive`. The LLM does NOT need to populate it — the service enforces presence of all 4 dimensions and the human-facing summarizer (in peaks-code or peaks-txt) computes `needsAttention` for display.
145
+ - `allPass === true` iff every dimension's `verdict === 'pass'` **and** the service itself has nothing to flag. The two fields are derived from the verdicts, never copied from the model's own summary: a model that wrote a fabricated `pass` plus a matching `allPass: true` cannot hand over an unsupported clean review.
146
+ - `needsAttention` is the list of dimension names whose verdict is `fail` or `inconclusive`, **plus** any dimension the service has to flag mechanically even though the reviewer passed it. The one such flag today is a delivered pre/post baseline whose own `VERDICT:` line reports `STRUCTURAL DRIFT DETECTED` (or a verdict line the service cannot classify): a detected structural removal may well be authorized, but a review that says "4/4 pass, nothing needs attention" directly above a diff that reports a removal it attached itself is self-contradicting, so that dimension is listed and `allPass` is `false`. The dimension's verdict is left as the reviewer wrote it — the call on whether a removal was authorized stays with the reviewer and the human. A second, related case is NOT a mechanical flag but a structural red the service explains: when **every** source on disk supporting a dimension is larger than the per-file cap (10,240 bytes, derived from `MAX_EVIDENCE_BYTES_TOTAL`), no source can ever be delivered under the `whole` rule, so the reviewer is told so in a `## Evidence delivery reachability (structural)` block and the dimension's summary carries a `delivery-reachability` marker naming the source and its byte count. Such a dimension is `inconclusive` by byte arithmetic — never by having passed and been overruled — and it appears in `needsAttention` through the ordinary non-`pass` route, which is what makes the CLI envelope state the reason instead of leaving a permanent red unexplained.
147
+ - The LLM does NOT need to populate `needsAttention` — the service enforces presence of all 4 dimensions and derives the field.
111
148
  - An `IncompleteFinalReviewError` is thrown when JSON is malformed or any required dimension is missing. That is a **gate failure**, not a `fail` verdict — the LLM call is invalid and must be re-prompted, not surfaced to the human.
112
149
 
113
150
  ## Confidence and what it means
114
151
 
115
152
  - `high` — evidence is concrete (named test file, named artifact, deterministic run).
116
153
  - `medium` — evidence is concrete but covers only part of the dimension; the LLM is being honest about coverage gaps.
117
- - `inconclusive` verdicts should always be `medium` or `low` confidence; `high` confidence on `inconclusive` is a contradiction and the LLM should be re-prompted.
154
+ - `inconclusive` verdicts should always be `medium` or `low` confidence; `high` confidence on `inconclusive` is a contradiction and the LLM should be re-prompted. The service does not rely on the re-prompt alone: it clamps a `high` on an `inconclusive` verdict to `medium` and marks the summary, so the contradiction cannot reach the human in the envelope even if the model keeps producing it.