peaks-loop 4.0.43 → 4.0.45
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +73 -0
- package/README-en.md +1 -1
- package/README.md +1 -1
- package/dist/cli/commands/codegraph-commands.d.ts +1 -0
- package/dist/cli/commands/codegraph-commands.js +239 -8
- package/dist/cli/commands/final-review-commands.d.ts +34 -10
- package/dist/cli/commands/final-review-commands.js +132 -34
- package/dist/cli/commands/share-commands.d.ts +49 -0
- package/dist/cli/commands/share-commands.js +114 -14
- package/dist/services/codegraph/codegraph-autorefresh.js +12 -0
- package/dist/services/codegraph/codegraph-exclude-integrity.d.ts +61 -0
- package/dist/services/codegraph/codegraph-exclude-integrity.js +98 -0
- package/dist/services/codegraph/codegraph-exclude-reconciler.d.ts +26 -0
- package/dist/services/codegraph/codegraph-exclude-reconciler.js +217 -0
- package/dist/services/codegraph/codegraph-exclude-repair.d.ts +102 -0
- package/dist/services/codegraph/codegraph-exclude-repair.js +266 -0
- package/dist/services/codegraph/codegraph-preflight-service.js +12 -0
- package/dist/services/codegraph/codegraph-service.d.ts +0 -1
- package/dist/services/codegraph/codegraph-service.js +5 -4
- package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.d.ts +29 -0
- package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.js +88 -0
- package/dist/services/doctor/doctor-service/plugin-registry.js +2 -0
- package/dist/services/doctor/doctor-service/types.d.ts +27 -0
- package/dist/services/final-review/final-review-service.d.ts +335 -1
- package/dist/services/final-review/final-review-service.js +1457 -6
- package/dist/services/final-review/index.d.ts +2 -1
- package/dist/services/final-review/index.js +2 -1
- package/dist/services/final-review/pre-post-diff.d.ts +137 -0
- package/dist/services/final-review/pre-post-diff.js +657 -0
- package/dist/services/prd/handoff-auto-regen.js +0 -1
- package/dist/services/prd/handoff-service.d.ts +9 -1
- package/dist/services/prd/handoff-service.js +48 -6
- package/package.json +7 -5
- package/skills/peaks-final-review/SKILL.md +79 -35
- package/skills/peaks-final-review/references/4-dimensions.md +42 -5
|
@@ -47,7 +47,15 @@ export declare function writeHandoff(handoff: Handoff, projectRoot: string): Pro
|
|
|
47
47
|
* malformed frontmatter. */
|
|
48
48
|
export declare function readHandoff(filePath: string): Promise<Handoff>;
|
|
49
49
|
/** Verify a handoff by re-reading and re-hashing. Returns a probe
|
|
50
|
-
* (never throws on hash mismatch — that's the outcome).
|
|
50
|
+
* (never throws on hash mismatch — that's the outcome).
|
|
51
|
+
*
|
|
52
|
+
* N1: the two failure classes are reported separately. The previous
|
|
53
|
+
* single `catch` folded every read failure into `file-missing`, so a
|
|
54
|
+
* handoff that WAS on disk but whose frontmatter the parser refused was
|
|
55
|
+
* reported as absent — sending an operator to look for a missing file
|
|
56
|
+
* that was right there. Only a genuine read failure (ENOENT, EACCES, …)
|
|
57
|
+
* is `file-missing` now; anything the parser rejects is
|
|
58
|
+
* `frontmatter-malformed`. */
|
|
51
59
|
export declare function verifyHandoff(filePath: string): Promise<HandoffProbe>;
|
|
52
60
|
/** Return the raw markdown content of a handoff file (frontmatter +
|
|
53
61
|
* body verbatim). For human display. */
|
|
@@ -66,15 +66,30 @@ export async function readHandoff(filePath) {
|
|
|
66
66
|
return parseHandoffContent(content);
|
|
67
67
|
}
|
|
68
68
|
/** Verify a handoff by re-reading and re-hashing. Returns a probe
|
|
69
|
-
* (never throws on hash mismatch — that's the outcome).
|
|
69
|
+
* (never throws on hash mismatch — that's the outcome).
|
|
70
|
+
*
|
|
71
|
+
* N1: the two failure classes are reported separately. The previous
|
|
72
|
+
* single `catch` folded every read failure into `file-missing`, so a
|
|
73
|
+
* handoff that WAS on disk but whose frontmatter the parser refused was
|
|
74
|
+
* reported as absent — sending an operator to look for a missing file
|
|
75
|
+
* that was right there. Only a genuine read failure (ENOENT, EACCES, …)
|
|
76
|
+
* is `file-missing` now; anything the parser rejects is
|
|
77
|
+
* `frontmatter-malformed`. */
|
|
70
78
|
export async function verifyHandoff(filePath) {
|
|
71
|
-
let
|
|
79
|
+
let content;
|
|
72
80
|
try {
|
|
73
|
-
|
|
81
|
+
content = await readFile(filePath, 'utf8');
|
|
74
82
|
}
|
|
75
83
|
catch {
|
|
76
84
|
return { ok: false, reason: 'file-missing' };
|
|
77
85
|
}
|
|
86
|
+
let handoff;
|
|
87
|
+
try {
|
|
88
|
+
handoff = parseHandoffContent(content);
|
|
89
|
+
}
|
|
90
|
+
catch {
|
|
91
|
+
return { ok: false, reason: 'frontmatter-malformed' };
|
|
92
|
+
}
|
|
78
93
|
if (handoff.frontmatter.schemaVersion !== HANDOFF_SCHEMA_VERSION) {
|
|
79
94
|
return {
|
|
80
95
|
ok: false,
|
|
@@ -119,20 +134,47 @@ function parseHandoffContent(content) {
|
|
|
119
134
|
if (!isHandoffFrontmatter(parsed)) {
|
|
120
135
|
throw new Error('handoff: frontmatter shape validation failed');
|
|
121
136
|
}
|
|
122
|
-
|
|
137
|
+
// N1: normalize the version to its canonical string form. `schemaVersion: 2`
|
|
138
|
+
// (bare) is valid YAML that parses to the NUMBER 2; `schemaVersion: '2'` is
|
|
139
|
+
// what `stringifyYaml` writes. Both mean schema version 2, so the parsed
|
|
140
|
+
// frontmatter is returned with the canonical `'2'` rather than the raw scalar
|
|
141
|
+
// — otherwise every downstream `=== '2'` comparison would depend on which
|
|
142
|
+
// producer wrote the file.
|
|
143
|
+
return {
|
|
144
|
+
frontmatter: { ...parsed, schemaVersion: HANDOFF_SCHEMA_VERSION },
|
|
145
|
+
body
|
|
146
|
+
};
|
|
123
147
|
}
|
|
124
148
|
function serializeHandoff(handoff) {
|
|
125
149
|
const yamlStr = stringifyYaml(handoff.frontmatter).trimEnd();
|
|
126
150
|
return `---\n${yamlStr}\n---\n${handoff.body}`;
|
|
127
151
|
}
|
|
152
|
+
/**
|
|
153
|
+
* N1: accept BOTH shapes of `schemaVersion` — the string `'2'` and the bare
|
|
154
|
+
* YAML number `2` — and reject any other value.
|
|
155
|
+
*
|
|
156
|
+
* The reader used to require `typeof v.schemaVersion === 'string'`. That made
|
|
157
|
+
* `readHandoff` refuse `prd/handoff.md` written by `handoff-auto-regen.ts`,
|
|
158
|
+
* which emits the unquoted `schemaVersion: 2`, while the
|
|
159
|
+
* `AUDIT_REQUIRES_HANDOFF` prereq — a SUBSTRING check for `schemaVersion: 2` —
|
|
160
|
+
* happily passed the same bytes. So the gate that exists to guarantee a
|
|
161
|
+
* readable handoff was satisfied by a handoff the parser would not read, and
|
|
162
|
+
* `peaks prd handoff verify` exited 1 on a healthy file.
|
|
163
|
+
*
|
|
164
|
+
* Quoting the writer instead is NOT a fix: it would delete the very substring
|
|
165
|
+
* the prereq pins, turning a broken read into a broken gate. The tolerant read
|
|
166
|
+
* is the only change that satisfies both consumers.
|
|
167
|
+
*/
|
|
168
|
+
function isSchemaVersion2(value) {
|
|
169
|
+
return value === HANDOFF_SCHEMA_VERSION || value === 2;
|
|
170
|
+
}
|
|
128
171
|
function isHandoffFrontmatter(value) {
|
|
129
172
|
if (!value || typeof value !== 'object')
|
|
130
173
|
return false;
|
|
131
174
|
const v = value;
|
|
132
175
|
return (typeof v.requestId === 'string' &&
|
|
133
176
|
typeof v.sessionId === 'string' &&
|
|
134
|
-
|
|
135
|
-
typeof v.schemaVersion === 'string' &&
|
|
177
|
+
isSchemaVersion2(v.schemaVersion) &&
|
|
136
178
|
typeof v.handoffHash === 'string' &&
|
|
137
179
|
typeof v.writtenAt === 'string' &&
|
|
138
180
|
Array.isArray(v.goals) &&
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "peaks-loop",
|
|
3
|
-
"version": "4.0.
|
|
3
|
+
"version": "4.0.45",
|
|
4
4
|
"description": "Loop Engineering CLI — workflow primitive / loop guards / evaluators / slice orchestration",
|
|
5
5
|
"author": "SquabbyZ",
|
|
6
6
|
"keywords": [
|
|
@@ -99,12 +99,13 @@
|
|
|
99
99
|
"better-sqlite3": "^12.11.1",
|
|
100
100
|
"commander": "^12.1.0",
|
|
101
101
|
"fzf": "^0.5.2",
|
|
102
|
+
"picomatch": "4.0.4",
|
|
102
103
|
"yaml": "^2.9.0",
|
|
103
104
|
"zod": "^4.4.3",
|
|
104
|
-
"peaks-loop-mut": "0.1.
|
|
105
|
-
"peaks-loop-shared": "0.0.
|
|
106
|
-
"peaks-loop-internal-runtime": "0.0.
|
|
107
|
-
"peaks-loop-shared
|
|
105
|
+
"peaks-loop-mut": "0.1.43",
|
|
106
|
+
"peaks-loop-shared-channel": "0.0.47",
|
|
107
|
+
"peaks-loop-internal-runtime": "0.0.30",
|
|
108
|
+
"peaks-loop-shared": "0.0.79"
|
|
108
109
|
},
|
|
109
110
|
"devDependencies": {
|
|
110
111
|
"@changesets/cli": "2.31.1",
|
|
@@ -114,6 +115,7 @@
|
|
|
114
115
|
"@stryker-mutator/vitest-runner": "^9.6.1",
|
|
115
116
|
"@types/better-sqlite3": "^7.6.13",
|
|
116
117
|
"@types/node": "^22.10.2",
|
|
118
|
+
"@types/picomatch": "4.0.3",
|
|
117
119
|
"@typescript-eslint/eslint-plugin": "8.66.0",
|
|
118
120
|
"@typescript-eslint/parser": "8.66.0",
|
|
119
121
|
"@vitest/coverage-istanbul": "^4.1.10",
|
|
@@ -69,7 +69,9 @@ interface PrepareFinalReviewOptions {
|
|
|
69
69
|
}
|
|
70
70
|
```
|
|
71
71
|
|
|
72
|
-
The service is the **gate primitive** that closes the 10% human / 90% LLM loop. It reads the approved audit-goal JSON from `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`,
|
|
72
|
+
The service is the **gate primitive** that closes the 10% human / 90% LLM loop. It reads the approved audit-goal JSON from `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`, **collects real evidence from disk** (`qa/test-reports`, `qa/test-cases`, `qa/security-findings`, `qa/performance-findings`, `rd/{tech-doc,bug-analysis,code-review,security-review}.md`, `prd/handoff.md`), inlines it into the 4-dim review prompt under byte bounds, calls an injected `LlmRunner` exactly once, parses the response, and validates that all four required dimensions are present. It throws `IncompleteFinalReviewError` on malformed JSON or missing dimensions — callers MUST treat that as a gate failure (return to human for re-prompting) and never let autonomous work proceed on a partial review.
|
|
73
|
+
|
|
74
|
+
> **Evidence-backed verdicts (added 2026-09-12).** The service previously sent the model only `successCriteria` and no evidence at all, so it could not honestly grade anything — a real run returned 4/4 `inconclusive`, and a model willing to confabulate could have returned `pass`. Verdicts are now gated **structurally**, not by prompt wording: any `pass` whose supporting sources were all missing/empty is rewritten to `inconclusive` + `confidence: low` (`fail` is never softened), and `allPass` is derived from the gated verdicts so it can only narrow. Every non-`found` evidence source renders an explicit `STATUS: MISSING (<why>)` in the prompt, so the model always knows what it does not know.
|
|
73
75
|
|
|
74
76
|
> The `LlmRunner` interface is intentionally minimal so this service reuses the same provider injection seam as `audit-goal-service` and the slice LLMArbitrator (`src/services/audit/audit-goal-service.ts:16`). No provider implementation is baked in at this layer.
|
|
75
77
|
|
|
@@ -85,41 +87,32 @@ All of the following MUST be true before invoking this skill:
|
|
|
85
87
|
|
|
86
88
|
If any precondition is missing, **STOP** and route back to the responsible skill. Do not paper over a missing artifact with a hand-written successCriteria — the review must reflect what the human originally approved.
|
|
87
89
|
|
|
88
|
-
## Invocation
|
|
89
|
-
|
|
90
|
-
> **Pre-flight finding (HARD):** The plan prose at `docs/superpowers/plans/2026-06-25-slice-topology-multipass-phase-4.md:146` documents the invocation as:
|
|
91
|
-
>
|
|
92
|
-
> ```bash
|
|
93
|
-
> peaks prepare-final-review --rid <rid> --json
|
|
94
|
-
> ```
|
|
95
|
-
>
|
|
96
|
-
> **This CLI command does NOT yet exist.** `prepareFinalReview()` is implemented as a service in `src/services/final-review/final-review-service.ts` and is unit-tested in `tests/unit/final-review/final-review-service.test.ts`, but it is **not wired to a CLI subcommand**. A `peaks prepare-final-review` subcommand is the planned forward-looking surface; the integration sits at the service layer, not the CLI layer, today.
|
|
97
|
-
>
|
|
98
|
-
> **Pick for the future CLI wrapper file (audit recommendation):** create a new `src/cli/commands/final-review-commands.ts` matching the `peaks-<group>-commands.ts` naming convention (`qa-commands.ts`, `code-review-commands.ts`, `audit-commands.ts`). A grep of `src/cli/commands/` confirms NO `final-review-commands.ts` and NO `prepareFinalReview` import in any CLI file. Rationale: the 4-dim business review is conceptually distinct from `peaks qa *` (which is autonomous gate verification) — it is the human-acceptance terminator, not an internal gate. A separate command group preserves that boundary.
|
|
99
|
-
|
|
100
|
-
### Current call path (today, until a CLI wrapper is built)
|
|
101
|
-
|
|
102
|
-
```ts
|
|
103
|
-
// peaks-code end-of-workflow, peaks-txt, or any other hand-rolled caller
|
|
104
|
-
import { prepareFinalReview } from './src/services/final-review/final-review-service.js';
|
|
105
|
-
import { auditGoalLlmRunner } from './src/services/audit/llm-runner.js'; // or your provider
|
|
106
|
-
|
|
107
|
-
const out = await prepareFinalReview('<rid>', {
|
|
108
|
-
projectRoot: '<absolute path>',
|
|
109
|
-
sessionId: '<sessionId>',
|
|
110
|
-
llmRunner: auditGoalLlmRunner
|
|
111
|
-
});
|
|
112
|
-
```
|
|
90
|
+
## Invocation
|
|
113
91
|
|
|
114
|
-
|
|
92
|
+
> **Correction (2026-09-12).** An earlier revision of this file claimed `peaks prepare-final-review`
|
|
93
|
+
> "does NOT yet exist" and told callers to hand-roll a `prepareFinalReview()` caller instead.
|
|
94
|
+
> **That was wrong.** The CLI wrapper exists and is registered —
|
|
95
|
+
> `src/cli/commands/final-review-commands.ts` (`W5 Fix M2`), command registered at its line ~130.
|
|
96
|
+
> The hand-rolled snippet that used to sit here also had the wrong flag shape (`--rid <rid>`);
|
|
97
|
+
> **the rid is positional**. Use the CLI.
|
|
115
98
|
|
|
116
99
|
```bash
|
|
117
|
-
|
|
118
|
-
# src/cli/commands/final-review-commands.ts
|
|
119
|
-
peaks prepare-final-review --rid <rid> [--session-id <sid>] --json
|
|
100
|
+
peaks prepare-final-review <rid> [--session-id <sid>] [--project <path>] [--llm-provider <name>] [--json]
|
|
120
101
|
```
|
|
121
102
|
|
|
122
|
-
|
|
103
|
+
- `<rid>` is **positional** (not `--rid`). It resolves to `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`.
|
|
104
|
+
- `--session-id` defaults to the active workspace binding when omitted.
|
|
105
|
+
- **`--llm-provider` defaults to `stub`, and `stub` is the only provider that works today.**
|
|
106
|
+
`stub` runs no real review — it returns a scaffold envelope so CI can prove the route is reachable.
|
|
107
|
+
Verified 2026-09-12: passing a real provider (`--llm-provider anthropic`) returns
|
|
108
|
+
`status: not-applicable` / `serviceWired: false` / `providerBinding: unknown` and tells you to re-run
|
|
109
|
+
with `stub`; the CLI's own nextAction calls real-provider binding "a follow-up slice". So there is
|
|
110
|
+
currently **no reachable path to a real 4-dim review** — confirming the route works is all `stub`
|
|
111
|
+
can do.
|
|
112
|
+
- `--json` is required for machine consumption (peaks-code, peaks-txt, downstream CI).
|
|
113
|
+
|
|
114
|
+
Calling the service directly (`prepareFinalReview(rid, { projectRoot, sessionId, llmRunner })`) remains
|
|
115
|
+
valid for callers that need a custom `LlmRunner` injection seam, but it is no longer the only path.
|
|
123
116
|
|
|
124
117
|
## Output
|
|
125
118
|
|
|
@@ -144,22 +137,73 @@ Each `DimensionEvidence` carries:
|
|
|
144
137
|
- `evidence` — list of `EvidenceItem` (`{ kind, description, artifact?, link? }`) with `EvidenceKind` ∈ `test-result | test-coverage | manual-spot-check | pre-post-diff | regression-suite | ac-mapping`
|
|
145
138
|
- `confidence` — `high | medium | low`
|
|
146
139
|
|
|
147
|
-
The service enforces that **all 4 dimensions are present**; a missing dimension throws `IncompleteFinalReviewError` and the call is a gate failure. Treat `allPass === true` + empty `needsAttention` as a clean handoff to the human.
|
|
140
|
+
The service enforces that **all 4 dimensions are present**; a missing dimension throws `IncompleteFinalReviewError` and the call is a gate failure. Treat `allPass === true` + empty `needsAttention` as a clean handoff to the human **only where it is reachable at all**: if the pre/post baseline for dimension 4 cannot be **computed** at all (see dimension 4 below), that dimension is permanently `inconclusive` and `allPass` is structurally `false` on every run — a `false` there means "no comparison was shown to the reviewer", not "the work regressed". The same delivery rule applies to **dimension 1's approved-scope contract** (`prd/handoff.md`): a contract that is absent, empty, unreadable, or too large to reach the reviewer costs that dimension its `pass` (a `scope-contract-gate` marker says so in the summary). Read both as "the reviewer was not given the document this verdict needs", never as "the work regressed". A computed baseline or a present contract is no longer at the mercy of the byte budget: both sources hold a floor in the allocator, so the budget-starved delivery failures left are the ones with no document behind them. One failure is NOT the budget's to fix, and the service says so in as many words: when **every** source on disk that supports a dimension is larger than the per-file cap (`MAX_EVIDENCE_BYTES_PER_FILE`, 10,240 bytes — derived from the total, so raising the total does not raise it), that dimension can never be delivered — not on this run and not on any run. The reviewer gets a `## Evidence delivery reachability (structural)` block naming the source and its size, and the dimension's summary carries a `delivery-reachability` marker with the same arithmetic. Read that `inconclusive` as "no evidence for this dimension can reach the reviewer", never as "the reviewer was unsure". Two cases worth naming: `needsAttention` is populated by the SERVICE too, not only by the verdicts — a delivered baseline whose own `VERDICT: ` line reports `STRUCTURAL DRIFT DETECTED` puts dimension 4 in `needsAttention` and clears `allPass` even when the reviewer passed it, because a detected removal is not "nothing needs attention"; and every dimension named by the reachability block above lands there too (it can never be `pass`), so the field tells you *why* it is red. And anything else — `allPass === false`, a `fail` verdict, or an `inconclusive` verdict — must come back to the LLM loop with a re-prompt (do not ask the human to interpret raw LLM output).
|
|
148
141
|
|
|
149
142
|
## The 4 dimensions (one-line summary)
|
|
150
143
|
|
|
151
144
|
Full evidence contract per dimension: `references/4-dimensions.md`.
|
|
152
145
|
|
|
153
|
-
1. **functional-completeness** — every AC from the approved audit-goal maps to a passing test (`evidence.kind === 'ac-mapping'` + `test-result`).
|
|
146
|
+
1. **functional-completeness** — every AC from the approved audit-goal maps to a passing test (`evidence.kind === 'ac-mapping'` + `test-result`), and the approved-scope contract (`prd/handoff.md`) was delivered to the reviewer in full.
|
|
154
147
|
2. **problem-resolution** — there is a targeted test for the original problem case (`evidence.kind === 'test-result'` against the original repro).
|
|
155
148
|
3. **no-new-bugs** — the regression suite is green AND the LLM surfaces 0 net-new failures (`evidence.kind === 'regression-suite'` + `manual-spot-check`).
|
|
156
149
|
4. **existing-functionality-intact** — a pre/post baseline diff (test count, public API surface, key behavior) shows no unintended drift (`evidence.kind === 'pre-post-diff'`).
|
|
157
150
|
|
|
151
|
+
> **Status of this dimension — updated 2026-09-12 (the `pre-post-diff` producer now ships).**
|
|
152
|
+
> The producer exists. `peaks prepare-final-review` runs a read-only git comparison of a base ref
|
|
153
|
+
> against the working tree and writes `.peaks/_runtime/<sessionId>/final-review/api-diff.txt`
|
|
154
|
+
> — the test-file and non-test `.ts` / `.tsx` source-file lists, the `it(` / `test(` case counts
|
|
155
|
+
> before/after, and the added/removed top-level `export` names over changed `.ts` / `.tsx` files —
|
|
156
|
+
> and maps that artifact into this dimension's `supports`. The service also attaches it to the
|
|
157
|
+
> dimension as an `EvidenceItem` of kind `pre-post-diff` whose `artifact` points at that file.
|
|
158
|
+
> The dimension can reach `pass` — when a baseline could actually be computed **and was delivered to the
|
|
159
|
+
> reviewer**. Both halves are required. The source holds a floor in the evidence allocation, so a
|
|
160
|
+
> budget that runs out no longer drops it; if it ever is dropped the reviewer's prompt says
|
|
161
|
+
> `STATUS: COMPUTED ON DISK, NOT DELIVERED`
|
|
162
|
+
> instead of claiming a comparison it does not carry, the dimension's `pass` is downgraded to
|
|
163
|
+
> `inconclusive`, and no `pre-post-diff` `EvidenceItem` is attached to it. A baseline the reviewer never
|
|
164
|
+
> saw is not evidence, exactly as a baseline that was never computed is not — the verdict may not
|
|
165
|
+
> outlive the evidence that was actually handed over. Two properties make that judgement structural
|
|
166
|
+
> rather than arithmetic: a source is delivered only when the bytes the reviewer received carry its
|
|
167
|
+
> conclusion (for this artifact, its opening `VERDICT:` line), and the delivered conclusion is then
|
|
168
|
+
> READ — a `STRUCTURAL DRIFT DETECTED` line puts this dimension in `needsAttention` and clears
|
|
169
|
+
> `allPass` even when the reviewer answered `pass`, because a detected removal is a question for a
|
|
170
|
+
> human, not a clean handoff.
|
|
171
|
+
>
|
|
172
|
+
> **A baseline that cannot be computed is never invented.** No base ref resolving, a base ref that
|
|
173
|
+
> resolves to HEAD itself (an empty range — what a shallow clone's `merge-base` produces on its
|
|
174
|
+
> default path), or a project that is not a git work tree, all mean the same thing: no artifact is
|
|
175
|
+
> written, the reviewer is told why in the prompt, and a `pass` on this dimension is downgraded to
|
|
176
|
+
> `inconclusive` — for EVERY one of those causes, with no exemptions. "This project keeps no
|
|
177
|
+
> baseline" is the CAUSE of the missing evidence; it is not a reason to trust the claim the
|
|
178
|
+
> evidence was supposed to support, and a permanently-`inconclusive` dimension is the honest
|
|
179
|
+
> reading of a comparison that never happened. The base ref comes from `--base <ref>`; the default
|
|
180
|
+
> is the merge-base with `origin/HEAD`, then `origin/main`, then `origin/master`, then `HEAD~1`,
|
|
181
|
+
> and when none of those resolve the reason says so and names `--base` as the way out.
|
|
182
|
+
>
|
|
183
|
+
> **What that means for `allPass`.** On a project that is not a git work tree — or where no base ref
|
|
184
|
+
> resolves — this dimension is permanently `inconclusive`, so `allPass === true` can never be reached
|
|
185
|
+
> there, however complete the other nine evidence sources are. That is the intended reading and not a
|
|
186
|
+
> bug to work around: nothing was ever compared, so there is no answer to hand over. Likewise, a
|
|
187
|
+
> project whose evidence set outgrows the reviewer's input budget leaves this block OMITTED, and the
|
|
188
|
+
> honest envelope then says `inconclusive` rather than asserting a comparison the reviewer was never
|
|
189
|
+
> shown.
|
|
190
|
+
>
|
|
191
|
+
> **Boundary:** the export comparison is a line-anchored regex, not a type checker — it detects a
|
|
192
|
+
> removed or renamed export and cannot detect a changed signature. Type-only exports count, both
|
|
193
|
+
> spellings (`export type { T }` and `export { type T }`). File-level DELETION is visible, because
|
|
194
|
+
> it is read from the file lists rather than inferred from the counts — a module with no
|
|
195
|
+
> `export ` line and no `it(` is still reported when it is deleted. And the reverse direction is
|
|
196
|
+
> guarded too: a name that appears on both sides of the diff, a `git mv`, and an `it(` ->
|
|
197
|
+
> `test.each(` conversion are reported as changes, not as removals. The artifact carries a
|
|
198
|
+
> wall-clock timestamp, so two runs over the same base differ on that line and on nothing else.
|
|
199
|
+
> And still do **not** "fix" a red dimension by re-mapping `qa/test-reports` into its `supports`:
|
|
200
|
+
> that turns the gate green without producing the baseline diff the definition above requires.
|
|
201
|
+
|
|
158
202
|
## Human's role
|
|
159
203
|
|
|
160
204
|
The human reviews evidence, **judges business outcomes (NOT code)**. The LLM produces structured evidence; the human's job is to:
|
|
161
205
|
|
|
162
|
-
- confirm that `allPass === true` corresponds to the business outcome they actually want (not just "tests are green");
|
|
206
|
+
- confirm that `allPass === true` corresponds to the business outcome they actually want (not just "tests are green", and — for dimension 4 — not just "a diff file exists somewhere on disk");
|
|
163
207
|
- decide what to do with `needsAttention` items — accept the LLM's verdict, override a `pass` to `fail` when the evidence is weak, or send the slice back to RD with a re-prompt;
|
|
164
208
|
- gate the release / archive action based on `allPass` + their own business review, not just on the LLM signal.
|
|
165
209
|
|
|
@@ -185,7 +229,7 @@ When handing off, emit: rid, `allPass`, `needsAttention[]`, output path, source
|
|
|
185
229
|
| `src/services/final-review/final-review-service.ts` | Authoritative service implementation. |
|
|
186
230
|
| `src/services/final-review/final-review-types.ts` | `FinalReviewOutput`, `DimensionEvidence`, verdict/evidence/confidence enums. |
|
|
187
231
|
| `src/services/audit/audit-goal-service.ts:16` | Line of evidence that `LlmRunner` is reusable across audit + final-review (service-level integration). |
|
|
188
|
-
| `tests/unit/final-review/final-review-service.test.ts` |
|
|
232
|
+
| `tests/unit/final-review/final-review-service.test.ts` | Service-level unit tests (8 cases: evidence inlining, the no-evidence⇒no-`pass` gate, prompt bounds, plus contract guards). |
|
|
189
233
|
| `docs/superpowers/plans/2026-06-25-slice-topology-multipass-phase-4.md:127` | Phase-4 plan prose (Task 14). |
|
|
190
234
|
| `skills/peaks-qa/SKILL.md` | Upstream QA skill — 4-dim review is downstream of all QA gates. |
|
|
191
235
|
| `skills/peaks-audit/SKILL.md` | Sibling skill — produces the `audit-goal` JSON that this skill consumes. |
|
|
@@ -20,10 +20,26 @@ Every acceptance criterion in the approved audit-goal at `.peaks/_runtime/<sessi
|
|
|
20
20
|
|
|
21
21
|
### Verdict semantics
|
|
22
22
|
|
|
23
|
-
- `pass` — every AC has a passing test
|
|
23
|
+
- `pass` — every AC has a passing test, the test suite is green, **and the approved-scope contract was delivered**: the `prd/handoff.md` source block reached the reviewer with its WHOLE byte count (`FOUND at … — N bytes`), because "complete" is defined against the approved scope and its non-goals, and a test report alone shows only that *something* was built.
|
|
24
24
|
- `fail` — one or more ACs are unmapped, the targeted test is failing, or the suite is red on a non-flaky ground.
|
|
25
25
|
- `inconclusive` (needs-human) — the mapping is plausible but the human needs to confirm that a passing test truly reflects the business intent of an AC. Example: a test exists for "config-service splits into 3 modules" but the human must decide whether the *seam* the test exercises is the seam they actually wanted.
|
|
26
26
|
|
|
27
|
+
### When the scope contract does not arrive — same rule as dimension 4
|
|
28
|
+
|
|
29
|
+
The service keys this dimension's `pass` on the **delivery** of its scope contract, not on its existence on disk, for the same reason dimension 4 keys its `pass` on the delivered baseline:
|
|
30
|
+
|
|
31
|
+
| Situation | What the reviewer is told | Verdict the service allows |
|
|
32
|
+
|---|---|---|
|
|
33
|
+
| `prd/handoff.md` inlined in full | `STATUS: FOUND at … — N bytes` | `pass` is available |
|
|
34
|
+
| `prd/handoff.md` present but the budget did not reach it, or the read was truncated | `STATUS: MISSING (omitted)`, or `TRUNCATED, showing the first N of M bytes` | `pass` is downgraded to `inconclusive`, with a `scope-contract-gate` marker in the summary |
|
|
35
|
+
| `prd/handoff.md` exists for this run but carries nothing (0 bytes / whitespace) | `STATUS: MISSING (empty)` | Same downgrade: the PRD phase ran and the reviewer was given no contract. An empty contract is a delivery failure, not an absent phase. |
|
|
36
|
+
| `prd/handoff.md` exists but this process could not read it (EACCES / EBUSY / EISDIR) | `STATUS: UNREADABLE` | Same downgrade. Deliberately NOT reported as "no PRD phase": the file is there, the read failed, and only one of those two facts is fixable. |
|
|
37
|
+
| No `prd/handoff.md` in this project at all — ENOENT (no PRD phase) | `STATUS: MISSING (missing)` | Not a delivery failure — the dimension is judged on the evidence that exists. Absence is not the same fact as non-delivery. |
|
|
38
|
+
|
|
39
|
+
The delivery rule is deliberately not "some bytes arrived": `qa-test-report` also supports this dimension, which is exactly how a `pass` used to survive while the contract that defines the dimension was inlined with **zero** bytes. See `enforceScopeContractDelivery()` in `src/services/final-review/final-review-service.ts`.
|
|
40
|
+
|
|
41
|
+
The contract source holds a **floor** in the byte allocator: its whole unit is reserved for this dimension, so a saturated run can no longer drop it as a side effect of the allocation order. The downgrade rows above are therefore about a document that is genuinely absent, empty or unreadable — not about a contract that merely happened to sit last in the order.
|
|
42
|
+
|
|
27
43
|
### Example
|
|
28
44
|
|
|
29
45
|
> `dimension: "functional-completeness"`, `verdict: "pass"`, `summary: "All 3 success criteria from the approved goal are covered by passing tests. AC-1 covered by config-service.modules.test.ts; AC-2 covered by config-service.api.test.ts (public API snapshot unchanged); AC-3 covered by coverage report at 100% lines/branches for the changed files."`, `evidence: [...]`, `confidence: "high"`.
|
|
@@ -92,10 +108,30 @@ A pre/post baseline diff shows no unintended drift in the test surface, public A
|
|
|
92
108
|
|
|
93
109
|
### Verdict semantics
|
|
94
110
|
|
|
95
|
-
- `pass` — every measured dimension (tests, API, behavior) is unchanged or changed only in ways that the slice was explicitly authorized to change (e.g. AC-2 says "add a new exported helper `resolveWithSchema()`" — that IS the authorized change).
|
|
111
|
+
- `pass` — the baseline was **compared and delivered**: the pre/post diff block reached the reviewer **with its conclusion** (`STATUS: FOUND`, and the `VERDICT:` line is in the delivered bytes — the verdict is the first thing in the artifact, so an over-cap slice still carries it), and every measured dimension (tests, API, behavior) is unchanged or changed only in ways that the slice was explicitly authorized to change (e.g. AC-2 says "add a new exported helper `resolveWithSchema()`" — that IS the authorized change).
|
|
96
112
|
- `fail` — an unauthorized change slipped in: an exported symbol disappeared, a test was deleted rather than updated, a behavior baseline drifted without a corresponding AC.
|
|
97
113
|
- `inconclusive` (needs-human) — a change is present that *could* be authorized drift or *could* be an unintentional regression. The human rules.
|
|
98
114
|
|
|
115
|
+
### When the baseline itself is missing — the one case where no verdict can be earned
|
|
116
|
+
|
|
117
|
+
A `pass` here rests on a **comparison**, so two things must both be true: the producer had to *compute* a baseline, and that baseline had to *reach the reviewer's prompt* — conclusion included. Neither substitutes for the other.
|
|
118
|
+
|
|
119
|
+
| Situation | What the reviewer is told | Verdict the service allows |
|
|
120
|
+
|---|---|---|
|
|
121
|
+
| Baseline computed and inlined | `STATUS: COMPUTED` + the diff block | `pass` is available |
|
|
122
|
+
| Baseline computed, dropped by the evidence budget | `STATUS: COMPUTED ON DISK, NOT DELIVERED` + the block marked `MISSING (omitted)` | `pass` is downgraded to `inconclusive` |
|
|
123
|
+
| Baseline inlined but larger than the per-file cap | `STATUS: FOUND … TRUNCATED, showing the first N of M bytes` — the slice still opens with the `VERDICT:` line | `pass` is available (the conclusion is in the delivered bytes) |
|
|
124
|
+
| No base ref resolves / base resolves to HEAD (empty range) | `STATUS: UNAVAILABLE` + the reason | `pass` is downgraded to `inconclusive` |
|
|
125
|
+
| The project is not a git work tree | `STATUS: UNAVAILABLE` + the reason | `pass` is downgraded to `inconclusive` — permanently, on every run |
|
|
126
|
+
|
|
127
|
+
The "dropped by the evidence budget" row is now a defensive branch rather than the expected failure: this source holds a **floor** in the allocator — its whole unit is reserved for this dimension — so when a baseline IS computed it is delivered, and the reviewer is never asked to judge this dimension against a comparison it was not shown. The rows that still fire are the ones where no baseline exists to deliver.
|
|
128
|
+
|
|
129
|
+
Three consequences worth stating plainly, because all three read like defects and are not:
|
|
130
|
+
|
|
131
|
+
- **A non-git project can never reach `allPass === true`.** Dimension 4 is permanently `inconclusive` there: no comparison exists to hand over, so there is no honest way to have it green. The service does not invent one, and a design-intent document (`rd/tech-doc.md`, `prd/handoff.md`) is not a substitute — it states what was intended, not what changed. There is no CLI way out either: `--base <ref>` names a COMMIT to compare against, so it cannot help a project that has no git history to resolve a ref in — the reason the reviewer is given is the project's state, not a missing argument.
|
|
132
|
+
- **A dimension whose evidence was omitted does not get a pass "because the file exists".** If the block is not in the prompt, the reviewer never saw it; the service downgrades the verdict and does **not** attach the artifact as an `EvidenceItem`, because an envelope citing evidence the reviewer never received is the forged clean handoff this gate exists to prevent.
|
|
133
|
+
- **A saturated run can redden more than one dimension, and that is the honest reading.** The evidence budget allocates each source WHOLE or not at all (never a partial slice), so a run whose sources do not all fit drops whole sources — and each dimension whose contract source was dropped loses its `pass`: dimension 4 when the baseline goes, dimension 1 when the approved-scope contract does.
|
|
134
|
+
|
|
99
135
|
### Example
|
|
100
136
|
|
|
101
137
|
> `dimension: "existing-functionality-intact"`, `verdict: "pass"`, `summary: "Public API snapshot: 0 symbols removed, 1 symbol added (resolveWithSchema — authorized by AC-2). Test count: +12 (new tests for resolveWithSchema), -3 (deleted tests for the old monolithic resolve that resolveWithSchema supersedes). CLI help text byte-identical to pre-fix golden."`, `evidence: [{ kind: "pre-post-diff", description: "Public API surface diff: +resolveWithSchema, -3 obsolete test files, no other deltas", artifact: ".peaks/_runtime/<sessionId>/final-review/api-diff.txt" }]`, `confidence: "high"`.
|
|
@@ -106,12 +142,13 @@ A pre/post baseline diff shows no unintended drift in the test surface, public A
|
|
|
106
142
|
|
|
107
143
|
The service contract (`src/services/final-review/final-review-service.ts:23-31` and `:88-93`):
|
|
108
144
|
|
|
109
|
-
- `allPass === true` iff every dimension's `verdict === 'pass'
|
|
110
|
-
- `needsAttention` is the list of dimension names whose verdict is `fail` or `inconclusive
|
|
145
|
+
- `allPass === true` iff every dimension's `verdict === 'pass'` **and** the service itself has nothing to flag. The two fields are derived from the verdicts, never copied from the model's own summary: a model that wrote a fabricated `pass` plus a matching `allPass: true` cannot hand over an unsupported clean review.
|
|
146
|
+
- `needsAttention` is the list of dimension names whose verdict is `fail` or `inconclusive`, **plus** any dimension the service has to flag mechanically even though the reviewer passed it. The one such flag today is a delivered pre/post baseline whose own `VERDICT:` line reports `STRUCTURAL DRIFT DETECTED` (or a verdict line the service cannot classify): a detected structural removal may well be authorized, but a review that says "4/4 pass, nothing needs attention" directly above a diff that reports a removal it attached itself is self-contradicting, so that dimension is listed and `allPass` is `false`. The dimension's verdict is left as the reviewer wrote it — the call on whether a removal was authorized stays with the reviewer and the human. A second, related case is NOT a mechanical flag but a structural red the service explains: when **every** source on disk supporting a dimension is larger than the per-file cap (10,240 bytes, derived from `MAX_EVIDENCE_BYTES_TOTAL`), no source can ever be delivered under the `whole` rule, so the reviewer is told so in a `## Evidence delivery reachability (structural)` block and the dimension's summary carries a `delivery-reachability` marker naming the source and its byte count. Such a dimension is `inconclusive` by byte arithmetic — never by having passed and been overruled — and it appears in `needsAttention` through the ordinary non-`pass` route, which is what makes the CLI envelope state the reason instead of leaving a permanent red unexplained.
|
|
147
|
+
- The LLM does NOT need to populate `needsAttention` — the service enforces presence of all 4 dimensions and derives the field.
|
|
111
148
|
- An `IncompleteFinalReviewError` is thrown when JSON is malformed or any required dimension is missing. That is a **gate failure**, not a `fail` verdict — the LLM call is invalid and must be re-prompted, not surfaced to the human.
|
|
112
149
|
|
|
113
150
|
## Confidence and what it means
|
|
114
151
|
|
|
115
152
|
- `high` — evidence is concrete (named test file, named artifact, deterministic run).
|
|
116
153
|
- `medium` — evidence is concrete but covers only part of the dimension; the LLM is being honest about coverage gaps.
|
|
117
|
-
- `inconclusive` verdicts should always be `medium` or `low` confidence; `high` confidence on `inconclusive` is a contradiction and the LLM should be re-prompted.
|
|
154
|
+
- `inconclusive` verdicts should always be `medium` or `low` confidence; `high` confidence on `inconclusive` is a contradiction and the LLM should be re-prompted. The service does not rely on the re-prompt alone: it clamps a `high` on an `inconclusive` verdict to `medium` and marks the summary, so the contradiction cannot reach the human in the envelope even if the model keeps producing it.
|