peaks-loop 4.0.43 → 4.0.44
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README-en.md +1 -1
- package/README.md +1 -1
- package/dist/cli/commands/codegraph-commands.js +191 -6
- package/dist/cli/commands/final-review-commands.d.ts +34 -10
- package/dist/cli/commands/final-review-commands.js +130 -34
- package/dist/cli/commands/share-commands.d.ts +49 -0
- package/dist/cli/commands/share-commands.js +114 -14
- package/dist/services/codegraph/codegraph-autorefresh.js +12 -0
- package/dist/services/codegraph/codegraph-exclude-integrity.d.ts +61 -0
- package/dist/services/codegraph/codegraph-exclude-integrity.js +98 -0
- package/dist/services/codegraph/codegraph-exclude-reconciler.d.ts +26 -0
- package/dist/services/codegraph/codegraph-exclude-reconciler.js +217 -0
- package/dist/services/codegraph/codegraph-exclude-repair.d.ts +102 -0
- package/dist/services/codegraph/codegraph-exclude-repair.js +266 -0
- package/dist/services/codegraph/codegraph-preflight-service.js +12 -0
- package/dist/services/codegraph/codegraph-service.d.ts +0 -1
- package/dist/services/codegraph/codegraph-service.js +5 -4
- package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.d.ts +29 -0
- package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.js +88 -0
- package/dist/services/doctor/doctor-service/plugin-registry.js +2 -0
- package/dist/services/doctor/doctor-service/types.d.ts +27 -0
- package/dist/services/final-review/final-review-service.d.ts +154 -0
- package/dist/services/final-review/final-review-service.js +621 -7
- package/dist/services/final-review/index.d.ts +1 -1
- package/dist/services/final-review/index.js +1 -1
- package/dist/services/prd/handoff-auto-regen.js +0 -1
- package/dist/services/prd/handoff-service.d.ts +9 -1
- package/dist/services/prd/handoff-service.js +48 -6
- package/package.json +7 -5
- package/skills/peaks-final-review/SKILL.md +43 -32
|
@@ -47,7 +47,15 @@ export declare function writeHandoff(handoff: Handoff, projectRoot: string): Pro
|
|
|
47
47
|
* malformed frontmatter. */
|
|
48
48
|
export declare function readHandoff(filePath: string): Promise<Handoff>;
|
|
49
49
|
/** Verify a handoff by re-reading and re-hashing. Returns a probe
|
|
50
|
-
* (never throws on hash mismatch — that's the outcome).
|
|
50
|
+
* (never throws on hash mismatch — that's the outcome).
|
|
51
|
+
*
|
|
52
|
+
* N1: the two failure classes are reported separately. The previous
|
|
53
|
+
* single `catch` folded every read failure into `file-missing`, so a
|
|
54
|
+
* handoff that WAS on disk but whose frontmatter the parser refused was
|
|
55
|
+
* reported as absent — sending an operator to look for a missing file
|
|
56
|
+
* that was right there. Only a genuine read failure (ENOENT, EACCES, …)
|
|
57
|
+
* is `file-missing` now; anything the parser rejects is
|
|
58
|
+
* `frontmatter-malformed`. */
|
|
51
59
|
export declare function verifyHandoff(filePath: string): Promise<HandoffProbe>;
|
|
52
60
|
/** Return the raw markdown content of a handoff file (frontmatter +
|
|
53
61
|
* body verbatim). For human display. */
|
|
@@ -66,15 +66,30 @@ export async function readHandoff(filePath) {
|
|
|
66
66
|
return parseHandoffContent(content);
|
|
67
67
|
}
|
|
68
68
|
/** Verify a handoff by re-reading and re-hashing. Returns a probe
|
|
69
|
-
* (never throws on hash mismatch — that's the outcome).
|
|
69
|
+
* (never throws on hash mismatch — that's the outcome).
|
|
70
|
+
*
|
|
71
|
+
* N1: the two failure classes are reported separately. The previous
|
|
72
|
+
* single `catch` folded every read failure into `file-missing`, so a
|
|
73
|
+
* handoff that WAS on disk but whose frontmatter the parser refused was
|
|
74
|
+
* reported as absent — sending an operator to look for a missing file
|
|
75
|
+
* that was right there. Only a genuine read failure (ENOENT, EACCES, …)
|
|
76
|
+
* is `file-missing` now; anything the parser rejects is
|
|
77
|
+
* `frontmatter-malformed`. */
|
|
70
78
|
export async function verifyHandoff(filePath) {
|
|
71
|
-
let
|
|
79
|
+
let content;
|
|
72
80
|
try {
|
|
73
|
-
|
|
81
|
+
content = await readFile(filePath, 'utf8');
|
|
74
82
|
}
|
|
75
83
|
catch {
|
|
76
84
|
return { ok: false, reason: 'file-missing' };
|
|
77
85
|
}
|
|
86
|
+
let handoff;
|
|
87
|
+
try {
|
|
88
|
+
handoff = parseHandoffContent(content);
|
|
89
|
+
}
|
|
90
|
+
catch {
|
|
91
|
+
return { ok: false, reason: 'frontmatter-malformed' };
|
|
92
|
+
}
|
|
78
93
|
if (handoff.frontmatter.schemaVersion !== HANDOFF_SCHEMA_VERSION) {
|
|
79
94
|
return {
|
|
80
95
|
ok: false,
|
|
@@ -119,20 +134,47 @@ function parseHandoffContent(content) {
|
|
|
119
134
|
if (!isHandoffFrontmatter(parsed)) {
|
|
120
135
|
throw new Error('handoff: frontmatter shape validation failed');
|
|
121
136
|
}
|
|
122
|
-
|
|
137
|
+
// N1: normalize the version to its canonical string form. `schemaVersion: 2`
|
|
138
|
+
// (bare) is valid YAML that parses to the NUMBER 2; `schemaVersion: '2'` is
|
|
139
|
+
// what `stringifyYaml` writes. Both mean schema version 2, so the parsed
|
|
140
|
+
// frontmatter is returned with the canonical `'2'` rather than the raw scalar
|
|
141
|
+
// — otherwise every downstream `=== '2'` comparison would depend on which
|
|
142
|
+
// producer wrote the file.
|
|
143
|
+
return {
|
|
144
|
+
frontmatter: { ...parsed, schemaVersion: HANDOFF_SCHEMA_VERSION },
|
|
145
|
+
body
|
|
146
|
+
};
|
|
123
147
|
}
|
|
124
148
|
function serializeHandoff(handoff) {
|
|
125
149
|
const yamlStr = stringifyYaml(handoff.frontmatter).trimEnd();
|
|
126
150
|
return `---\n${yamlStr}\n---\n${handoff.body}`;
|
|
127
151
|
}
|
|
152
|
+
/**
|
|
153
|
+
* N1: accept BOTH shapes of `schemaVersion` — the string `'2'` and the bare
|
|
154
|
+
* YAML number `2` — and reject any other value.
|
|
155
|
+
*
|
|
156
|
+
* The reader used to require `typeof v.schemaVersion === 'string'`. That made
|
|
157
|
+
* `readHandoff` refuse `prd/handoff.md` written by `handoff-auto-regen.ts`,
|
|
158
|
+
* which emits the unquoted `schemaVersion: 2`, while the
|
|
159
|
+
* `AUDIT_REQUIRES_HANDOFF` prereq — a SUBSTRING check for `schemaVersion: 2` —
|
|
160
|
+
* happily passed the same bytes. So the gate that exists to guarantee a
|
|
161
|
+
* readable handoff was satisfied by a handoff the parser would not read, and
|
|
162
|
+
* `peaks prd handoff verify` exited 1 on a healthy file.
|
|
163
|
+
*
|
|
164
|
+
* Quoting the writer instead is NOT a fix: it would delete the very substring
|
|
165
|
+
* the prereq pins, turning a broken read into a broken gate. The tolerant read
|
|
166
|
+
* is the only change that satisfies both consumers.
|
|
167
|
+
*/
|
|
168
|
+
function isSchemaVersion2(value) {
|
|
169
|
+
return value === HANDOFF_SCHEMA_VERSION || value === 2;
|
|
170
|
+
}
|
|
128
171
|
function isHandoffFrontmatter(value) {
|
|
129
172
|
if (!value || typeof value !== 'object')
|
|
130
173
|
return false;
|
|
131
174
|
const v = value;
|
|
132
175
|
return (typeof v.requestId === 'string' &&
|
|
133
176
|
typeof v.sessionId === 'string' &&
|
|
134
|
-
|
|
135
|
-
typeof v.schemaVersion === 'string' &&
|
|
177
|
+
isSchemaVersion2(v.schemaVersion) &&
|
|
136
178
|
typeof v.handoffHash === 'string' &&
|
|
137
179
|
typeof v.writtenAt === 'string' &&
|
|
138
180
|
Array.isArray(v.goals) &&
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "peaks-loop",
|
|
3
|
-
"version": "4.0.
|
|
3
|
+
"version": "4.0.44",
|
|
4
4
|
"description": "Loop Engineering CLI — workflow primitive / loop guards / evaluators / slice orchestration",
|
|
5
5
|
"author": "SquabbyZ",
|
|
6
6
|
"keywords": [
|
|
@@ -99,12 +99,13 @@
|
|
|
99
99
|
"better-sqlite3": "^12.11.1",
|
|
100
100
|
"commander": "^12.1.0",
|
|
101
101
|
"fzf": "^0.5.2",
|
|
102
|
+
"picomatch": "4.0.4",
|
|
102
103
|
"yaml": "^2.9.0",
|
|
103
104
|
"zod": "^4.4.3",
|
|
104
|
-
"peaks-loop-
|
|
105
|
-
"peaks-loop-
|
|
106
|
-
"peaks-loop-
|
|
107
|
-
"peaks-loop-shared
|
|
105
|
+
"peaks-loop-internal-runtime": "0.0.29",
|
|
106
|
+
"peaks-loop-mut": "0.1.42",
|
|
107
|
+
"peaks-loop-shared-channel": "0.0.46",
|
|
108
|
+
"peaks-loop-shared": "0.0.78"
|
|
108
109
|
},
|
|
109
110
|
"devDependencies": {
|
|
110
111
|
"@changesets/cli": "2.31.1",
|
|
@@ -114,6 +115,7 @@
|
|
|
114
115
|
"@stryker-mutator/vitest-runner": "^9.6.1",
|
|
115
116
|
"@types/better-sqlite3": "^7.6.13",
|
|
116
117
|
"@types/node": "^22.10.2",
|
|
118
|
+
"@types/picomatch": "4.0.3",
|
|
117
119
|
"@typescript-eslint/eslint-plugin": "8.66.0",
|
|
118
120
|
"@typescript-eslint/parser": "8.66.0",
|
|
119
121
|
"@vitest/coverage-istanbul": "^4.1.10",
|
|
@@ -69,7 +69,9 @@ interface PrepareFinalReviewOptions {
|
|
|
69
69
|
}
|
|
70
70
|
```
|
|
71
71
|
|
|
72
|
-
The service is the **gate primitive** that closes the 10% human / 90% LLM loop. It reads the approved audit-goal JSON from `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`,
|
|
72
|
+
The service is the **gate primitive** that closes the 10% human / 90% LLM loop. It reads the approved audit-goal JSON from `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`, **collects real evidence from disk** (`qa/test-reports`, `qa/test-cases`, `qa/security-findings`, `qa/performance-findings`, `rd/{tech-doc,bug-analysis,code-review,security-review}.md`, `prd/handoff.md`), inlines it into the 4-dim review prompt under byte bounds, calls an injected `LlmRunner` exactly once, parses the response, and validates that all four required dimensions are present. It throws `IncompleteFinalReviewError` on malformed JSON or missing dimensions — callers MUST treat that as a gate failure (return to human for re-prompting) and never let autonomous work proceed on a partial review.
|
|
73
|
+
|
|
74
|
+
> **Evidence-backed verdicts (added 2026-09-12).** The service previously sent the model only `successCriteria` and no evidence at all, so it could not honestly grade anything — a real run returned 4/4 `inconclusive`, and a model willing to confabulate could have returned `pass`. Verdicts are now gated **structurally**, not by prompt wording: any `pass` whose supporting sources were all missing/empty is rewritten to `inconclusive` + `confidence: low` (`fail` is never softened), and `allPass` is derived from the gated verdicts so it can only narrow. Every non-`found` evidence source renders an explicit `STATUS: MISSING (<why>)` in the prompt, so the model always knows what it does not know.
|
|
73
75
|
|
|
74
76
|
> The `LlmRunner` interface is intentionally minimal so this service reuses the same provider injection seam as `audit-goal-service` and the slice LLMArbitrator (`src/services/audit/audit-goal-service.ts:16`). No provider implementation is baked in at this layer.
|
|
75
77
|
|
|
@@ -85,41 +87,32 @@ All of the following MUST be true before invoking this skill:
|
|
|
85
87
|
|
|
86
88
|
If any precondition is missing, **STOP** and route back to the responsible skill. Do not paper over a missing artifact with a hand-written successCriteria — the review must reflect what the human originally approved.
|
|
87
89
|
|
|
88
|
-
## Invocation
|
|
90
|
+
## Invocation
|
|
89
91
|
|
|
90
|
-
> **
|
|
91
|
-
>
|
|
92
|
-
>
|
|
93
|
-
>
|
|
94
|
-
>
|
|
95
|
-
>
|
|
96
|
-
> **This CLI command does NOT yet exist.** `prepareFinalReview()` is implemented as a service in `src/services/final-review/final-review-service.ts` and is unit-tested in `tests/unit/final-review/final-review-service.test.ts`, but it is **not wired to a CLI subcommand**. A `peaks prepare-final-review` subcommand is the planned forward-looking surface; the integration sits at the service layer, not the CLI layer, today.
|
|
97
|
-
>
|
|
98
|
-
> **Pick for the future CLI wrapper file (audit recommendation):** create a new `src/cli/commands/final-review-commands.ts` matching the `peaks-<group>-commands.ts` naming convention (`qa-commands.ts`, `code-review-commands.ts`, `audit-commands.ts`). A grep of `src/cli/commands/` confirms NO `final-review-commands.ts` and NO `prepareFinalReview` import in any CLI file. Rationale: the 4-dim business review is conceptually distinct from `peaks qa *` (which is autonomous gate verification) — it is the human-acceptance terminator, not an internal gate. A separate command group preserves that boundary.
|
|
99
|
-
|
|
100
|
-
### Current call path (today, until a CLI wrapper is built)
|
|
101
|
-
|
|
102
|
-
```ts
|
|
103
|
-
// peaks-code end-of-workflow, peaks-txt, or any other hand-rolled caller
|
|
104
|
-
import { prepareFinalReview } from './src/services/final-review/final-review-service.js';
|
|
105
|
-
import { auditGoalLlmRunner } from './src/services/audit/llm-runner.js'; // or your provider
|
|
106
|
-
|
|
107
|
-
const out = await prepareFinalReview('<rid>', {
|
|
108
|
-
projectRoot: '<absolute path>',
|
|
109
|
-
sessionId: '<sessionId>',
|
|
110
|
-
llmRunner: auditGoalLlmRunner
|
|
111
|
-
});
|
|
112
|
-
```
|
|
113
|
-
|
|
114
|
-
### Planned call path (after the CLI wrapper lands)
|
|
92
|
+
> **Correction (2026-09-12).** An earlier revision of this file claimed `peaks prepare-final-review`
|
|
93
|
+
> "does NOT yet exist" and told callers to hand-roll a `prepareFinalReview()` caller instead.
|
|
94
|
+
> **That was wrong.** The CLI wrapper exists and is registered —
|
|
95
|
+
> `src/cli/commands/final-review-commands.ts` (`W5 Fix M2`), command registered at its line ~130.
|
|
96
|
+
> The hand-rolled snippet that used to sit here also had the wrong flag shape (`--rid <rid>`);
|
|
97
|
+
> **the rid is positional**. Use the CLI.
|
|
115
98
|
|
|
116
99
|
```bash
|
|
117
|
-
|
|
118
|
-
# src/cli/commands/final-review-commands.ts
|
|
119
|
-
peaks prepare-final-review --rid <rid> [--session-id <sid>] --json
|
|
100
|
+
peaks prepare-final-review <rid> [--session-id <sid>] [--project <path>] [--llm-provider <name>] [--json]
|
|
120
101
|
```
|
|
121
102
|
|
|
122
|
-
|
|
103
|
+
- `<rid>` is **positional** (not `--rid`). It resolves to `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`.
|
|
104
|
+
- `--session-id` defaults to the active workspace binding when omitted.
|
|
105
|
+
- **`--llm-provider` defaults to `stub`, and `stub` is the only provider that works today.**
|
|
106
|
+
`stub` runs no real review — it returns a scaffold envelope so CI can prove the route is reachable.
|
|
107
|
+
Verified 2026-09-12: passing a real provider (`--llm-provider anthropic`) returns
|
|
108
|
+
`status: not-applicable` / `serviceWired: false` / `providerBinding: unknown` and tells you to re-run
|
|
109
|
+
with `stub`; the CLI's own nextAction calls real-provider binding "a follow-up slice". So there is
|
|
110
|
+
currently **no reachable path to a real 4-dim review** — confirming the route works is all `stub`
|
|
111
|
+
can do.
|
|
112
|
+
- `--json` is required for machine consumption (peaks-code, peaks-txt, downstream CI).
|
|
113
|
+
|
|
114
|
+
Calling the service directly (`prepareFinalReview(rid, { projectRoot, sessionId, llmRunner })`) remains
|
|
115
|
+
valid for callers that need a custom `LlmRunner` injection seam, but it is no longer the only path.
|
|
123
116
|
|
|
124
117
|
## Output
|
|
125
118
|
|
|
@@ -155,6 +148,24 @@ Full evidence contract per dimension: `references/4-dimensions.md`.
|
|
|
155
148
|
3. **no-new-bugs** — the regression suite is green AND the LLM surfaces 0 net-new failures (`evidence.kind === 'regression-suite'` + `manual-spot-check`).
|
|
156
149
|
4. **existing-functionality-intact** — a pre/post baseline diff (test count, public API surface, key behavior) shows no unintended drift (`evidence.kind === 'pre-post-diff'`).
|
|
157
150
|
|
|
151
|
+
> **⚠️ This dimension currently cannot pass — read before acting on it (verified 2026-09-12).**
|
|
152
|
+
> `pre-post-diff` is a declared `EvidenceKind`, but **nothing in peaks-loop produces that artifact**.
|
|
153
|
+
> The evidence actually mapped to this dimension is `rd/tech-doc.md` (design intent) and
|
|
154
|
+
> `prd/handoff.md` (scope / non-goals) — neither is a baseline diff, and the model correctly
|
|
155
|
+
> reports that ("the only FOUND source… is a design-intent document, not a regression assessment").
|
|
156
|
+
> `peaks scan api-diff <doc>` is *not* a producer: it diffs an API **document**, not the code surface.
|
|
157
|
+
>
|
|
158
|
+
> **Consequences:** `allPass === true` is **unreachable by construction**, for every workflow.
|
|
159
|
+
> This dimension will return `inconclusive` with an empty `evidence[]` even when the work is
|
|
160
|
+
> perfect. Treat that as a **tooling** state, not as evidence of a regression — and do **not**
|
|
161
|
+
> "fix" it by re-mapping `qa/test-reports` into this dimension's `supports`, which would turn the
|
|
162
|
+
> gate green without producing the baseline diff the definition above requires.
|
|
163
|
+
>
|
|
164
|
+
> **Real fix (unbuilt):** a producer for the pre/post baseline diff (test-count delta, public-API
|
|
165
|
+
> surface snapshot) written to `.peaks/_runtime/<sessionId>/final-review/api-diff.txt`, then mapped
|
|
166
|
+
> into this dimension's `supports`. Until that ships, `needsAttention` always contains this
|
|
167
|
+
> dimension — a permanently-red gate that reviewers will otherwise learn to ignore.
|
|
168
|
+
|
|
158
169
|
## Human's role
|
|
159
170
|
|
|
160
171
|
The human reviews evidence, **judges business outcomes (NOT code)**. The LLM produces structured evidence; the human's job is to:
|
|
@@ -185,7 +196,7 @@ When handing off, emit: rid, `allPass`, `needsAttention[]`, output path, source
|
|
|
185
196
|
| `src/services/final-review/final-review-service.ts` | Authoritative service implementation. |
|
|
186
197
|
| `src/services/final-review/final-review-types.ts` | `FinalReviewOutput`, `DimensionEvidence`, verdict/evidence/confidence enums. |
|
|
187
198
|
| `src/services/audit/audit-goal-service.ts:16` | Line of evidence that `LlmRunner` is reusable across audit + final-review (service-level integration). |
|
|
188
|
-
| `tests/unit/final-review/final-review-service.test.ts` |
|
|
199
|
+
| `tests/unit/final-review/final-review-service.test.ts` | Service-level unit tests (8 cases: evidence inlining, the no-evidence⇒no-`pass` gate, prompt bounds, plus contract guards). |
|
|
189
200
|
| `docs/superpowers/plans/2026-06-25-slice-topology-multipass-phase-4.md:127` | Phase-4 plan prose (Task 14). |
|
|
190
201
|
| `skills/peaks-qa/SKILL.md` | Upstream QA skill — 4-dim review is downstream of all QA gates. |
|
|
191
202
|
| `skills/peaks-audit/SKILL.md` | Sibling skill — produces the `audit-goal` JSON that this skill consumes. |
|