peaks-loop 4.0.43 → 4.0.44

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/dist/cli/commands/codegraph-commands.js +191 -6
  5. package/dist/cli/commands/final-review-commands.d.ts +34 -10
  6. package/dist/cli/commands/final-review-commands.js +130 -34
  7. package/dist/cli/commands/share-commands.d.ts +49 -0
  8. package/dist/cli/commands/share-commands.js +114 -14
  9. package/dist/services/codegraph/codegraph-autorefresh.js +12 -0
  10. package/dist/services/codegraph/codegraph-exclude-integrity.d.ts +61 -0
  11. package/dist/services/codegraph/codegraph-exclude-integrity.js +98 -0
  12. package/dist/services/codegraph/codegraph-exclude-reconciler.d.ts +26 -0
  13. package/dist/services/codegraph/codegraph-exclude-reconciler.js +217 -0
  14. package/dist/services/codegraph/codegraph-exclude-repair.d.ts +102 -0
  15. package/dist/services/codegraph/codegraph-exclude-repair.js +266 -0
  16. package/dist/services/codegraph/codegraph-preflight-service.js +12 -0
  17. package/dist/services/codegraph/codegraph-service.d.ts +0 -1
  18. package/dist/services/codegraph/codegraph-service.js +5 -4
  19. package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.d.ts +29 -0
  20. package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.js +88 -0
  21. package/dist/services/doctor/doctor-service/plugin-registry.js +2 -0
  22. package/dist/services/doctor/doctor-service/types.d.ts +27 -0
  23. package/dist/services/final-review/final-review-service.d.ts +154 -0
  24. package/dist/services/final-review/final-review-service.js +621 -7
  25. package/dist/services/final-review/index.d.ts +1 -1
  26. package/dist/services/final-review/index.js +1 -1
  27. package/dist/services/prd/handoff-auto-regen.js +0 -1
  28. package/dist/services/prd/handoff-service.d.ts +9 -1
  29. package/dist/services/prd/handoff-service.js +48 -6
  30. package/package.json +7 -5
  31. package/skills/peaks-final-review/SKILL.md +43 -32
@@ -47,7 +47,15 @@ export declare function writeHandoff(handoff: Handoff, projectRoot: string): Pro
47
47
  * malformed frontmatter. */
48
48
  export declare function readHandoff(filePath: string): Promise<Handoff>;
49
49
  /** Verify a handoff by re-reading and re-hashing. Returns a probe
50
- * (never throws on hash mismatch — that's the outcome). */
50
+ * (never throws on hash mismatch — that's the outcome).
51
+ *
52
+ * N1: the two failure classes are reported separately. The previous
53
+ * single `catch` folded every read failure into `file-missing`, so a
54
+ * handoff that WAS on disk but whose frontmatter the parser refused was
55
+ * reported as absent — sending an operator to look for a missing file
56
+ * that was right there. Only a genuine read failure (ENOENT, EACCES, …)
57
+ * is `file-missing` now; anything the parser rejects is
58
+ * `frontmatter-malformed`. */
51
59
  export declare function verifyHandoff(filePath: string): Promise<HandoffProbe>;
52
60
  /** Return the raw markdown content of a handoff file (frontmatter +
53
61
  * body verbatim). For human display. */
@@ -66,15 +66,30 @@ export async function readHandoff(filePath) {
66
66
  return parseHandoffContent(content);
67
67
  }
68
68
  /** Verify a handoff by re-reading and re-hashing. Returns a probe
69
- * (never throws on hash mismatch — that's the outcome). */
69
+ * (never throws on hash mismatch — that's the outcome).
70
+ *
71
+ * N1: the two failure classes are reported separately. The previous
72
+ * single `catch` folded every read failure into `file-missing`, so a
73
+ * handoff that WAS on disk but whose frontmatter the parser refused was
74
+ * reported as absent — sending an operator to look for a missing file
75
+ * that was right there. Only a genuine read failure (ENOENT, EACCES, …)
76
+ * is `file-missing` now; anything the parser rejects is
77
+ * `frontmatter-malformed`. */
70
78
  export async function verifyHandoff(filePath) {
71
- let handoff;
79
+ let content;
72
80
  try {
73
- handoff = await readHandoff(filePath);
81
+ content = await readFile(filePath, 'utf8');
74
82
  }
75
83
  catch {
76
84
  return { ok: false, reason: 'file-missing' };
77
85
  }
86
+ let handoff;
87
+ try {
88
+ handoff = parseHandoffContent(content);
89
+ }
90
+ catch {
91
+ return { ok: false, reason: 'frontmatter-malformed' };
92
+ }
78
93
  if (handoff.frontmatter.schemaVersion !== HANDOFF_SCHEMA_VERSION) {
79
94
  return {
80
95
  ok: false,
@@ -119,20 +134,47 @@ function parseHandoffContent(content) {
119
134
  if (!isHandoffFrontmatter(parsed)) {
120
135
  throw new Error('handoff: frontmatter shape validation failed');
121
136
  }
122
- return { frontmatter: parsed, body };
137
+ // N1: normalize the version to its canonical string form. `schemaVersion: 2`
138
+ // (bare) is valid YAML that parses to the NUMBER 2; `schemaVersion: '2'` is
139
+ // what `stringifyYaml` writes. Both mean schema version 2, so the parsed
140
+ // frontmatter is returned with the canonical `'2'` rather than the raw scalar
141
+ // — otherwise every downstream `=== '2'` comparison would depend on which
142
+ // producer wrote the file.
143
+ return {
144
+ frontmatter: { ...parsed, schemaVersion: HANDOFF_SCHEMA_VERSION },
145
+ body
146
+ };
123
147
  }
124
148
  function serializeHandoff(handoff) {
125
149
  const yamlStr = stringifyYaml(handoff.frontmatter).trimEnd();
126
150
  return `---\n${yamlStr}\n---\n${handoff.body}`;
127
151
  }
152
+ /**
153
+ * N1: accept BOTH shapes of `schemaVersion` — the string `'2'` and the bare
154
+ * YAML number `2` — and reject any other value.
155
+ *
156
+ * The reader used to require `typeof v.schemaVersion === 'string'`. That made
157
+ * `readHandoff` refuse `prd/handoff.md` written by `handoff-auto-regen.ts`,
158
+ * which emits the unquoted `schemaVersion: 2`, while the
159
+ * `AUDIT_REQUIRES_HANDOFF` prereq — a SUBSTRING check for `schemaVersion: 2` —
160
+ * happily passed the same bytes. So the gate that exists to guarantee a
161
+ * readable handoff was satisfied by a handoff the parser would not read, and
162
+ * `peaks prd handoff verify` exited 1 on a healthy file.
163
+ *
164
+ * Quoting the writer instead is NOT a fix: it would delete the very substring
165
+ * the prereq pins, turning a broken read into a broken gate. The tolerant read
166
+ * is the only change that satisfies both consumers.
167
+ */
168
+ function isSchemaVersion2(value) {
169
+ return value === HANDOFF_SCHEMA_VERSION || value === 2;
170
+ }
128
171
  function isHandoffFrontmatter(value) {
129
172
  if (!value || typeof value !== 'object')
130
173
  return false;
131
174
  const v = value;
132
175
  return (typeof v.requestId === 'string' &&
133
176
  typeof v.sessionId === 'string' &&
134
- typeof v.sessionId === 'string' &&
135
- typeof v.schemaVersion === 'string' &&
177
+ isSchemaVersion2(v.schemaVersion) &&
136
178
  typeof v.handoffHash === 'string' &&
137
179
  typeof v.writtenAt === 'string' &&
138
180
  Array.isArray(v.goals) &&
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "peaks-loop",
3
- "version": "4.0.43",
3
+ "version": "4.0.44",
4
4
  "description": "Loop Engineering CLI — workflow primitive / loop guards / evaluators / slice orchestration",
5
5
  "author": "SquabbyZ",
6
6
  "keywords": [
@@ -99,12 +99,13 @@
99
99
  "better-sqlite3": "^12.11.1",
100
100
  "commander": "^12.1.0",
101
101
  "fzf": "^0.5.2",
102
+ "picomatch": "4.0.4",
102
103
  "yaml": "^2.9.0",
103
104
  "zod": "^4.4.3",
104
- "peaks-loop-mut": "0.1.41",
105
- "peaks-loop-shared": "0.0.77",
106
- "peaks-loop-internal-runtime": "0.0.28",
107
- "peaks-loop-shared-channel": "0.0.45"
105
+ "peaks-loop-internal-runtime": "0.0.29",
106
+ "peaks-loop-mut": "0.1.42",
107
+ "peaks-loop-shared-channel": "0.0.46",
108
+ "peaks-loop-shared": "0.0.78"
108
109
  },
109
110
  "devDependencies": {
110
111
  "@changesets/cli": "2.31.1",
@@ -114,6 +115,7 @@
114
115
  "@stryker-mutator/vitest-runner": "^9.6.1",
115
116
  "@types/better-sqlite3": "^7.6.13",
116
117
  "@types/node": "^22.10.2",
118
+ "@types/picomatch": "4.0.3",
117
119
  "@typescript-eslint/eslint-plugin": "8.66.0",
118
120
  "@typescript-eslint/parser": "8.66.0",
119
121
  "@vitest/coverage-istanbul": "^4.1.10",
@@ -69,7 +69,9 @@ interface PrepareFinalReviewOptions {
69
69
  }
70
70
  ```
71
71
 
72
- The service is the **gate primitive** that closes the 10% human / 90% LLM loop. It reads the approved audit-goal JSON from `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`, calls an injected `LlmRunner` exactly once with a 4-dim review prompt, parses the response, and validates that all four required dimensions are present. It throws `IncompleteFinalReviewError` on malformed JSON or missing dimensions — callers MUST treat that as a gate failure (return to human for re-prompting) and never let autonomous work proceed on a partial review.
72
+ The service is the **gate primitive** that closes the 10% human / 90% LLM loop. It reads the approved audit-goal JSON from `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`, **collects real evidence from disk** (`qa/test-reports`, `qa/test-cases`, `qa/security-findings`, `qa/performance-findings`, `rd/{tech-doc,bug-analysis,code-review,security-review}.md`, `prd/handoff.md`), inlines it into the 4-dim review prompt under byte bounds, calls an injected `LlmRunner` exactly once, parses the response, and validates that all four required dimensions are present. It throws `IncompleteFinalReviewError` on malformed JSON or missing dimensions — callers MUST treat that as a gate failure (return to human for re-prompting) and never let autonomous work proceed on a partial review.
73
+
74
+ > **Evidence-backed verdicts (added 2026-09-12).** The service previously sent the model only `successCriteria` and no evidence at all, so it could not honestly grade anything — a real run returned 4/4 `inconclusive`, and a model willing to confabulate could have returned `pass`. Verdicts are now gated **structurally**, not by prompt wording: any `pass` whose supporting sources were all missing/empty is rewritten to `inconclusive` + `confidence: low` (`fail` is never softened), and `allPass` is derived from the gated verdicts so it can only narrow. Every non-`found` evidence source renders an explicit `STATUS: MISSING (<why>)` in the prompt, so the model always knows what it does not know.
73
75
 
74
76
  > The `LlmRunner` interface is intentionally minimal so this service reuses the same provider injection seam as `audit-goal-service` and the slice LLMArbitrator (`src/services/audit/audit-goal-service.ts:16`). No provider implementation is baked in at this layer.
75
77
 
@@ -85,41 +87,32 @@ All of the following MUST be true before invoking this skill:
85
87
 
86
88
  If any precondition is missing, **STOP** and route back to the responsible skill. Do not paper over a missing artifact with a hand-written successCriteria — the review must reflect what the human originally approved.
87
89
 
88
- ## Invocation (CLI wrapper status — READ FIRST)
90
+ ## Invocation
89
91
 
90
- > **Pre-flight finding (HARD):** The plan prose at `docs/superpowers/plans/2026-06-25-slice-topology-multipass-phase-4.md:146` documents the invocation as:
91
- >
92
- > ```bash
93
- > peaks prepare-final-review --rid <rid> --json
94
- > ```
95
- >
96
- > **This CLI command does NOT yet exist.** `prepareFinalReview()` is implemented as a service in `src/services/final-review/final-review-service.ts` and is unit-tested in `tests/unit/final-review/final-review-service.test.ts`, but it is **not wired to a CLI subcommand**. A `peaks prepare-final-review` subcommand is the planned forward-looking surface; the integration sits at the service layer, not the CLI layer, today.
97
- >
98
- > **Pick for the future CLI wrapper file (audit recommendation):** create a new `src/cli/commands/final-review-commands.ts` matching the `peaks-<group>-commands.ts` naming convention (`qa-commands.ts`, `code-review-commands.ts`, `audit-commands.ts`). A grep of `src/cli/commands/` confirms NO `final-review-commands.ts` and NO `prepareFinalReview` import in any CLI file. Rationale: the 4-dim business review is conceptually distinct from `peaks qa *` (which is autonomous gate verification) — it is the human-acceptance terminator, not an internal gate. A separate command group preserves that boundary.
99
-
100
- ### Current call path (today, until a CLI wrapper is built)
101
-
102
- ```ts
103
- // peaks-code end-of-workflow, peaks-txt, or any other hand-rolled caller
104
- import { prepareFinalReview } from './src/services/final-review/final-review-service.js';
105
- import { auditGoalLlmRunner } from './src/services/audit/llm-runner.js'; // or your provider
106
-
107
- const out = await prepareFinalReview('<rid>', {
108
- projectRoot: '<absolute path>',
109
- sessionId: '<sessionId>',
110
- llmRunner: auditGoalLlmRunner
111
- });
112
- ```
113
-
114
- ### Planned call path (after the CLI wrapper lands)
92
+ > **Correction (2026-09-12).** An earlier revision of this file claimed `peaks prepare-final-review`
93
+ > "does NOT yet exist" and told callers to hand-roll a `prepareFinalReview()` caller instead.
94
+ > **That was wrong.** The CLI wrapper exists and is registered —
95
+ > `src/cli/commands/final-review-commands.ts` (`W5 Fix M2`), command registered at its line ~130.
96
+ > The hand-rolled snippet that used to sit here also had the wrong flag shape (`--rid <rid>`);
97
+ > **the rid is positional**. Use the CLI.
115
98
 
116
99
  ```bash
117
- # Forward-looking — does not work yet. Track under
118
- # src/cli/commands/final-review-commands.ts
119
- peaks prepare-final-review --rid <rid> [--session-id <sid>] --json
100
+ peaks prepare-final-review <rid> [--session-id <sid>] [--project <path>] [--llm-provider <name>] [--json]
120
101
  ```
121
102
 
122
- `--rid` is required and resolves to `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`. `--json` is required for machine consumption (peaks-code, peaks-txt, downstream CI). The wrapper will resolve `--session-id` from `.peaks/_runtime/current-change` when omitted, matching the workspace binding pattern used by `peaks request *`.
103
+ - `<rid>` is **positional** (not `--rid`). It resolves to `.peaks/_runtime/<sessionId>/audit-goal/<rid>.json`.
104
+ - `--session-id` defaults to the active workspace binding when omitted.
105
+ - **`--llm-provider` defaults to `stub`, and `stub` is the only provider that works today.**
106
+ `stub` runs no real review — it returns a scaffold envelope so CI can prove the route is reachable.
107
+ Verified 2026-09-12: passing a real provider (`--llm-provider anthropic`) returns
108
+ `status: not-applicable` / `serviceWired: false` / `providerBinding: unknown` and tells you to re-run
109
+ with `stub`; the CLI's own nextAction calls real-provider binding "a follow-up slice". So there is
110
+ currently **no reachable path to a real 4-dim review** — confirming the route works is all `stub`
111
+ can do.
112
+ - `--json` is required for machine consumption (peaks-code, peaks-txt, downstream CI).
113
+
114
+ Calling the service directly (`prepareFinalReview(rid, { projectRoot, sessionId, llmRunner })`) remains
115
+ valid for callers that need a custom `LlmRunner` injection seam, but it is no longer the only path.
123
116
 
124
117
  ## Output
125
118
 
@@ -155,6 +148,24 @@ Full evidence contract per dimension: `references/4-dimensions.md`.
155
148
  3. **no-new-bugs** — the regression suite is green AND the LLM surfaces 0 net-new failures (`evidence.kind === 'regression-suite'` + `manual-spot-check`).
156
149
  4. **existing-functionality-intact** — a pre/post baseline diff (test count, public API surface, key behavior) shows no unintended drift (`evidence.kind === 'pre-post-diff'`).
157
150
 
151
+ > **⚠️ This dimension currently cannot pass — read before acting on it (verified 2026-09-12).**
152
+ > `pre-post-diff` is a declared `EvidenceKind`, but **nothing in peaks-loop produces that artifact**.
153
+ > The evidence actually mapped to this dimension is `rd/tech-doc.md` (design intent) and
154
+ > `prd/handoff.md` (scope / non-goals) — neither is a baseline diff, and the model correctly
155
+ > reports that ("the only FOUND source… is a design-intent document, not a regression assessment").
156
+ > `peaks scan api-diff <doc>` is *not* a producer: it diffs an API **document**, not the code surface.
157
+ >
158
+ > **Consequences:** `allPass === true` is **unreachable by construction**, for every workflow.
159
+ > This dimension will return `inconclusive` with an empty `evidence[]` even when the work is
160
+ > perfect. Treat that as a **tooling** state, not as evidence of a regression — and do **not**
161
+ > "fix" it by re-mapping `qa/test-reports` into this dimension's `supports`, which would turn the
162
+ > gate green without producing the baseline diff the definition above requires.
163
+ >
164
+ > **Real fix (unbuilt):** a producer for the pre/post baseline diff (test-count delta, public-API
165
+ > surface snapshot) written to `.peaks/_runtime/<sessionId>/final-review/api-diff.txt`, then mapped
166
+ > into this dimension's `supports`. Until that ships, `needsAttention` always contains this
167
+ > dimension — a permanently-red gate that reviewers will otherwise learn to ignore.
168
+
158
169
  ## Human's role
159
170
 
160
171
  The human reviews evidence, **judges business outcomes (NOT code)**. The LLM produces structured evidence; the human's job is to:
@@ -185,7 +196,7 @@ When handing off, emit: rid, `allPass`, `needsAttention[]`, output path, source
185
196
  | `src/services/final-review/final-review-service.ts` | Authoritative service implementation. |
186
197
  | `src/services/final-review/final-review-types.ts` | `FinalReviewOutput`, `DimensionEvidence`, verdict/evidence/confidence enums. |
187
198
  | `src/services/audit/audit-goal-service.ts:16` | Line of evidence that `LlmRunner` is reusable across audit + final-review (service-level integration). |
188
- | `tests/unit/final-review/final-review-service.test.ts` | Existing service-level unit tests (5 cases). |
199
+ | `tests/unit/final-review/final-review-service.test.ts` | Service-level unit tests (8 cases: evidence inlining, the no-evidence⇒no-`pass` gate, prompt bounds, plus contract guards). |
189
200
  | `docs/superpowers/plans/2026-06-25-slice-topology-multipass-phase-4.md:127` | Phase-4 plan prose (Task 14). |
190
201
  | `skills/peaks-qa/SKILL.md` | Upstream QA skill — 4-dim review is downstream of all QA gates. |
191
202
  | `skills/peaks-audit/SKILL.md` | Sibling skill — produces the `audit-goal` JSON that this skill consumes. |