peaks-loop 4.0.44 → 4.0.46

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. package/CHANGELOG.md +59 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/dist/cli/commands/code-runtime-commands.js +30 -19
  5. package/dist/cli/commands/codegraph-commands.d.ts +1 -0
  6. package/dist/cli/commands/codegraph-commands.js +49 -3
  7. package/dist/cli/commands/final-review-commands.js +3 -1
  8. package/dist/services/code/auto-compact-lifecycle.d.ts +39 -3
  9. package/dist/services/code/auto-compact-lifecycle.js +41 -6
  10. package/dist/services/code/auto-compact-orchestrator.js +20 -11
  11. package/dist/services/compact-statusline/compact-lifecycle-store.d.ts +11 -2
  12. package/dist/services/compact-statusline/compact-lifecycle-store.js +19 -1
  13. package/dist/services/compact-statusline/compact-statusline-service.d.ts +1 -1
  14. package/dist/services/compact-statusline/compact-statusline-service.js +22 -0
  15. package/dist/services/final-review/final-review-service.d.ts +224 -44
  16. package/dist/services/final-review/final-review-service.js +934 -97
  17. package/dist/services/final-review/index.d.ts +2 -1
  18. package/dist/services/final-review/index.js +2 -1
  19. package/dist/services/final-review/pre-post-diff.d.ts +137 -0
  20. package/dist/services/final-review/pre-post-diff.js +657 -0
  21. package/dist/services/skills/skill-statusline-renderer.js +30 -5
  22. package/dist/services/skills/statusline-palette.d.ts +6 -0
  23. package/dist/services/skills/statusline-palette.js +4 -1
  24. package/package.json +5 -5
  25. package/skills/peaks-code/SKILL.md +2 -2
  26. package/skills/peaks-final-review/SKILL.md +51 -18
  27. package/skills/peaks-final-review/references/4-dimensions.md +42 -5
@@ -9,12 +9,18 @@ export interface LlmRunner {
9
9
  };
10
10
  }>;
11
11
  }
12
- import type { FinalReviewOutput } from './final-review-types.js';
12
+ import type { DimensionKind, FinalReviewOutput } from './final-review-types.js';
13
13
  import type { CapabilityAuditResult } from '../capability-audit-service/types.js';
14
14
  export interface PrepareFinalReviewOptions {
15
15
  readonly projectRoot: string;
16
16
  readonly sessionId: string;
17
17
  readonly llmRunner: LlmRunner;
18
+ /**
19
+ * Explicit base ref for the pre/post baseline diff. Unset ⇒ the producer
20
+ * resolves the merge-base with the upstream default branch itself, and
21
+ * reports the dimension `unavailable` when it cannot.
22
+ */
23
+ readonly baseRef?: string;
18
24
  }
19
25
  export declare class IncompleteFinalReviewError extends Error {
20
26
  readonly code: "INCOMPLETE_FINAL_REVIEW";
@@ -43,52 +49,91 @@ export declare class EmptyReviewReplyError extends Error {
43
49
  */
44
50
  export declare const MAX_EMPTY_REPLY_ATTEMPTS = 3;
45
51
  /**
46
- * Per-file evidence cap. Real evidence artifacts in this repo run 11–18 KB
47
- * (`rd/tech-doc.md`, `rd/code-review.md`, `qa/*-findings-*.md`); 8 KB keeps the
48
- * head of every file (header + verdict + first tables) without letting one
49
- * verbose artifact crowd out the other sources. Enforced in BYTES against the
50
- * raw buffer, so multi-byte (CJK) content cannot slip past the cap.
51
- */
52
- export declare const MAX_EVIDENCE_BYTES_PER_FILE: number;
53
- /**
54
- * Total evidence budget across all sources. 32 KB ≈ 8k tokens of input, which
52
+ * Total evidence budget across all sources. 40 KB ≈ 10k tokens of input, which
55
53
  * keeps the prompt far inside any modern context window. Sources that do not
56
54
  * fit are reported as OMITTED — never dropped silently.
57
55
  *
58
- * This is an INPUT cap and stays fixed. The output ceiling that has to sit
59
- * opposite it is derived per call by `outputBudgetForEvidence()` below — the
60
- * two used to drift apart, and that drift was the defect.
56
+ * This is an INPUT cap and stays fixed per run. The output ceiling that has to
57
+ * sit opposite it is derived per call by `outputBudgetForEvidence()` below —
58
+ * the two used to drift apart, and that drift was the defect.
59
+ *
60
+ * RE-EVALUATED 2026-09-12 (F-BLOCK) — 32 KiB did NOT hold once the tenth
61
+ * source (`final-review-pre-post-diff`) was appended. Measured on this repo's
62
+ * own run (`2026-09-12-session-e37ef0`, rid
63
+ * `2026-09-12-codegraph-exclude-integrity`), the ten sources are 2,621 / 8,164
64
+ * / 9,492 / 10,839 / 11,422 / 13,050 / 13,462 / 13,852 / 17,745 / 20,543
65
+ * bytes — 121,190 bytes on disk, 76,321 bytes once the 8 KiB per-file cap is
66
+ * applied. Four sources at the cap spent the old 32,768 to the byte, so the
67
+ * appended tenth (added LAST, by design) was the first block the budget
68
+ * dropped — on every run, and on the compact set QA measured too (9 x 3.8 KB
69
+ * ≈ 34 KB, which the old cap already could not hold).
70
+ *
71
+ * 40 KiB is a budget that fits the measured ten-source set at the allocator's
72
+ * per-source unit (5 x 8 KiB units + the smaller ones), and it is the largest
73
+ * this budget may grow to today: the derived output ceiling is
74
+ * `3000 + 12288 + bytes/4`, so it reaches MAX_OUTPUT_TOKENS (32,000) at exactly
75
+ * 66,848 bytes of inlined evidence and is clamped from there on — a cap at or
76
+ * above that point buys the reviewer no more output room at all.
77
+ *
78
+ * F-CAP — an earlier version of this comment claimed 40 KiB (40,960 =
79
+ * 10 x 4,096) was "the smallest cap that affords one floor-sized slice to EVERY
80
+ * source in the ten-source set". That was arithmetic about a budget, not a
81
+ * statement about the allocator: sources are capped at
82
+ * MAX_EVIDENCE_BYTES_PER_FILE (10 KiB each, derived — see below) while a floor
83
+ * is per DIMENSION (four of them), so 10 x 4,096 was never what any source
84
+ * received. Re-measured on
85
+ * the same two saturated fixtures (QA round 4: the 9 x 40 KB fixture and this
86
+ * repo's own file sizes), raising the cap from 32 KiB to 40 KiB left **4 of the
87
+ * 10 sources inlined with zero bytes** — `rd/security-review`,
88
+ * `rd/bug-analysis`, `prd/handoff` and the appended
89
+ * `final-review-pre-post-diff` — and it did not deliver the baseline either;
90
+ * all it did was change WHICH source was starved (at 32 KiB that source was
91
+ * `rd/code-review`).
92
+ *
93
+ * So read this constant as "how much evidence fits", never as "which sources
94
+ * arrive". Arrival is decided by the allocator below (one whole unit per source,
95
+ * with a dimension's floor reserving its holder's unit) and, for the two sources
96
+ * a dimension's verdict may not outlive, by the delivery gates on top of it.
97
+ * And note the honest consequence, stated rather than papered over: on a
98
+ * saturated run the allocator still omits sources, so `prd/handoff.md` and the
99
+ * pre/post diff can be dropped — which is exactly why
100
+ * `enforceScopeContractDelivery()` and `enforcePrePostDiffAvailability()` exist
101
+ * and why a `pass` on those two dimensions is keyed on delivery, not on
102
+ * existence.
61
103
  */
62
104
  export declare const MAX_EVIDENCE_BYTES_TOTAL: number;
63
105
  /**
64
- * Reserved floor per dimension — the anti-starvation guarantee.
65
- *
66
- * The allocator below used to be strictly first-come-first-served: each source
67
- * took `min(perFileCap, budgetLeft)` in source order. With this repo's own
68
- * 9-source evidence set (`2026-09-12-session-e37ef0`, measured) sources 1-4
69
- * consumed the whole 32 KiB — 4 x 8,192 = 32,768, the cap to the byte — before
70
- * source 5 was even opened. `existing-functionality-intact` is supplied ONLY by
71
- * `rd/tech-doc.md` (6th) and `prd/handoff.md` (9th), so that one dimension
72
- * reached the reviewer with zero evidence on every run and its verdict was
73
- * structurally locked to `inconclusive` no matter how good the work was. A gate
74
- * that is always red is noise, and an operator trained to ignore noise has no
75
- * gate at all — the same harm as a gate that never fires, only quieter.
76
- *
77
- * So each dimension with at least one readable source on disk gets one floor
78
- * reserved for the FIRST such source, and no source that is not that holder may
79
- * spend it. The reservation is a floor, never a quota: it is released the
80
- * instant its holder is served, and whatever the holder does not use flows back
81
- * into the sequential allocation unchanged.
82
- *
83
- * Why 4 KiB: the per-file cap exists to keep the "header + verdict + first
84
- * tables" — the part a reviewer actually cites. Measured on the same run's
85
- * artifacts (9 files, 8,164-20,543 bytes each): every one of them states its
86
- * verdict inside the first ~700 bytes. 4 KiB is ~5x that, so a floor holder is
87
- * not there for depth — it is there so its dimension is not blind. Four
88
- * dimensions x 4 KiB = 16 KiB of the 32 KiB cap, so at least half the budget
89
- * still flows through the sequential path below.
90
- */
91
- export declare const MIN_EVIDENCE_BYTES_PER_DIMENSION: number;
106
+ * Per-file evidence cap — and the allocator's UNIT. DERIVED, never typed.
107
+ *
108
+ * F5 — this used to be the literal `8 * 1024`, and the whole floor guarantee
109
+ * below was silently conditioned on `4 x perFile <= total`: the reservation
110
+ * promises every pending holder its own unit, and that promise holds only while
111
+ * the four units fit the budget together. At `8 * 1024` the inequality held by
112
+ * coincidence (4 x 8,192 = 32,768 <= 40,960) and nothing in the code said so —
113
+ * raising the cap past 10,240 would have starved all four dimensions in the
114
+ * same run, which reads as four independent red gates rather than one broken
115
+ * constant. The cap is therefore DIVIDED OUT OF the total: the relation is
116
+ * definitional instead of remembered, and `assertFloorReservationAffordable()`
117
+ * still checks it at the point of use (the division must also be exact).
118
+ *
119
+ * The value is 10,240 because that is `40,960 / 4` — the largest unit the
120
+ * four-dimension reservation can afford. It is also the smallest cap at which
121
+ * the decisive evidence source of this repo's own run can be delivered WHOLE:
122
+ * `qa/test-reports/<rid>.md` measured 9,492 bytes there, and `whole` is the
123
+ * delivery rule for every source whose producer publishes no conclusion literal
124
+ * (see `isDelivered`), so a cap under 9,492 makes `problem-resolution` and
125
+ * `no-new-bugs` undeliverable on every run — the always-red gate this module
126
+ * refuses to ship. Enforced in BYTES against the raw buffer, so multi-byte
127
+ * (CJK) content cannot slip past the cap.
128
+ *
129
+ * A source is inlined as `min(bytes, this cap)` — its whole unit — or not at
130
+ * all. So a `TRUNCATED` source block can only ever mean "the FILE is bigger
131
+ * than this cap"; it can no longer mean "the budget ran out while this file was
132
+ * being copied in". That distinction is the point of the all-or-nothing
133
+ * allocator below: the second reading is what let one byte of a 4,226-byte
134
+ * artifact be counted as delivered evidence (F-BLOCK-1BYTE).
135
+ */
136
+ export declare const MAX_EVIDENCE_BYTES_PER_FILE: number;
92
137
  /**
93
138
  * Floor — also the value that shipped before this fix, so no evidence set can
94
139
  * end up with a smaller budget than it had. ~3000 tokens is enough for the
@@ -110,9 +155,12 @@ export declare const MIN_OUTPUT_TOKENS = 3000;
110
155
  * truncated on 2 of 10 real runs.
111
156
  *
112
157
  * 12288 (12 KiB) is sized so the largest evidence pack the input caps allow
113
- * lands at 23480 (see the formula below) — about 1.8x the largest value ever
114
- * OBSERVED to truncate (13240), which is the margin the observed variance
115
- * asks for. The numbers are in the block comment above.
158
+ * lands at 25528 (see the formula below; the input cap is 40 KiB as of the
159
+ * F-BLOCK re-evaluation, and this floor was checked against the raised cap, not
160
+ * against the 32 KiB the measurements above were taken at — a 12 KiB headroom
161
+ * under a bigger cap is the conservative direction) — about 1.9x the largest
162
+ * value ever OBSERVED to truncate (13240), which is the margin the observed
163
+ * variance asks for. The numbers are in the block comment above.
116
164
  */
117
165
  export declare const REASONING_HEADROOM_TOKENS: number;
118
166
  /**
@@ -174,6 +222,137 @@ export declare function outputBudgetForEvidence(includedEvidenceBytes: number):
174
222
  * still gets a call made.
175
223
  */
176
224
  export declare function resolveOutputBudget(includedEvidenceBytes: number, env?: NodeJS.ProcessEnv): number;
225
+ /**
226
+ * What DELIVERED means for one source — the module's ONE delivery definition,
227
+ * declared per source and read in exactly one place (`isDelivered`).
228
+ *
229
+ * whole the reviewer must have received the whole document. A
230
+ * truncated slice is not a weaker version of a document, it is a
231
+ * DIFFERENT document, and the conclusion may be in the part that
232
+ * was cut — which is precisely the state `found` used to call
233
+ * "delivered".
234
+ * conclusion the reviewer must have received the literal that IS the
235
+ * document's conclusion. Used where the producer publishes one
236
+ * (the pre/post diff opens with its `VERDICT:` line).
237
+ *
238
+ * `whole` is the rule wherever the producer is another role's skill and
239
+ * publishes no conclusion literal: the module may not GUESS where a document's
240
+ * conclusion lives. The two are the same judgement — "did the reviewer receive
241
+ * the conclusion" — checked at the only place the module can check it.
242
+ */
243
+ type DeliveryRule = {
244
+ readonly kind: 'whole';
245
+ } | {
246
+ readonly kind: 'conclusion';
247
+ readonly marker: string;
248
+ };
249
+ interface EvidenceSource {
250
+ /** Stable id quoted by the model in its citations. */
251
+ readonly key: string;
252
+ readonly label: string;
253
+ /** Path segments under `.peaks/_runtime/<sessionId>/`. */
254
+ readonly segments: readonly string[];
255
+ /** Dimensions this source can supply evidence for. */
256
+ readonly supports: readonly DimensionKind[];
257
+ /** What DELIVERED means for this source. See `isDelivered()` — every source
258
+ * must declare one, and the declaration is the only thing the module's
259
+ * delivery judgement reads. */
260
+ readonly delivery: DeliveryRule;
261
+ }
262
+ /**
263
+ * `found` is the only status that carries bytes. The other four exist so the
264
+ * prompt can name *why* a source carries nothing: an absent file, one that
265
+ * exists but is empty, one that exists but could not be READ, and one that did
266
+ * not fit the byte budget are four different facts, and the model is told all
267
+ * four explicitly.
268
+ *
269
+ * F4 — `missing` and `unreadable` used to be one status. `readFileSync`'s catch
270
+ * collapsed EACCES / EBUSY / EPERM into the same `raw === null` as ENOENT, so a
271
+ * contract file that EXISTS and could not be opened was reported to the
272
+ * reviewer — and, worse, to the delivery gate — as "there was no PRD phase".
273
+ * Those are opposite facts about a run: one says "nothing to deliver", the
274
+ * other says "there is something to deliver and it did not arrive".
275
+ */
276
+ type EvidenceStatus = 'found' | 'empty' | 'missing' | 'unreadable' | 'omitted';
277
+ interface CollectedEvidence {
278
+ readonly source: EvidenceSource;
279
+ readonly relativePath: string;
280
+ readonly absolutePath: string;
281
+ readonly status: EvidenceStatus;
282
+ /** Full size on disk (0 when nothing could be read). */
283
+ readonly totalBytes: number;
284
+ /** Bytes actually inlined into the prompt. */
285
+ readonly includedBytes: number;
286
+ readonly content: string;
287
+ /** Why this source carries no evidence (non-`found` statuses only). */
288
+ readonly reason: string;
289
+ }
290
+ /**
291
+ * F5 — the floor promise, checked where it is relied on instead of remembered.
292
+ *
293
+ * The allocator's loop invariant is `budgetLeft >= sum(units of pending
294
+ * holders)`, and its INITIAL condition is
295
+ * `REQUIRED_DIMENSIONS.length x MAX_EVIDENCE_BYTES_PER_FILE <=
296
+ * MAX_EVIDENCE_BYTES_TOTAL`. While that holds, every holder is served when it
297
+ * is reached and the reservation is a guarantee; the moment it fails, all four
298
+ * dimensions are starved in the same run — which surfaces as four independent
299
+ * red gates rather than as one broken constant, and is therefore the kind of
300
+ * breakage nobody diagnoses correctly.
301
+ *
302
+ * The cap is derived from the total so the inequality cannot be typed wrong,
303
+ * and this check covers the two ways a derivation can still go bad: a
304
+ * non-integer quotient (a fifth dimension, say) and a future re-typing of
305
+ * either constant. It is exported because a test asserts it directly.
306
+ */
307
+ export declare function assertFloorReservationAffordable(): void;
308
+ /**
309
+ * H2 — the reason a dimension is red PERMANENTLY, as opposed to merely unfed on
310
+ * this run. Reported per source, so the arithmetic is checkable by the reader.
311
+ */
312
+ export interface UndeliverableDimensionEvidence {
313
+ readonly dimension: DimensionKind;
314
+ readonly sources: readonly {
315
+ readonly key: string;
316
+ readonly relativePath: string;
317
+ readonly totalBytes: number;
318
+ }[];
319
+ }
320
+ /**
321
+ * H2 — every required dimension whose verdict is locked to `inconclusive` by
322
+ * BYTE ARITHMETIC rather than by the reviewer's judgement.
323
+ *
324
+ * The defect this exists for: `qa/test-reports/<rid>.md` measured 9,492 bytes
325
+ * on this repo's own run and is the ONLY source on disk for `problem-resolution`
326
+ * and `no-new-bugs` (the other sources that support them are absent), against a
327
+ * per-file cap of 10,240 — a 748-byte margin on a file that is REWRITTEN every
328
+ * round and only grows. The moment it crosses 10,240 both dimensions go
329
+ * permanently red, and NOTHING said so: `assertFloorReservationAffordable()`
330
+ * only checks the constant-level relation (`4 x cap <= total`), never whether
331
+ * any actual source fits the cap it must live under, so the red handoff read as
332
+ * "the reviewer was unsure" instead of "no evidence can ever reach the
333
+ * reviewer". A gate that is always red and never explains itself is noise, and
334
+ * this is the same "always red" harm the floor reservation was built to remove
335
+ * — one layer down.
336
+ *
337
+ * The conditions, all three of which must hold, are chosen so the report cannot
338
+ * be noise:
339
+ * 1. NOTHING on disk can back the dimension (no `isDelivered`), and
340
+ * 2. there IS something on disk to deliver (a source that was never written
341
+ * is not a delivery failure — same reasoning as the scope-contract gate's
342
+ * `missing` exemption: a run with no QA phase has no report to lose), and
343
+ * 3. EVERY one of those on-disk sources is structurally undeliverable — a
344
+ * single source that merely did not fit TODAY (budget-exhausted `omitted`)
345
+ * is a different, self-correcting state and is left to the allocator.
346
+ *
347
+ * The report is consumed by the prompt (stated to the reviewer), by the
348
+ * envelope (a marker on each dimension's summary) and by `needsAttention`
349
+ * (which also clears `allPass`) — see `enforceDeliveryReachability` and
350
+ * `renderDeliveryReachabilityStatus`. It is a LOUD STATEMENT, not a silent
351
+ * downgrade, and it is deliberately not a throw: a crash would destroy the
352
+ * evidence for the three dimensions that ARE deliverable, and the honest fact
353
+ * here is per-dimension, so it is reported per-dimension.
354
+ */
355
+ export declare function undeliverableDimensions(collected: readonly CollectedEvidence[]): readonly UndeliverableDimensionEvidence[];
177
356
  export declare function prepareFinalReview(rid: string, opts: PrepareFinalReviewOptions): Promise<FinalReviewOutput>;
178
357
  export declare function decideFifthDimension(input: {
179
358
  readonly audit: CapabilityAuditResult | null;
@@ -182,3 +361,4 @@ export declare function decideFifthDimension(input: {
182
361
  readonly verdict: 'pass' | 'fail' | 'inconclusive';
183
362
  readonly reason: string;
184
363
  };
364
+ export {};