peaks-loop 4.0.43 → 4.0.44

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/dist/cli/commands/codegraph-commands.js +191 -6
  5. package/dist/cli/commands/final-review-commands.d.ts +34 -10
  6. package/dist/cli/commands/final-review-commands.js +130 -34
  7. package/dist/cli/commands/share-commands.d.ts +49 -0
  8. package/dist/cli/commands/share-commands.js +114 -14
  9. package/dist/services/codegraph/codegraph-autorefresh.js +12 -0
  10. package/dist/services/codegraph/codegraph-exclude-integrity.d.ts +61 -0
  11. package/dist/services/codegraph/codegraph-exclude-integrity.js +98 -0
  12. package/dist/services/codegraph/codegraph-exclude-reconciler.d.ts +26 -0
  13. package/dist/services/codegraph/codegraph-exclude-reconciler.js +217 -0
  14. package/dist/services/codegraph/codegraph-exclude-repair.d.ts +102 -0
  15. package/dist/services/codegraph/codegraph-exclude-repair.js +266 -0
  16. package/dist/services/codegraph/codegraph-preflight-service.js +12 -0
  17. package/dist/services/codegraph/codegraph-service.d.ts +0 -1
  18. package/dist/services/codegraph/codegraph-service.js +5 -4
  19. package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.d.ts +29 -0
  20. package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.js +88 -0
  21. package/dist/services/doctor/doctor-service/plugin-registry.js +2 -0
  22. package/dist/services/doctor/doctor-service/types.d.ts +27 -0
  23. package/dist/services/final-review/final-review-service.d.ts +154 -0
  24. package/dist/services/final-review/final-review-service.js +621 -7
  25. package/dist/services/final-review/index.d.ts +1 -1
  26. package/dist/services/final-review/index.js +1 -1
  27. package/dist/services/prd/handoff-auto-regen.js +0 -1
  28. package/dist/services/prd/handoff-service.d.ts +9 -1
  29. package/dist/services/prd/handoff-service.js +48 -6
  30. package/package.json +7 -5
  31. package/skills/peaks-final-review/SKILL.md +43 -32
@@ -20,6 +20,160 @@ export declare class IncompleteFinalReviewError extends Error {
20
20
  readonly code: "INCOMPLETE_FINAL_REVIEW";
21
21
  constructor(message: string);
22
22
  }
23
+ /**
24
+ * N4 — the reply carried no text block at all.
25
+ *
26
+ * Measured 2/3 on this repo's own machine, and it is NOT truncation: the
27
+ * provider answered with a response whose `content` has no `text` block (a
28
+ * reasoning-only turn, a refusal, or a content filter), so there is no JSON to
29
+ * parse and no budget to raise — an operator sent to "raise the budget" for
30
+ * this failure would be sent the wrong way. It gets its own class, its own
31
+ * `code`, and a message that says so, so it is diagnosable instead of being
32
+ * flattened into "not valid JSON".
33
+ */
34
+ export declare class EmptyReviewReplyError extends Error {
35
+ readonly code: "EMPTY_FINAL_REVIEW_REPLY";
36
+ constructor(message: string);
37
+ }
38
+ /**
39
+ * How many times an empty reply is retried before it is reported. The failure
40
+ * was 2/3 on the observed machine — intermittent, not systematic — so a small
41
+ * bounded retry converts most of it into a completed review, while 3 attempts
42
+ * keeps a genuinely broken provider from being hammered.
43
+ */
44
+ export declare const MAX_EMPTY_REPLY_ATTEMPTS = 3;
45
+ /**
46
+ * Per-file evidence cap. Real evidence artifacts in this repo run 11–18 KB
47
+ * (`rd/tech-doc.md`, `rd/code-review.md`, `qa/*-findings-*.md`); 8 KB keeps the
48
+ * head of every file (header + verdict + first tables) without letting one
49
+ * verbose artifact crowd out the other sources. Enforced in BYTES against the
50
+ * raw buffer, so multi-byte (CJK) content cannot slip past the cap.
51
+ */
52
+ export declare const MAX_EVIDENCE_BYTES_PER_FILE: number;
53
+ /**
54
+ * Total evidence budget across all sources. 32 KB ≈ 8k tokens of input, which
55
+ * keeps the prompt far inside any modern context window. Sources that do not
56
+ * fit are reported as OMITTED — never dropped silently.
57
+ *
58
+ * This is an INPUT cap and stays fixed. The output ceiling that has to sit
59
+ * opposite it is derived per call by `outputBudgetForEvidence()` below — the
60
+ * two used to drift apart, and that drift was the defect.
61
+ */
62
+ export declare const MAX_EVIDENCE_BYTES_TOTAL: number;
63
+ /**
64
+ * Reserved floor per dimension — the anti-starvation guarantee.
65
+ *
66
+ * The allocator below used to be strictly first-come-first-served: each source
67
+ * took `min(perFileCap, budgetLeft)` in source order. With this repo's own
68
+ * 9-source evidence set (`2026-09-12-session-e37ef0`, measured) sources 1-4
69
+ * consumed the whole 32 KiB — 4 x 8,192 = 32,768, the cap to the byte — before
70
+ * source 5 was even opened. `existing-functionality-intact` is supplied ONLY by
71
+ * `rd/tech-doc.md` (6th) and `prd/handoff.md` (9th), so that one dimension
72
+ * reached the reviewer with zero evidence on every run and its verdict was
73
+ * structurally locked to `inconclusive` no matter how good the work was. A gate
74
+ * that is always red is noise, and an operator trained to ignore noise has no
75
+ * gate at all — the same harm as a gate that never fires, only quieter.
76
+ *
77
+ * So each dimension with at least one readable source on disk gets one floor
78
+ * reserved for the FIRST such source, and no source that is not that holder may
79
+ * spend it. The reservation is a floor, never a quota: it is released the
80
+ * instant its holder is served, and whatever the holder does not use flows back
81
+ * into the sequential allocation unchanged.
82
+ *
83
+ * Why 4 KiB: the per-file cap exists to keep the "header + verdict + first
84
+ * tables" — the part a reviewer actually cites. Measured on the same run's
85
+ * artifacts (9 files, 8,164-20,543 bytes each): every one of them states its
86
+ * verdict inside the first ~700 bytes. 4 KiB is ~5x that, so a floor holder is
87
+ * not there for depth — it is there so its dimension is not blind. Four
88
+ * dimensions x 4 KiB = 16 KiB of the 32 KiB cap, so at least half the budget
89
+ * still flows through the sequential path below.
90
+ */
91
+ export declare const MIN_EVIDENCE_BYTES_PER_DIMENSION: number;
92
+ /**
93
+ * Floor — also the value that shipped before this fix, so no evidence set can
94
+ * end up with a smaller budget than it had. ~3000 tokens is enough for the
95
+ * envelope skeleton plus a short paragraph per dimension.
96
+ */
97
+ export declare const MIN_OUTPUT_TOKENS = 3000;
98
+ /**
99
+ * Headroom for the part of the reply that is not the envelope.
100
+ *
101
+ * A Messages-API-compatible endpoint applies `max_tokens` to the WHOLE
102
+ * response, and a reasoning model spends it on hidden reasoning before it
103
+ * emits a single character of the 4-dim envelope. Measured on this repo's own
104
+ * machine (2026-09-12, rid `2026-09-12-codegraph-exclude-integrity`,
105
+ * `deepseek-flash[1M]` via `api.deepseek.com/anthropic`): `max_tokens=8192`
106
+ * came back with `output_tokens=8192` and only **574** visible characters —
107
+ * the entire budget went to reasoning. A bytes-per-token estimate of the
108
+ * visible output cannot see that cost, which is why the previous formula
109
+ * budgeted 7096 for a reply that needs 10108 — and why a 13240 budget still
110
+ * truncated on 2 of 10 real runs.
111
+ *
112
+ * 12288 (12 KiB) is sized so the largest evidence pack the input caps allow
113
+ * lands at 23480 (see the formula below) — about 1.8x the largest value ever
114
+ * OBSERVED to truncate (13240), which is the margin the observed variance
115
+ * asks for. The numbers are in the block comment above.
116
+ */
117
+ export declare const REASONING_HEADROOM_TOKENS: number;
118
+ /**
119
+ * Ceiling, 32_000: the value a real run on this machine was forced to in order
120
+ * to complete the envelope at all, and the largest this endpoint was observed
121
+ * to accept. 16384 was tried first and truncated 2/10 — a ceiling that is
122
+ * merely "above the last successful measurement" is not above the requirement,
123
+ * because the requirement moves with the model's reasoning spend.
124
+ *
125
+ * A model that caps output at 8192 will refuse this. That is still strictly
126
+ * better than shipping a budget measured to be too small, and the env lever
127
+ * below lets an operator pull it down without a code change.
128
+ */
129
+ export declare const MAX_OUTPUT_TOKENS = 32000;
130
+ /**
131
+ * Environment lever. The old failure message told the operator to "raise the
132
+ * budget" while the budget was a module constant with no CLI flag and no env
133
+ * var — an instruction that could not be carried out from any surface the
134
+ * operator has. This is that lever.
135
+ *
136
+ * The value is the output ceiling in tokens; it OVERRIDES the derivation below
137
+ * (it is not a bonus added to it). Unset/invalid/out-of-range handling is in
138
+ * `resolveOutputBudget`.
139
+ */
140
+ export declare const MAX_OUTPUT_TOKENS_ENV = "PEAKS_FINAL_REVIEW_MAX_OUTPUT_TOKENS";
141
+ /**
142
+ * Absolute upper bound the env lever may reach. An endpoint that accepts
143
+ * `max_tokens` at all accepts this; anything above it is a typo (a stray extra
144
+ * digit), not an intent, and clamping is safer than sending it.
145
+ */
146
+ export declare const HARD_MAX_OUTPUT_TOKENS = 64000;
147
+ /**
148
+ * Inlined bytes that buy one extra output token — 4:1.
149
+ *
150
+ * This term prices the visible envelope (4 x `summary` + `evidence[]` +
151
+ * `confidence` + `overallSummary`) against the evidence the model is required
152
+ * to cite. It was 8:1, which put the 32 KiB pack at 4096 tokens of visible
153
+ * output; the same pack has been observed to complete at 10108 and to truncate
154
+ * at 13240, so 8:1 was pricing the visible side BELOW its own measurement.
155
+ * 4:1 doubles it to 8192. It is still not treated as the whole budget — see
156
+ * `REASONING_HEADROOM_TOKENS`.
157
+ */
158
+ export declare const EVIDENCE_BYTES_PER_OUTPUT_TOKEN = 4;
159
+ /**
160
+ * Output ceiling for a call whose prompt carries `includedEvidenceBytes` bytes
161
+ * of inlined evidence. Pure, total, and clamped on both ends — the same
162
+ * evidence pack always yields the same budget.
163
+ */
164
+ export declare function outputBudgetForEvidence(includedEvidenceBytes: number): number;
165
+ /**
166
+ * The budget the call actually uses: the derived one, unless
167
+ * `PEAKS_FINAL_REVIEW_MAX_OUTPUT_TOKENS` overrides it.
168
+ *
169
+ * An override that is not a positive integer THROWS rather than being ignored:
170
+ * a silent fallback would leave an operator who passed a bad value with the
171
+ * exact experience this lever exists to remove — a budget they cannot move.
172
+ * Out-of-range values are clamped, not rejected, so a model needing more than
173
+ * `HARD_MAX_OUTPUT_TOKENS` (or a model needing less than `MIN_OUTPUT_TOKENS`)
174
+ * still gets a call made.
175
+ */
176
+ export declare function resolveOutputBudget(includedEvidenceBytes: number, env?: NodeJS.ProcessEnv): number;
23
177
  export declare function prepareFinalReview(rid: string, opts: PrepareFinalReviewOptions): Promise<FinalReviewOutput>;
24
178
  export declare function decideFifthDimension(input: {
25
179
  readonly audit: CapabilityAuditResult | null;