peaks-loop 4.0.43 → 4.0.45
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +73 -0
- package/README-en.md +1 -1
- package/README.md +1 -1
- package/dist/cli/commands/codegraph-commands.d.ts +1 -0
- package/dist/cli/commands/codegraph-commands.js +239 -8
- package/dist/cli/commands/final-review-commands.d.ts +34 -10
- package/dist/cli/commands/final-review-commands.js +132 -34
- package/dist/cli/commands/share-commands.d.ts +49 -0
- package/dist/cli/commands/share-commands.js +114 -14
- package/dist/services/codegraph/codegraph-autorefresh.js +12 -0
- package/dist/services/codegraph/codegraph-exclude-integrity.d.ts +61 -0
- package/dist/services/codegraph/codegraph-exclude-integrity.js +98 -0
- package/dist/services/codegraph/codegraph-exclude-reconciler.d.ts +26 -0
- package/dist/services/codegraph/codegraph-exclude-reconciler.js +217 -0
- package/dist/services/codegraph/codegraph-exclude-repair.d.ts +102 -0
- package/dist/services/codegraph/codegraph-exclude-repair.js +266 -0
- package/dist/services/codegraph/codegraph-preflight-service.js +12 -0
- package/dist/services/codegraph/codegraph-service.d.ts +0 -1
- package/dist/services/codegraph/codegraph-service.js +5 -4
- package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.d.ts +29 -0
- package/dist/services/doctor/doctor-service/checks/codegraph-exclude-integrity.js +88 -0
- package/dist/services/doctor/doctor-service/plugin-registry.js +2 -0
- package/dist/services/doctor/doctor-service/types.d.ts +27 -0
- package/dist/services/final-review/final-review-service.d.ts +335 -1
- package/dist/services/final-review/final-review-service.js +1457 -6
- package/dist/services/final-review/index.d.ts +2 -1
- package/dist/services/final-review/index.js +2 -1
- package/dist/services/final-review/pre-post-diff.d.ts +137 -0
- package/dist/services/final-review/pre-post-diff.js +657 -0
- package/dist/services/prd/handoff-auto-regen.js +0 -1
- package/dist/services/prd/handoff-service.d.ts +9 -1
- package/dist/services/prd/handoff-service.js +48 -6
- package/package.json +7 -5
- package/skills/peaks-final-review/SKILL.md +79 -35
- package/skills/peaks-final-review/references/4-dimensions.md +42 -5
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Check: codegraph exclude integrity (`capability:codegraph-exclude-integrity`).
|
|
3
|
+
*
|
|
4
|
+
* The companion to `capability:codegraph`. That check answers "is the
|
|
5
|
+
* upstream package resolvable at the pinned version"; this one answers
|
|
6
|
+
* "does the project's `.codegraph/config.json` exclude rules silently
|
|
7
|
+
* drop source files git tracks".
|
|
8
|
+
*
|
|
9
|
+
* Why it exists: upstream ships a default `exclude` template matched by
|
|
10
|
+
* *directory name*, and `config.json`'s `exclude` array replaces those
|
|
11
|
+
* defaults wholesale. Any default rule whose directory name collides
|
|
12
|
+
* with a real source directory drops tracked files from the index while
|
|
13
|
+
* `peaks codegraph status` still printed `[OK] Index is up to date`.
|
|
14
|
+
*
|
|
15
|
+
* Read-only by construction — it consumes the same inspector
|
|
16
|
+
* `peaks codegraph status` gates on and never writes the config.
|
|
17
|
+
*
|
|
18
|
+
* Failure posture:
|
|
19
|
+
* - codegraph not initialized in the inspected root (no
|
|
20
|
+
* `.codegraph/config.json`) → `ok: true`; there is nothing to
|
|
21
|
+
* reconcile and a fresh clone must not fail the doctor.
|
|
22
|
+
* - a confirmed gap → `ok: false` (blocking; the index is provably
|
|
23
|
+
* incomplete and the fix is one command).
|
|
24
|
+
* - could not evaluate (not a git work tree, malformed config) →
|
|
25
|
+
* `ok: false, severity: 'warning'` so the doctor reports the blind
|
|
26
|
+
* spot without flipping the exit code on an unrelated failure.
|
|
27
|
+
*/
|
|
28
|
+
import type { DoctorCheckPlugin } from '../types.js';
|
|
29
|
+
export declare const check: DoctorCheckPlugin;
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Check: codegraph exclude integrity (`capability:codegraph-exclude-integrity`).
|
|
3
|
+
*
|
|
4
|
+
* The companion to `capability:codegraph`. That check answers "is the
|
|
5
|
+
* upstream package resolvable at the pinned version"; this one answers
|
|
6
|
+
* "does the project's `.codegraph/config.json` exclude rules silently
|
|
7
|
+
* drop source files git tracks".
|
|
8
|
+
*
|
|
9
|
+
* Why it exists: upstream ships a default `exclude` template matched by
|
|
10
|
+
* *directory name*, and `config.json`'s `exclude` array replaces those
|
|
11
|
+
* defaults wholesale. Any default rule whose directory name collides
|
|
12
|
+
* with a real source directory drops tracked files from the index while
|
|
13
|
+
* `peaks codegraph status` still printed `[OK] Index is up to date`.
|
|
14
|
+
*
|
|
15
|
+
* Read-only by construction — it consumes the same inspector
|
|
16
|
+
* `peaks codegraph status` gates on and never writes the config.
|
|
17
|
+
*
|
|
18
|
+
* Failure posture:
|
|
19
|
+
* - codegraph not initialized in the inspected root (no
|
|
20
|
+
* `.codegraph/config.json`) → `ok: true`; there is nothing to
|
|
21
|
+
* reconcile and a fresh clone must not fail the doctor.
|
|
22
|
+
* - a confirmed gap → `ok: false` (blocking; the index is provably
|
|
23
|
+
* incomplete and the fix is one command).
|
|
24
|
+
* - could not evaluate (not a git work tree, malformed config) →
|
|
25
|
+
* `ok: false, severity: 'warning'` so the doctor reports the blind
|
|
26
|
+
* spot without flipping the exit code on an unrelated failure.
|
|
27
|
+
*/
|
|
28
|
+
import { getErrorMessage } from 'peaks-loop-shared/result';
|
|
29
|
+
import { inspectCodegraphExcludeIntegrity, isCodegraphExcludeConfigPresent } from '../../../codegraph/codegraph-exclude-integrity.js';
|
|
30
|
+
const CHECK_ID = 'capability:codegraph-exclude-integrity';
|
|
31
|
+
/** How many rules / offending files the message names before eliding. */
|
|
32
|
+
const MAX_NAMED = 5;
|
|
33
|
+
function defaultProbe() {
|
|
34
|
+
const projectRoot = process.cwd();
|
|
35
|
+
// No config → codegraph was never initialized here, so no exclude
|
|
36
|
+
// list is in play and there is nothing to report.
|
|
37
|
+
return isCodegraphExcludeConfigPresent(projectRoot)
|
|
38
|
+
? inspectCodegraphExcludeIntegrity(projectRoot)
|
|
39
|
+
: null;
|
|
40
|
+
}
|
|
41
|
+
function renderGapMessage(excludedTrackedCount, trackedSourceCount, rulesToRemove, violations) {
|
|
42
|
+
const namedRules = rulesToRemove.slice(0, MAX_NAMED).join(', ');
|
|
43
|
+
const elidedRules = rulesToRemove.length > MAX_NAMED ? `, … (+${rulesToRemove.length - MAX_NAMED})` : '';
|
|
44
|
+
const namedFiles = violations
|
|
45
|
+
.slice(0, MAX_NAMED)
|
|
46
|
+
.map((violation) => `${violation.path} <- ${violation.matchedRule}`)
|
|
47
|
+
.join('; ');
|
|
48
|
+
const elidedFiles = violations.length > MAX_NAMED ? `; … (+${violations.length - MAX_NAMED})` : '';
|
|
49
|
+
return `codegraph index is incomplete: ${excludedTrackedCount} of ${trackedSourceCount} tracked source files are blocked by ${rulesToRemove.length} exclude rule(s) [${namedRules}${elidedRules}]. Blocked: ${namedFiles}${elidedFiles}. Run \`peaks codegraph repair-exclude --project <root>\` to drop them and rebuild the index.`;
|
|
50
|
+
}
|
|
51
|
+
function run({ options }) {
|
|
52
|
+
const probe = options.codegraphIntegrityProbe ?? defaultProbe;
|
|
53
|
+
let report;
|
|
54
|
+
try {
|
|
55
|
+
report = probe();
|
|
56
|
+
}
|
|
57
|
+
catch (error) {
|
|
58
|
+
return [{
|
|
59
|
+
id: CHECK_ID,
|
|
60
|
+
ok: false,
|
|
61
|
+
severity: 'warning',
|
|
62
|
+
message: `codegraph exclude integrity could not be evaluated: ${getErrorMessage(error)}`
|
|
63
|
+
}];
|
|
64
|
+
}
|
|
65
|
+
if (report === null) {
|
|
66
|
+
return [{
|
|
67
|
+
id: CHECK_ID,
|
|
68
|
+
ok: true,
|
|
69
|
+
message: 'codegraph is not initialized in this project (no .codegraph/config.json); the exclude list is not in play yet'
|
|
70
|
+
}];
|
|
71
|
+
}
|
|
72
|
+
if (!report.gap) {
|
|
73
|
+
return [{
|
|
74
|
+
id: CHECK_ID,
|
|
75
|
+
ok: true,
|
|
76
|
+
message: `codegraph exclude list drops no tracked source file (${report.trackedSourceCount} tracked source file(s) admitted by include)`
|
|
77
|
+
}];
|
|
78
|
+
}
|
|
79
|
+
return [{
|
|
80
|
+
id: CHECK_ID,
|
|
81
|
+
ok: false,
|
|
82
|
+
message: renderGapMessage(report.excludedTrackedCount, report.trackedSourceCount, report.rulesToRemove, report.violations)
|
|
83
|
+
}];
|
|
84
|
+
}
|
|
85
|
+
export const check = {
|
|
86
|
+
name: 'codegraph-exclude-integrity',
|
|
87
|
+
run
|
|
88
|
+
};
|
|
@@ -42,6 +42,7 @@ import { check as workspaceInit } from './checks/workspace-init.js';
|
|
|
42
42
|
import { check as statuslineInstall } from './checks/statusline-install.js';
|
|
43
43
|
import { check as statuslineRuntime } from './checks/statusline-runtime.js';
|
|
44
44
|
import { check as codegraphCapability } from './checks/codegraph-capability.js';
|
|
45
|
+
import { check as codegraphExcludeIntegrity } from './checks/codegraph-exclude-integrity.js';
|
|
45
46
|
import { check as distSourceVersion } from './checks/dist-source-version.js';
|
|
46
47
|
import { check as multiBinaryDrift } from './checks/multi-binary-drift.js';
|
|
47
48
|
import { check as workspaceLayout } from './checks/workspace-layout.js';
|
|
@@ -70,6 +71,7 @@ export const PLUGINS = [
|
|
|
70
71
|
statuslineInstall, // id "statusline:install"
|
|
71
72
|
statuslineRuntime, // id "statusline:runtime"
|
|
72
73
|
codegraphCapability, // id "capability:codegraph"
|
|
74
|
+
codegraphExcludeIntegrity, // id "capability:codegraph-exclude-integrity"
|
|
73
75
|
distSourceVersion, // id "build:dist-version-matches-source"
|
|
74
76
|
multiBinaryDrift, // id "build:multi-binary-drift"
|
|
75
77
|
workspaceLayout, // id "build:workspace-layout-canonical"
|
|
@@ -75,6 +75,25 @@ export type CodegraphCapabilityProbe = {
|
|
|
75
75
|
*/
|
|
76
76
|
managedPath: CodegraphManagedPathInfo | null;
|
|
77
77
|
};
|
|
78
|
+
/**
|
|
79
|
+
* Structural shape of the codegraph exclude-integrity report the
|
|
80
|
+
* `capability:codegraph-exclude-integrity` check gates on. Declared
|
|
81
|
+
* structurally (rather than imported from the codegraph service) to
|
|
82
|
+
* keep this type module dependency-free — the default probe returns a
|
|
83
|
+
* `CodegraphExcludeIntegrityReport`, which is assignable here.
|
|
84
|
+
*/
|
|
85
|
+
export type CodegraphExcludeIntegrityProbe = {
|
|
86
|
+
readonly configPath: string;
|
|
87
|
+
readonly gap: boolean;
|
|
88
|
+
readonly trackedSourceCount: number;
|
|
89
|
+
readonly excludedTrackedCount: number;
|
|
90
|
+
readonly rulesToRemove: readonly string[];
|
|
91
|
+
/** One entry per (file, rule) pair. */
|
|
92
|
+
readonly violations: readonly {
|
|
93
|
+
readonly path: string;
|
|
94
|
+
readonly matchedRule: string;
|
|
95
|
+
}[];
|
|
96
|
+
};
|
|
78
97
|
export type DistVersionComparison = {
|
|
79
98
|
dist: string | null;
|
|
80
99
|
source: string;
|
|
@@ -229,6 +248,14 @@ export type DoctorOptions = {
|
|
|
229
248
|
* `process.cwd()`.
|
|
230
249
|
*/
|
|
231
250
|
codegraphManagedPathProbe?: () => CodegraphManagedPathInfo | null;
|
|
251
|
+
/**
|
|
252
|
+
* Optional override for the `capability:codegraph-exclude-integrity`
|
|
253
|
+
* check. Returns the integrity report, or `null` when codegraph is
|
|
254
|
+
* not initialized in the inspected root (nothing to reconcile). When
|
|
255
|
+
* omitted, the check inspects `process.cwd()`. Throwing is allowed
|
|
256
|
+
* and reported as a non-blocking warning.
|
|
257
|
+
*/
|
|
258
|
+
codegraphIntegrityProbe?: () => CodegraphExcludeIntegrityProbe | null;
|
|
232
259
|
skillPresenceProbe?: () => DoctorSkillPresence | null;
|
|
233
260
|
skillPresenceFreshnessThresholdMs?: number;
|
|
234
261
|
statusLineInstalledProbe?: () => boolean;
|
|
@@ -9,17 +9,350 @@ export interface LlmRunner {
|
|
|
9
9
|
};
|
|
10
10
|
}>;
|
|
11
11
|
}
|
|
12
|
-
import type { FinalReviewOutput } from './final-review-types.js';
|
|
12
|
+
import type { DimensionKind, FinalReviewOutput } from './final-review-types.js';
|
|
13
13
|
import type { CapabilityAuditResult } from '../capability-audit-service/types.js';
|
|
14
14
|
export interface PrepareFinalReviewOptions {
|
|
15
15
|
readonly projectRoot: string;
|
|
16
16
|
readonly sessionId: string;
|
|
17
17
|
readonly llmRunner: LlmRunner;
|
|
18
|
+
/**
|
|
19
|
+
* Explicit base ref for the pre/post baseline diff. Unset ⇒ the producer
|
|
20
|
+
* resolves the merge-base with the upstream default branch itself, and
|
|
21
|
+
* reports the dimension `unavailable` when it cannot.
|
|
22
|
+
*/
|
|
23
|
+
readonly baseRef?: string;
|
|
18
24
|
}
|
|
19
25
|
export declare class IncompleteFinalReviewError extends Error {
|
|
20
26
|
readonly code: "INCOMPLETE_FINAL_REVIEW";
|
|
21
27
|
constructor(message: string);
|
|
22
28
|
}
|
|
29
|
+
/**
|
|
30
|
+
* N4 — the reply carried no text block at all.
|
|
31
|
+
*
|
|
32
|
+
* Measured 2/3 on this repo's own machine, and it is NOT truncation: the
|
|
33
|
+
* provider answered with a response whose `content` has no `text` block (a
|
|
34
|
+
* reasoning-only turn, a refusal, or a content filter), so there is no JSON to
|
|
35
|
+
* parse and no budget to raise — an operator sent to "raise the budget" for
|
|
36
|
+
* this failure would be sent the wrong way. It gets its own class, its own
|
|
37
|
+
* `code`, and a message that says so, so it is diagnosable instead of being
|
|
38
|
+
* flattened into "not valid JSON".
|
|
39
|
+
*/
|
|
40
|
+
export declare class EmptyReviewReplyError extends Error {
|
|
41
|
+
readonly code: "EMPTY_FINAL_REVIEW_REPLY";
|
|
42
|
+
constructor(message: string);
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* How many times an empty reply is retried before it is reported. The failure
|
|
46
|
+
* was 2/3 on the observed machine — intermittent, not systematic — so a small
|
|
47
|
+
* bounded retry converts most of it into a completed review, while 3 attempts
|
|
48
|
+
* keeps a genuinely broken provider from being hammered.
|
|
49
|
+
*/
|
|
50
|
+
export declare const MAX_EMPTY_REPLY_ATTEMPTS = 3;
|
|
51
|
+
/**
|
|
52
|
+
* Total evidence budget across all sources. 40 KB ≈ 10k tokens of input, which
|
|
53
|
+
* keeps the prompt far inside any modern context window. Sources that do not
|
|
54
|
+
* fit are reported as OMITTED — never dropped silently.
|
|
55
|
+
*
|
|
56
|
+
* This is an INPUT cap and stays fixed per run. The output ceiling that has to
|
|
57
|
+
* sit opposite it is derived per call by `outputBudgetForEvidence()` below —
|
|
58
|
+
* the two used to drift apart, and that drift was the defect.
|
|
59
|
+
*
|
|
60
|
+
* RE-EVALUATED 2026-09-12 (F-BLOCK) — 32 KiB did NOT hold once the tenth
|
|
61
|
+
* source (`final-review-pre-post-diff`) was appended. Measured on this repo's
|
|
62
|
+
* own run (`2026-09-12-session-e37ef0`, rid
|
|
63
|
+
* `2026-09-12-codegraph-exclude-integrity`), the ten sources are 2,621 / 8,164
|
|
64
|
+
* / 9,492 / 10,839 / 11,422 / 13,050 / 13,462 / 13,852 / 17,745 / 20,543
|
|
65
|
+
* bytes — 121,190 bytes on disk, 76,321 bytes once the 8 KiB per-file cap is
|
|
66
|
+
* applied. Four sources at the cap spent the old 32,768 to the byte, so the
|
|
67
|
+
* appended tenth (added LAST, by design) was the first block the budget
|
|
68
|
+
* dropped — on every run, and on the compact set QA measured too (9 x 3.8 KB
|
|
69
|
+
* ≈ 34 KB, which the old cap already could not hold).
|
|
70
|
+
*
|
|
71
|
+
* 40 KiB is a budget that fits the measured ten-source set at the allocator's
|
|
72
|
+
* per-source unit (5 x 8 KiB units + the smaller ones), and it is the largest
|
|
73
|
+
* this budget may grow to today: the derived output ceiling is
|
|
74
|
+
* `3000 + 12288 + bytes/4`, so it reaches MAX_OUTPUT_TOKENS (32,000) at exactly
|
|
75
|
+
* 66,848 bytes of inlined evidence and is clamped from there on — a cap at or
|
|
76
|
+
* above that point buys the reviewer no more output room at all.
|
|
77
|
+
*
|
|
78
|
+
* F-CAP — an earlier version of this comment claimed 40 KiB (40,960 =
|
|
79
|
+
* 10 x 4,096) was "the smallest cap that affords one floor-sized slice to EVERY
|
|
80
|
+
* source in the ten-source set". That was arithmetic about a budget, not a
|
|
81
|
+
* statement about the allocator: sources are capped at
|
|
82
|
+
* MAX_EVIDENCE_BYTES_PER_FILE (10 KiB each, derived — see below) while a floor
|
|
83
|
+
* is per DIMENSION (four of them), so 10 x 4,096 was never what any source
|
|
84
|
+
* received. Re-measured on
|
|
85
|
+
* the same two saturated fixtures (QA round 4: the 9 x 40 KB fixture and this
|
|
86
|
+
* repo's own file sizes), raising the cap from 32 KiB to 40 KiB left **4 of the
|
|
87
|
+
* 10 sources inlined with zero bytes** — `rd/security-review`,
|
|
88
|
+
* `rd/bug-analysis`, `prd/handoff` and the appended
|
|
89
|
+
* `final-review-pre-post-diff` — and it did not deliver the baseline either;
|
|
90
|
+
* all it did was change WHICH source was starved (at 32 KiB that source was
|
|
91
|
+
* `rd/code-review`).
|
|
92
|
+
*
|
|
93
|
+
* So read this constant as "how much evidence fits", never as "which sources
|
|
94
|
+
* arrive". Arrival is decided by the allocator below (one whole unit per source,
|
|
95
|
+
* with a dimension's floor reserving its holder's unit) and, for the two sources
|
|
96
|
+
* a dimension's verdict may not outlive, by the delivery gates on top of it.
|
|
97
|
+
* And note the honest consequence, stated rather than papered over: on a
|
|
98
|
+
* saturated run the allocator still omits sources, so `prd/handoff.md` and the
|
|
99
|
+
* pre/post diff can be dropped — which is exactly why
|
|
100
|
+
* `enforceScopeContractDelivery()` and `enforcePrePostDiffAvailability()` exist
|
|
101
|
+
* and why a `pass` on those two dimensions is keyed on delivery, not on
|
|
102
|
+
* existence.
|
|
103
|
+
*/
|
|
104
|
+
export declare const MAX_EVIDENCE_BYTES_TOTAL: number;
|
|
105
|
+
/**
|
|
106
|
+
* Per-file evidence cap — and the allocator's UNIT. DERIVED, never typed.
|
|
107
|
+
*
|
|
108
|
+
* F5 — this used to be the literal `8 * 1024`, and the whole floor guarantee
|
|
109
|
+
* below was silently conditioned on `4 x perFile <= total`: the reservation
|
|
110
|
+
* promises every pending holder its own unit, and that promise holds only while
|
|
111
|
+
* the four units fit the budget together. At `8 * 1024` the inequality held by
|
|
112
|
+
* coincidence (4 x 8,192 = 32,768 <= 40,960) and nothing in the code said so —
|
|
113
|
+
* raising the cap past 10,240 would have starved all four dimensions in the
|
|
114
|
+
* same run, which reads as four independent red gates rather than one broken
|
|
115
|
+
* constant. The cap is therefore DIVIDED OUT OF the total: the relation is
|
|
116
|
+
* definitional instead of remembered, and `assertFloorReservationAffordable()`
|
|
117
|
+
* still checks it at the point of use (the division must also be exact).
|
|
118
|
+
*
|
|
119
|
+
* The value is 10,240 because that is `40,960 / 4` — the largest unit the
|
|
120
|
+
* four-dimension reservation can afford. It is also the smallest cap at which
|
|
121
|
+
* the decisive evidence source of this repo's own run can be delivered WHOLE:
|
|
122
|
+
* `qa/test-reports/<rid>.md` measured 9,492 bytes there, and `whole` is the
|
|
123
|
+
* delivery rule for every source whose producer publishes no conclusion literal
|
|
124
|
+
* (see `isDelivered`), so a cap under 9,492 makes `problem-resolution` and
|
|
125
|
+
* `no-new-bugs` undeliverable on every run — the always-red gate this module
|
|
126
|
+
* refuses to ship. Enforced in BYTES against the raw buffer, so multi-byte
|
|
127
|
+
* (CJK) content cannot slip past the cap.
|
|
128
|
+
*
|
|
129
|
+
* A source is inlined as `min(bytes, this cap)` — its whole unit — or not at
|
|
130
|
+
* all. So a `TRUNCATED` source block can only ever mean "the FILE is bigger
|
|
131
|
+
* than this cap"; it can no longer mean "the budget ran out while this file was
|
|
132
|
+
* being copied in". That distinction is the point of the all-or-nothing
|
|
133
|
+
* allocator below: the second reading is what let one byte of a 4,226-byte
|
|
134
|
+
* artifact be counted as delivered evidence (F-BLOCK-1BYTE).
|
|
135
|
+
*/
|
|
136
|
+
export declare const MAX_EVIDENCE_BYTES_PER_FILE: number;
|
|
137
|
+
/**
|
|
138
|
+
* Floor — also the value that shipped before this fix, so no evidence set can
|
|
139
|
+
* end up with a smaller budget than it had. ~3000 tokens is enough for the
|
|
140
|
+
* envelope skeleton plus a short paragraph per dimension.
|
|
141
|
+
*/
|
|
142
|
+
export declare const MIN_OUTPUT_TOKENS = 3000;
|
|
143
|
+
/**
|
|
144
|
+
* Headroom for the part of the reply that is not the envelope.
|
|
145
|
+
*
|
|
146
|
+
* A Messages-API-compatible endpoint applies `max_tokens` to the WHOLE
|
|
147
|
+
* response, and a reasoning model spends it on hidden reasoning before it
|
|
148
|
+
* emits a single character of the 4-dim envelope. Measured on this repo's own
|
|
149
|
+
* machine (2026-09-12, rid `2026-09-12-codegraph-exclude-integrity`,
|
|
150
|
+
* `deepseek-flash[1M]` via `api.deepseek.com/anthropic`): `max_tokens=8192`
|
|
151
|
+
* came back with `output_tokens=8192` and only **574** visible characters —
|
|
152
|
+
* the entire budget went to reasoning. A bytes-per-token estimate of the
|
|
153
|
+
* visible output cannot see that cost, which is why the previous formula
|
|
154
|
+
* budgeted 7096 for a reply that needs 10108 — and why a 13240 budget still
|
|
155
|
+
* truncated on 2 of 10 real runs.
|
|
156
|
+
*
|
|
157
|
+
* 12288 (12 KiB) is sized so the largest evidence pack the input caps allow
|
|
158
|
+
* lands at 25528 (see the formula below; the input cap is 40 KiB as of the
|
|
159
|
+
* F-BLOCK re-evaluation, and this floor was checked against the raised cap, not
|
|
160
|
+
* against the 32 KiB the measurements above were taken at — a 12 KiB headroom
|
|
161
|
+
* under a bigger cap is the conservative direction) — about 1.9x the largest
|
|
162
|
+
* value ever OBSERVED to truncate (13240), which is the margin the observed
|
|
163
|
+
* variance asks for. The numbers are in the block comment above.
|
|
164
|
+
*/
|
|
165
|
+
export declare const REASONING_HEADROOM_TOKENS: number;
|
|
166
|
+
/**
|
|
167
|
+
* Ceiling, 32_000: the value a real run on this machine was forced to in order
|
|
168
|
+
* to complete the envelope at all, and the largest this endpoint was observed
|
|
169
|
+
* to accept. 16384 was tried first and truncated 2/10 — a ceiling that is
|
|
170
|
+
* merely "above the last successful measurement" is not above the requirement,
|
|
171
|
+
* because the requirement moves with the model's reasoning spend.
|
|
172
|
+
*
|
|
173
|
+
* A model that caps output at 8192 will refuse this. That is still strictly
|
|
174
|
+
* better than shipping a budget measured to be too small, and the env lever
|
|
175
|
+
* below lets an operator pull it down without a code change.
|
|
176
|
+
*/
|
|
177
|
+
export declare const MAX_OUTPUT_TOKENS = 32000;
|
|
178
|
+
/**
|
|
179
|
+
* Environment lever. The old failure message told the operator to "raise the
|
|
180
|
+
* budget" while the budget was a module constant with no CLI flag and no env
|
|
181
|
+
* var — an instruction that could not be carried out from any surface the
|
|
182
|
+
* operator has. This is that lever.
|
|
183
|
+
*
|
|
184
|
+
* The value is the output ceiling in tokens; it OVERRIDES the derivation below
|
|
185
|
+
* (it is not a bonus added to it). Unset/invalid/out-of-range handling is in
|
|
186
|
+
* `resolveOutputBudget`.
|
|
187
|
+
*/
|
|
188
|
+
export declare const MAX_OUTPUT_TOKENS_ENV = "PEAKS_FINAL_REVIEW_MAX_OUTPUT_TOKENS";
|
|
189
|
+
/**
|
|
190
|
+
* Absolute upper bound the env lever may reach. An endpoint that accepts
|
|
191
|
+
* `max_tokens` at all accepts this; anything above it is a typo (a stray extra
|
|
192
|
+
* digit), not an intent, and clamping is safer than sending it.
|
|
193
|
+
*/
|
|
194
|
+
export declare const HARD_MAX_OUTPUT_TOKENS = 64000;
|
|
195
|
+
/**
|
|
196
|
+
* Inlined bytes that buy one extra output token — 4:1.
|
|
197
|
+
*
|
|
198
|
+
* This term prices the visible envelope (4 x `summary` + `evidence[]` +
|
|
199
|
+
* `confidence` + `overallSummary`) against the evidence the model is required
|
|
200
|
+
* to cite. It was 8:1, which put the 32 KiB pack at 4096 tokens of visible
|
|
201
|
+
* output; the same pack has been observed to complete at 10108 and to truncate
|
|
202
|
+
* at 13240, so 8:1 was pricing the visible side BELOW its own measurement.
|
|
203
|
+
* 4:1 doubles it to 8192. It is still not treated as the whole budget — see
|
|
204
|
+
* `REASONING_HEADROOM_TOKENS`.
|
|
205
|
+
*/
|
|
206
|
+
export declare const EVIDENCE_BYTES_PER_OUTPUT_TOKEN = 4;
|
|
207
|
+
/**
|
|
208
|
+
* Output ceiling for a call whose prompt carries `includedEvidenceBytes` bytes
|
|
209
|
+
* of inlined evidence. Pure, total, and clamped on both ends — the same
|
|
210
|
+
* evidence pack always yields the same budget.
|
|
211
|
+
*/
|
|
212
|
+
export declare function outputBudgetForEvidence(includedEvidenceBytes: number): number;
|
|
213
|
+
/**
|
|
214
|
+
* The budget the call actually uses: the derived one, unless
|
|
215
|
+
* `PEAKS_FINAL_REVIEW_MAX_OUTPUT_TOKENS` overrides it.
|
|
216
|
+
*
|
|
217
|
+
* An override that is not a positive integer THROWS rather than being ignored:
|
|
218
|
+
* a silent fallback would leave an operator who passed a bad value with the
|
|
219
|
+
* exact experience this lever exists to remove — a budget they cannot move.
|
|
220
|
+
* Out-of-range values are clamped, not rejected, so a model needing more than
|
|
221
|
+
* `HARD_MAX_OUTPUT_TOKENS` (or a model needing less than `MIN_OUTPUT_TOKENS`)
|
|
222
|
+
* still gets a call made.
|
|
223
|
+
*/
|
|
224
|
+
export declare function resolveOutputBudget(includedEvidenceBytes: number, env?: NodeJS.ProcessEnv): number;
|
|
225
|
+
/**
|
|
226
|
+
* What DELIVERED means for one source — the module's ONE delivery definition,
|
|
227
|
+
* declared per source and read in exactly one place (`isDelivered`).
|
|
228
|
+
*
|
|
229
|
+
* whole the reviewer must have received the whole document. A
|
|
230
|
+
* truncated slice is not a weaker version of a document, it is a
|
|
231
|
+
* DIFFERENT document, and the conclusion may be in the part that
|
|
232
|
+
* was cut — which is precisely the state `found` used to call
|
|
233
|
+
* "delivered".
|
|
234
|
+
* conclusion the reviewer must have received the literal that IS the
|
|
235
|
+
* document's conclusion. Used where the producer publishes one
|
|
236
|
+
* (the pre/post diff opens with its `VERDICT:` line).
|
|
237
|
+
*
|
|
238
|
+
* `whole` is the rule wherever the producer is another role's skill and
|
|
239
|
+
* publishes no conclusion literal: the module may not GUESS where a document's
|
|
240
|
+
* conclusion lives. The two are the same judgement — "did the reviewer receive
|
|
241
|
+
* the conclusion" — checked at the only place the module can check it.
|
|
242
|
+
*/
|
|
243
|
+
type DeliveryRule = {
|
|
244
|
+
readonly kind: 'whole';
|
|
245
|
+
} | {
|
|
246
|
+
readonly kind: 'conclusion';
|
|
247
|
+
readonly marker: string;
|
|
248
|
+
};
|
|
249
|
+
interface EvidenceSource {
|
|
250
|
+
/** Stable id quoted by the model in its citations. */
|
|
251
|
+
readonly key: string;
|
|
252
|
+
readonly label: string;
|
|
253
|
+
/** Path segments under `.peaks/_runtime/<sessionId>/`. */
|
|
254
|
+
readonly segments: readonly string[];
|
|
255
|
+
/** Dimensions this source can supply evidence for. */
|
|
256
|
+
readonly supports: readonly DimensionKind[];
|
|
257
|
+
/** What DELIVERED means for this source. See `isDelivered()` — every source
|
|
258
|
+
* must declare one, and the declaration is the only thing the module's
|
|
259
|
+
* delivery judgement reads. */
|
|
260
|
+
readonly delivery: DeliveryRule;
|
|
261
|
+
}
|
|
262
|
+
/**
|
|
263
|
+
* `found` is the only status that carries bytes. The other four exist so the
|
|
264
|
+
* prompt can name *why* a source carries nothing: an absent file, one that
|
|
265
|
+
* exists but is empty, one that exists but could not be READ, and one that did
|
|
266
|
+
* not fit the byte budget are four different facts, and the model is told all
|
|
267
|
+
* four explicitly.
|
|
268
|
+
*
|
|
269
|
+
* F4 — `missing` and `unreadable` used to be one status. `readFileSync`'s catch
|
|
270
|
+
* collapsed EACCES / EBUSY / EPERM into the same `raw === null` as ENOENT, so a
|
|
271
|
+
* contract file that EXISTS and could not be opened was reported to the
|
|
272
|
+
* reviewer — and, worse, to the delivery gate — as "there was no PRD phase".
|
|
273
|
+
* Those are opposite facts about a run: one says "nothing to deliver", the
|
|
274
|
+
* other says "there is something to deliver and it did not arrive".
|
|
275
|
+
*/
|
|
276
|
+
type EvidenceStatus = 'found' | 'empty' | 'missing' | 'unreadable' | 'omitted';
|
|
277
|
+
interface CollectedEvidence {
|
|
278
|
+
readonly source: EvidenceSource;
|
|
279
|
+
readonly relativePath: string;
|
|
280
|
+
readonly absolutePath: string;
|
|
281
|
+
readonly status: EvidenceStatus;
|
|
282
|
+
/** Full size on disk (0 when nothing could be read). */
|
|
283
|
+
readonly totalBytes: number;
|
|
284
|
+
/** Bytes actually inlined into the prompt. */
|
|
285
|
+
readonly includedBytes: number;
|
|
286
|
+
readonly content: string;
|
|
287
|
+
/** Why this source carries no evidence (non-`found` statuses only). */
|
|
288
|
+
readonly reason: string;
|
|
289
|
+
}
|
|
290
|
+
/**
|
|
291
|
+
* F5 — the floor promise, checked where it is relied on instead of remembered.
|
|
292
|
+
*
|
|
293
|
+
* The allocator's loop invariant is `budgetLeft >= sum(units of pending
|
|
294
|
+
* holders)`, and its INITIAL condition is
|
|
295
|
+
* `REQUIRED_DIMENSIONS.length x MAX_EVIDENCE_BYTES_PER_FILE <=
|
|
296
|
+
* MAX_EVIDENCE_BYTES_TOTAL`. While that holds, every holder is served when it
|
|
297
|
+
* is reached and the reservation is a guarantee; the moment it fails, all four
|
|
298
|
+
* dimensions are starved in the same run — which surfaces as four independent
|
|
299
|
+
* red gates rather than as one broken constant, and is therefore the kind of
|
|
300
|
+
* breakage nobody diagnoses correctly.
|
|
301
|
+
*
|
|
302
|
+
* The cap is derived from the total so the inequality cannot be typed wrong,
|
|
303
|
+
* and this check covers the two ways a derivation can still go bad: a
|
|
304
|
+
* non-integer quotient (a fifth dimension, say) and a future re-typing of
|
|
305
|
+
* either constant. It is exported because a test asserts it directly.
|
|
306
|
+
*/
|
|
307
|
+
export declare function assertFloorReservationAffordable(): void;
|
|
308
|
+
/**
|
|
309
|
+
* H2 — the reason a dimension is red PERMANENTLY, as opposed to merely unfed on
|
|
310
|
+
* this run. Reported per source, so the arithmetic is checkable by the reader.
|
|
311
|
+
*/
|
|
312
|
+
export interface UndeliverableDimensionEvidence {
|
|
313
|
+
readonly dimension: DimensionKind;
|
|
314
|
+
readonly sources: readonly {
|
|
315
|
+
readonly key: string;
|
|
316
|
+
readonly relativePath: string;
|
|
317
|
+
readonly totalBytes: number;
|
|
318
|
+
}[];
|
|
319
|
+
}
|
|
320
|
+
/**
|
|
321
|
+
* H2 — every required dimension whose verdict is locked to `inconclusive` by
|
|
322
|
+
* BYTE ARITHMETIC rather than by the reviewer's judgement.
|
|
323
|
+
*
|
|
324
|
+
* The defect this exists for: `qa/test-reports/<rid>.md` measured 9,492 bytes
|
|
325
|
+
* on this repo's own run and is the ONLY source on disk for `problem-resolution`
|
|
326
|
+
* and `no-new-bugs` (the other sources that support them are absent), against a
|
|
327
|
+
* per-file cap of 10,240 — a 748-byte margin on a file that is REWRITTEN every
|
|
328
|
+
* round and only grows. The moment it crosses 10,240 both dimensions go
|
|
329
|
+
* permanently red, and NOTHING said so: `assertFloorReservationAffordable()`
|
|
330
|
+
* only checks the constant-level relation (`4 x cap <= total`), never whether
|
|
331
|
+
* any actual source fits the cap it must live under, so the red handoff read as
|
|
332
|
+
* "the reviewer was unsure" instead of "no evidence can ever reach the
|
|
333
|
+
* reviewer". A gate that is always red and never explains itself is noise, and
|
|
334
|
+
* this is the same "always red" harm the floor reservation was built to remove
|
|
335
|
+
* — one layer down.
|
|
336
|
+
*
|
|
337
|
+
* The conditions, all three of which must hold, are chosen so the report cannot
|
|
338
|
+
* be noise:
|
|
339
|
+
* 1. NOTHING on disk can back the dimension (no `isDelivered`), and
|
|
340
|
+
* 2. there IS something on disk to deliver (a source that was never written
|
|
341
|
+
* is not a delivery failure — same reasoning as the scope-contract gate's
|
|
342
|
+
* `missing` exemption: a run with no QA phase has no report to lose), and
|
|
343
|
+
* 3. EVERY one of those on-disk sources is structurally undeliverable — a
|
|
344
|
+
* single source that merely did not fit TODAY (budget-exhausted `omitted`)
|
|
345
|
+
* is a different, self-correcting state and is left to the allocator.
|
|
346
|
+
*
|
|
347
|
+
* The report is consumed by the prompt (stated to the reviewer), by the
|
|
348
|
+
* envelope (a marker on each dimension's summary) and by `needsAttention`
|
|
349
|
+
* (which also clears `allPass`) — see `enforceDeliveryReachability` and
|
|
350
|
+
* `renderDeliveryReachabilityStatus`. It is a LOUD STATEMENT, not a silent
|
|
351
|
+
* downgrade, and it is deliberately not a throw: a crash would destroy the
|
|
352
|
+
* evidence for the three dimensions that ARE deliverable, and the honest fact
|
|
353
|
+
* here is per-dimension, so it is reported per-dimension.
|
|
354
|
+
*/
|
|
355
|
+
export declare function undeliverableDimensions(collected: readonly CollectedEvidence[]): readonly UndeliverableDimensionEvidence[];
|
|
23
356
|
export declare function prepareFinalReview(rid: string, opts: PrepareFinalReviewOptions): Promise<FinalReviewOutput>;
|
|
24
357
|
export declare function decideFifthDimension(input: {
|
|
25
358
|
readonly audit: CapabilityAuditResult | null;
|
|
@@ -28,3 +361,4 @@ export declare function decideFifthDimension(input: {
|
|
|
28
361
|
readonly verdict: 'pass' | 'fail' | 'inconclusive';
|
|
29
362
|
readonly reason: string;
|
|
30
363
|
};
|
|
364
|
+
export {};
|