peaks-loop 4.0.46 → 4.0.48
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +54 -0
- package/README-en.md +1 -1
- package/README.md +1 -1
- package/agents/karpathy-reviewer.md +11 -10
- package/dist/cli/cli-helpers.d.ts +34 -0
- package/dist/cli/cli-helpers.js +57 -0
- package/dist/cli/commands/code-job-shape-commands.js +8 -0
- package/dist/cli/commands/code-runtime-commands.d.ts +22 -0
- package/dist/cli/commands/code-runtime-commands.js +139 -16
- package/dist/cli/commands/compact-command.js +241 -1
- package/dist/cli/commands/config-commands.js +15 -9
- package/dist/cli/commands/container-commands.js +3 -3
- package/dist/cli/commands/core/skill-command.js +45 -10
- package/dist/cli/commands/cron-commands.js +2 -1
- package/dist/cli/commands/dashboard-long-run.js +6 -0
- package/dist/cli/commands/dispatch-commands.js +11 -1
- package/dist/cli/commands/doctor/invoke-from-code.js +6 -0
- package/dist/cli/commands/e2e-verify.js +3 -3
- package/dist/cli/commands/governance-classify-contract-commands.js +1 -0
- package/dist/cli/commands/hooks-commands.js +14 -5
- package/dist/cli/commands/job-commands.js +8 -0
- package/dist/cli/commands/loop-commands.js +1 -0
- package/dist/cli/commands/loop-eval-commands.js +15 -0
- package/dist/cli/commands/perf-audit-commands.js +2 -0
- package/dist/cli/commands/playwright-commands.js +14 -1
- package/dist/cli/commands/prd-commands.js +1 -1
- package/dist/cli/commands/qa-commands.js +22 -0
- package/dist/cli/commands/reinject-command.d.ts +72 -0
- package/dist/cli/commands/reinject-command.js +174 -0
- package/dist/cli/commands/request-commands.js +14 -3
- package/dist/cli/commands/scan-commands.js +1 -1
- package/dist/cli/commands/security-audit-commands.js +2 -0
- package/dist/cli/commands/shadcn-commands.js +1 -0
- package/dist/cli/commands/slice-integrate-commands.js +5 -0
- package/dist/cli/commands/statusline-commands.js +44 -4
- package/dist/cli/commands/sub-agent/detached.d.ts +14 -1
- package/dist/cli/commands/sub-agent/detached.js +47 -22
- package/dist/cli/commands/sub-agent-shutdown-commands.js +11 -0
- package/dist/cli/commands/test-commands.js +2 -1
- package/dist/cli/commands/verdict-aggregate-command.js +95 -13
- package/dist/cli/commands/vm-commands.js +7 -7
- package/dist/cli/commands/workflow-commands.js +1 -1
- package/dist/cli/commands/workspace/init-command.js +24 -2
- package/dist/cli/commands/worktree-lease-commands.js +4 -4
- package/dist/cli/index.js +10 -3
- package/dist/cli/program.js +5 -0
- package/dist/hooks/pre-tool-use-sub-agent.js +1 -1
- package/dist/services/adapter/adapter-registry.js +1 -1
- package/dist/services/artifacts/artifact-prerequisites.d.ts +38 -7
- package/dist/services/artifacts/artifact-prerequisites.js +130 -65
- package/dist/services/artifacts/artifact-service.js +1 -1
- package/dist/services/artifacts/request-artifact-service.d.ts +8 -0
- package/dist/services/artifacts/request-artifact-service.js +18 -8
- package/dist/services/artifacts/request-artifact-state-helpers.d.ts +57 -0
- package/dist/services/artifacts/request-artifact-state-helpers.js +91 -10
- package/dist/services/audit-independent/perf-audit-service.d.ts +9 -0
- package/dist/services/audit-independent/perf-audit-service.js +27 -5
- package/dist/services/audit-independent/security-audit-service.d.ts +12 -2
- package/dist/services/audit-independent/security-audit-service.js +28 -6
- package/dist/services/capability-guard-runner/contracts/J01.js +2 -1
- package/dist/services/capability-guard-runner/contracts/J02.js +3 -3
- package/dist/services/capability-guard-runner/contracts/J04.js +4 -2
- package/dist/services/capability-guard-runner/contracts/J07.js +2 -1
- package/dist/services/code/auto-compact-lifecycle.d.ts +130 -1
- package/dist/services/code/auto-compact-lifecycle.js +180 -4
- package/dist/services/code/auto-compact-orchestrator.d.ts +53 -9
- package/dist/services/code/auto-compact-orchestrator.js +166 -34
- package/dist/services/code/compact-event-settle.d.ts +122 -0
- package/dist/services/code/compact-event-settle.js +219 -0
- package/dist/services/code/orchestrator-can-do.d.ts +4 -2
- package/dist/services/code/orchestrator-can-do.js +37 -5
- package/dist/services/codegraph/codegraph-exclude-reconciler.js +2 -1
- package/dist/services/codegraph/codegraph-process-runner.js +3 -2
- package/dist/services/compact/request-transition-hook.js +5 -2
- package/dist/services/compact-history/compact-history-service.d.ts +75 -0
- package/dist/services/compact-history/compact-history-service.js +49 -0
- package/dist/services/config/config-restore.d.ts +12 -1
- package/dist/services/config/config-restore.js +35 -4
- package/dist/services/config/config-rollback.js +6 -1
- package/dist/services/config/config-safety.d.ts +52 -0
- package/dist/services/config/config-safety.js +75 -1
- package/dist/services/context/auto-compact-dispatcher.d.ts +7 -37
- package/dist/services/context/auto-compact-dispatcher.js +113 -40
- package/dist/services/context/auto-compact-reader.d.ts +68 -28
- package/dist/services/context/auto-compact-reader.js +155 -1
- package/dist/services/context/auto-compact-types.d.ts +89 -12
- package/dist/services/context/auto-compact-types.js +16 -32
- package/dist/services/context/harness-context-witness.d.ts +310 -0
- package/dist/services/context/harness-context-witness.js +606 -0
- package/dist/services/context/harness-window-config.d.ts +412 -0
- package/dist/services/context/harness-window-config.js +607 -0
- package/dist/services/context/main-session-monitor.d.ts +27 -0
- package/dist/services/context/main-session-monitor.js +32 -1
- package/dist/services/context/post-compact-reinjection.d.ts +221 -0
- package/dist/services/context/post-compact-reinjection.js +491 -0
- package/dist/services/dispatch/merge-back-runner.js +5 -5
- package/dist/services/dispatch/service-shutdown.js +3 -3
- package/dist/services/doc/doc-generator.js +2 -1
- package/dist/services/env/shell-probe.js +1 -1
- package/dist/services/evidence/evidence-generator.js +86 -49
- package/dist/services/final-review/final-review-service.d.ts +9 -0
- package/dist/services/final-review/final-review-service.js +36 -12
- package/dist/services/fuzzy-matching/fzf-pick-service.js +2 -0
- package/dist/services/hooks/auto-compact-hook-install.d.ts +10 -2
- package/dist/services/hooks/auto-compact-hook-install.js +8 -0
- package/dist/services/ide/adapters/claude-code-adapter.d.ts +107 -3
- package/dist/services/ide/adapters/claude-code-adapter.js +154 -7
- package/dist/services/ide/ide-registry.d.ts +31 -0
- package/dist/services/ide/ide-registry.js +35 -0
- package/dist/services/ide/ide-types.d.ts +59 -0
- package/dist/services/job/job-state-store.js +7 -0
- package/dist/services/lint/detect-eslint.js +2 -2
- package/dist/services/lint/eslint-runner.js +3 -1
- package/dist/services/loop/evaluator-dispatcher.js +2 -1
- package/dist/services/memory/project-memory-service/index/kind-dispatch.js +1 -1
- package/dist/services/memory/project-memory-service/store/paths.d.ts +9 -1
- package/dist/services/memory/project-memory-service/store/paths.js +15 -6
- package/dist/services/polyrepo/polyrepo-dispatcher.js +11 -0
- package/dist/services/prd/best-practice-auto-trigger.js +1 -0
- package/dist/services/prd/handoff-auto-regen.js +31 -27
- package/dist/services/prd/handoff-frontmatter.d.ts +44 -0
- package/dist/services/prd/handoff-frontmatter.js +75 -0
- package/dist/services/prd/handoff-service.d.ts +41 -2
- package/dist/services/prd/handoff-service.js +81 -8
- package/dist/services/prd/handoff-types.d.ts +3 -2
- package/dist/services/prd/handoff-types.js +3 -2
- package/dist/services/qa/qa-business-review-state.js +9 -0
- package/dist/services/release/version-precheck-service.d.ts +2 -1
- package/dist/services/release/version-precheck-service.js +82 -12
- package/dist/services/runtime/vendor-adapter.d.ts +29 -4
- package/dist/services/runtime/vendors/claude-code.js +1 -1
- package/dist/services/runtime/vendors/codex.js +1 -1
- package/dist/services/runtime/vendors/copilot.js +1 -1
- package/dist/services/sc/sc-service.js +1 -1
- package/dist/services/scan/diff-scope-service.js +2 -2
- package/dist/services/scan/file-size-scan.js +2 -2
- package/dist/services/scan/karpathy-service.js +2 -2
- package/dist/services/scan/orphan-service.js +2 -1
- package/dist/services/scan/type-sanity-service.js +2 -2
- package/dist/services/session/session-checkpoint-service.js +8 -0
- package/dist/services/skill/resume-detector.js +29 -11
- package/dist/services/skillhub/tar-runtime.js +1 -0
- package/dist/services/skills/hooks-codegate-superpowers.d.ts +14 -0
- package/dist/services/skills/hooks-codegate-superpowers.js +99 -3
- package/dist/services/skills/hooks-settings-service.d.ts +12 -0
- package/dist/services/skills/hooks-settings-service.js +91 -14
- package/dist/services/skills/session-start-hook-constants.d.ts +86 -0
- package/dist/services/skills/session-start-hook-constants.js +86 -0
- package/dist/services/skills/skill-presence-service.js +9 -0
- package/dist/services/skills/skill-statusline-service.d.ts +14 -0
- package/dist/services/slice/slice-check-service.js +31 -12
- package/dist/services/slice/slice-decompose-runners.js +2 -1
- package/dist/services/slice/slice-review-state.js +8 -0
- package/dist/services/upgrade/upgrade-service.js +1 -0
- package/dist/services/workflow/pipeline-verify-gate-support.d.ts +47 -10
- package/dist/services/workflow/pipeline-verify-gate-support.js +212 -93
- package/dist/services/workflow/pipeline-verify-service.js +24 -23
- package/dist/services/workflow/pipeline-verify-types.d.ts +10 -3
- package/dist/services/workflow/workflow-skip-service.js +2 -1
- package/dist/services/workspace/claude-settings-template.d.ts +56 -8
- package/dist/services/workspace/claude-settings-template.js +98 -20
- package/dist/services/workspace/migrate-service.js +1 -1
- package/dist/services/workspace/workspace-claude-settings-materializer.js +124 -9
- package/dist/services/workspace/workspace-service.js +8 -0
- package/dist/services/worktree/host-worktree-reconciler.js +1 -0
- package/dist/services/worktree/long-path-cleanup.js +3 -2
- package/dist/shared/process.js +1 -1
- package/package.json +6 -6
- package/scripts/install-skills.mjs +1 -0
- package/scripts/watch.mjs +3 -1
- package/skills/bee/peaks-perf-audit/SKILL.md +1 -1
- package/skills/bee/peaks-prd/SKILL.md +8 -6
- package/skills/bee/peaks-qa/SKILL.md +7 -7
- package/skills/bee/peaks-qa/references/qa-runbook.md +2 -2
- package/skills/bee/peaks-qa/references/qa-transition-gates.md +7 -7
- package/skills/bee/peaks-rd/SKILL.md +10 -8
- package/skills/bee/peaks-rd/references/artifact-per-request.md +2 -2
- package/skills/bee/peaks-rd/references/parallel-review-fanout.md +7 -5
- package/skills/bee/peaks-rd/references/rd-fanout-contracts.md +13 -13
- package/skills/bee/peaks-rd/references/rd-runbook.md +9 -5
- package/skills/bee/peaks-rd/references/rd-transition-gates.md +9 -7
- package/skills/bee/peaks-rd/references/writing-handoff-frontmatter.md +6 -6
- package/skills/bee/peaks-reviewer/SKILL.md +1 -1
- package/skills/bee/peaks-sc/SKILL.md +1 -1
- package/skills/bee/peaks-security-audit/SKILL.md +1 -1
- package/skills/bee/peaks-txt/SKILL.md +1 -1
- package/skills/bee/peaks-ui/SKILL.md +1 -1
- package/skills/peaks-audit/SKILL.md +1 -1
- package/skills/peaks-code/SKILL.md +3 -3
- package/skills/peaks-code/references/a2a-artifact-mapping.md +3 -3
- package/skills/peaks-code/references/local-artifact-workspace.md +1 -1
- package/skills/peaks-code/references/resume-detection.md +13 -7
- package/skills/peaks-code/references/runbook.md +3 -2
- package/skills/peaks-code/references/session-overload-signal-index.md +2 -1
- package/skills/peaks-code/references/sub-agent-dispatch.md +1 -1
- package/skills/peaks-code/references/workflow-gates-and-types.md +8 -6
- package/skills/peaks-content/SKILL.md +1 -1
- package/skills/peaks-doctor/SKILL.md +1 -1
- package/skills/peaks-final-review/SKILL.md +1 -1
- package/skills/peaks-ide/SKILL.md +1 -1
- package/skills/peaks-issue-fix-orchestrator/SKILL.md +1 -1
- package/skills/peaks-resume/SKILL.md +1 -1
- package/skills/peaks-slice-decompose/SKILL.md +1 -1
- package/skills/peaks-solo/SKILL.md +1 -1
- package/skills/peaks-sop/SKILL.md +1 -1
- package/skills/peaks-status/SKILL.md +1 -1
- package/skills/peaks-test/SKILL.md +1 -1
|
@@ -0,0 +1,606 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Harness context witness — peaks-loop's second scale, captured from the
|
|
3
|
+
* harness instead of re-derived by peaks-loop.
|
|
4
|
+
*
|
|
5
|
+
* The harness pipes a JSON session payload on the statusline command's stdin,
|
|
6
|
+
* and that payload is the ONLY supported programming channel that carries the
|
|
7
|
+
* harness's OWN context numbers (`context_window.used_percentage` and
|
|
8
|
+
* `context_window.context_window_size`). peaks-loop's own ratio is derived
|
|
9
|
+
* separately (transcript estimate) and divides by a window peaks-loop writes
|
|
10
|
+
* into the harness's settings. Two derivations, two authors — and until now
|
|
11
|
+
* nothing compared them. This module is the comparison.
|
|
12
|
+
*
|
|
13
|
+
* WHAT IS NOT COMPARED, and why — read before changing anything here:
|
|
14
|
+
* `context_window_size` is the MODEL window. peaks-loop's denominator is the
|
|
15
|
+
* AUTO-COMPACT window it configures. They are semantically different objects
|
|
16
|
+
* that happen to be equal on the machine this slice was written on. The
|
|
17
|
+
* comparison is therefore `peaks ratio` vs `used_percentage` — two fractions
|
|
18
|
+
* "used / some window" — and NEVER a comparison of the two windows as if they
|
|
19
|
+
* were one field. A guard that compared the windows would report agreement on
|
|
20
|
+
* the one machine where the two numbers coincide and would have no idea why.
|
|
21
|
+
*
|
|
22
|
+
* `context_window_size` IS read as the denominator of the harness's OWN
|
|
23
|
+
* `used_percentage`, to recover the unit of a payload that does not state one
|
|
24
|
+
* (`readPercentageUnit`). That is reading the harness's field against the
|
|
25
|
+
* harness's own window; it is not a denominator for the comparison above.
|
|
26
|
+
*
|
|
27
|
+
* WHY A RESIDUAL, AND WHY THIS BUDGET (`witnessToleranceTokens` + the residual
|
|
28
|
+
* in `compareHarnessWitness`).
|
|
29
|
+
*
|
|
30
|
+
* WHY A THIRD ANSWER. This guard has three outcomes, not two: after "they
|
|
31
|
+
* agree" and "they disagree" there is "this sample cannot tell". A guard with
|
|
32
|
+
* only two answers cannot distinguish "correctly silent" from "broken", which
|
|
33
|
+
* is the failure mode this repository has a memory about. `unverifiable` is
|
|
34
|
+
* that third answer, and it is the honest one whenever the two numbers are not
|
|
35
|
+
* known to describe the same moment — and also whenever they describe the same
|
|
36
|
+
* moment but the sample is too coarse to separate a real window difference from
|
|
37
|
+
* the budget's own uncertainty, which is the state a small or stale witness
|
|
38
|
+
* puts this guard in.
|
|
39
|
+
*
|
|
40
|
+
* THE FILE IS A RENDER LEDGER, NOT ONLY A WITNESS. It is written on every
|
|
41
|
+
* render that receives a payload, including a render whose payload carried no
|
|
42
|
+
* usable percentage. That is what makes `absent` diagnosable: "no file" means
|
|
43
|
+
* nothing rendered here, and "a file with no percentage" means one did and its
|
|
44
|
+
* payload had nothing to read. See `.peaks/_runtime/<sid>/rd/tech-doc.md` §17.
|
|
45
|
+
*
|
|
46
|
+
* ...BUT A LOWER-INFORMATION RENDER DOES NOT OVERWRITE A HIGHER-INFORMATION
|
|
47
|
+
* ONE (repair cycle 3). The ledger rule is what makes the FIRST render of a
|
|
48
|
+
* session land; the no-downgrade rule is what stops a later contextless render
|
|
49
|
+
* from silencing a comparison the earlier one could still make. `capturedAt`
|
|
50
|
+
* keeps naming the render that produced the surviving record, so nothing on
|
|
51
|
+
* disk claims to be fresher than it is. See `writeHarnessWitness`.
|
|
52
|
+
*
|
|
53
|
+
* COST (the statusline runs this on every render): one small file write on the
|
|
54
|
+
* render path, one small file read on the probe path. No network, no directory
|
|
55
|
+
* walk, no transcript read, no new dependency.
|
|
56
|
+
*/
|
|
57
|
+
import { existsSync, readFileSync, renameSync, rmSync, writeFileSync } from 'node:fs';
|
|
58
|
+
import { dirname } from 'node:path';
|
|
59
|
+
import { getSessionDir } from '../session/getSessionDir.js';
|
|
60
|
+
export const HARNESS_CONTEXT_WITNESS_FILE = 'harness-context-witness.json';
|
|
61
|
+
export const WITNESS_SCHEMA_VERSION = 2;
|
|
62
|
+
/**
|
|
63
|
+
* Contributor 1 of the budget: the harness reports a rounded percentage.
|
|
64
|
+
* "Pre-calculated percentage of context window used" is not documented as
|
|
65
|
+
* fractional, so the conservative reading is an integer percent, i.e. up to
|
|
66
|
+
* ±0.5 percentage points = ±0.005 of the window. If the real payload turns out
|
|
67
|
+
* to carry decimals this term shrinks tenfold and the guard gets sharper; the
|
|
68
|
+
* assumption is visible in the first witness file, where
|
|
69
|
+
* `usageTokens / modelWindowTokens` and `usedPercentage` must agree.
|
|
70
|
+
*/
|
|
71
|
+
export const WITNESS_PERCENT_ROUNDING_FRACTION = 0.005;
|
|
72
|
+
/**
|
|
73
|
+
* Contributor 2: the two sides do not sum exactly the same token quantities.
|
|
74
|
+
* MEASURED, not guessed — on 2026-09-13 the harness's own pre-compact count
|
|
75
|
+
* (963,306 tokens) exceeded peaks-loop's transcript estimate (961,658) by
|
|
76
|
+
* 1,648 tokens = 0.171%. Used as a fraction of the tokens, not of the window.
|
|
77
|
+
*/
|
|
78
|
+
export const WITNESS_NUMERATOR_FRACTION = 0.0017;
|
|
79
|
+
/**
|
|
80
|
+
* The smallest window difference this guard claims to detect, and therefore
|
|
81
|
+
* the yardstick for "is this sample sharp enough to answer at all".
|
|
82
|
+
*
|
|
83
|
+
* 3% is not arbitrary: the harness compacts a native-1M model at ~967,000 by
|
|
84
|
+
* default while peaks-loop writes the model ceiling, so 1,000,000 vs 967,000 —
|
|
85
|
+
* a 3.3% difference — is the smallest real-world disagreement between the two
|
|
86
|
+
* denominators today. Anything the guard reports as "agree" while its own
|
|
87
|
+
* budget is wider than that difference is a claim it cannot support.
|
|
88
|
+
*
|
|
89
|
+
* NOTE (repair cycle 2): the constant is a floor on the SAMPLE, and the sample
|
|
90
|
+
* it floors is the WITNESS's, not peaks-loop's — a gap of this size leaves a
|
|
91
|
+
* residual proportional to the witness's ratio, so a stale (or low) witness
|
|
92
|
+
* shrinks the signal while the budget's rounding term does not shrink with it.
|
|
93
|
+
* See `sampleSupportsAgreement`, which is where this constant is applied. At
|
|
94
|
+
* zero skew it was already correctly calibrated (measured onset 0.176 of the
|
|
95
|
+
* window against a predicted 0.177); the fault was that only zero skew was
|
|
96
|
+
* correctly calibrated.
|
|
97
|
+
*/
|
|
98
|
+
export const MIN_DETECTABLE_WINDOW_DIFFERENCE = 0.03;
|
|
99
|
+
/**
|
|
100
|
+
* The largest raw value a FRACTION reading can carry. Above it the fraction
|
|
101
|
+
* reading is out of range, so the value has exactly ONE in-range reading —
|
|
102
|
+
* percent — and the unit needs no other evidence. (The repo's env / statusline
|
|
103
|
+
* readers use 1.5; that threshold calls values in (1, 1.5] fractions by fiat
|
|
104
|
+
* and then has to throw them away, which is why this reader does not reuse it.)
|
|
105
|
+
*/
|
|
106
|
+
const FRACTION_MAX = 1;
|
|
107
|
+
/** Upper bound of a sane percentage scale; above this the payload is not a percent. */
|
|
108
|
+
const PERCENT_SCALE_MAX = 100;
|
|
109
|
+
/** The three prompt-side components peaks-loop's own `rawTokens` sums. */
|
|
110
|
+
const USAGE_TOKEN_KEYS = ['input_tokens', 'cache_read_input_tokens', 'cache_creation_input_tokens'];
|
|
111
|
+
function finiteNumber(value) {
|
|
112
|
+
return typeof value === 'number' && Number.isFinite(value) ? value : null;
|
|
113
|
+
}
|
|
114
|
+
/**
|
|
115
|
+
* Pick the reading of a raw `used_percentage` whose unit the payload does not
|
|
116
|
+
* state.
|
|
117
|
+
*
|
|
118
|
+
* Above `FRACTION_MAX` only the percent reading is in range, so the value
|
|
119
|
+
* decides. At or below it both readings are in range and the value cannot
|
|
120
|
+
* decide, so the witness's OWN token snapshot does: the two readings differ by
|
|
121
|
+
* exactly 100x, which puts their geometric midpoint at `raw / 10`.
|
|
122
|
+
* `unestablished` when the payload did not carry enough to compute that ratio:
|
|
123
|
+
* the value is in range on both scales and nothing in the payload says which
|
|
124
|
+
* one the harness meant.
|
|
125
|
+
*
|
|
126
|
+
* NOTE: this reads the harness's field against the harness's own window. It is
|
|
127
|
+
* not the comparison's denominator (see the module header).
|
|
128
|
+
*/
|
|
129
|
+
function readPercentageUnit(raw, tokensPerWindow) {
|
|
130
|
+
if (raw > FRACTION_MAX)
|
|
131
|
+
return 'percent';
|
|
132
|
+
if (tokensPerWindow === null)
|
|
133
|
+
return 'unestablished';
|
|
134
|
+
return tokensPerWindow < raw / 10 ? 'percent' : 'fraction';
|
|
135
|
+
}
|
|
136
|
+
/**
|
|
137
|
+
* Normalise a harness percentage to 0..1, keeping the raw value either way.
|
|
138
|
+
*
|
|
139
|
+
* A value on neither scale (negative, or above 100) is refused rather than
|
|
140
|
+
* clamped — a witness nobody can interpret must not be compared — but it is
|
|
141
|
+
* still RETURNED with its raw value attached, because discarding it is what
|
|
142
|
+
* left a refused payload indistinguishable from a render that never happened.
|
|
143
|
+
*
|
|
144
|
+
* A value that is in range on BOTH scales, with no snapshot to settle which,
|
|
145
|
+
* is refused the same way — no reading is taken. Applying the repo's
|
|
146
|
+
* fraction-first convention here instead is what read a "used 1%" payload as
|
|
147
|
+
* 100% and turned a unit misfire into a confident disagreement that blamed the
|
|
148
|
+
* two denominators (see `unusableReason`).
|
|
149
|
+
*/
|
|
150
|
+
function resolvePercentage(raw, tokensPerWindow) {
|
|
151
|
+
const value = finiteNumber(raw);
|
|
152
|
+
if (value === null || value < 0 || value > PERCENT_SCALE_MAX) {
|
|
153
|
+
return { value: null, raw: value, unit: null };
|
|
154
|
+
}
|
|
155
|
+
const unit = readPercentageUnit(value, tokensPerWindow);
|
|
156
|
+
if (unit === 'unestablished')
|
|
157
|
+
return { value: null, raw: value, unit };
|
|
158
|
+
return { value: unit === 'percent' ? value / PERCENT_SCALE_MAX : value, raw: value, unit };
|
|
159
|
+
}
|
|
160
|
+
/** Sum the prompt-side usage components, or `null` when none is present. */
|
|
161
|
+
function sumUsageTokens(currentUsage) {
|
|
162
|
+
if (currentUsage === null || typeof currentUsage !== 'object' || Array.isArray(currentUsage)) {
|
|
163
|
+
return null;
|
|
164
|
+
}
|
|
165
|
+
const record = currentUsage;
|
|
166
|
+
let sum = 0;
|
|
167
|
+
let seen = false;
|
|
168
|
+
for (const key of USAGE_TOKEN_KEYS) {
|
|
169
|
+
const value = finiteNumber(record[key]);
|
|
170
|
+
if (value !== null) {
|
|
171
|
+
sum += value;
|
|
172
|
+
seen = true;
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
return seen ? sum : null;
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
* Read the harness's context block out of a parsed statusline stdin payload.
|
|
179
|
+
* Returns `null` only when there was no payload at all (a render on a TTY, or
|
|
180
|
+
* a manual `peaks statusline`) — in which case there is nothing to record and
|
|
181
|
+
* nothing to say. A payload that arrived but carried no usable percentage is
|
|
182
|
+
* recorded, not dropped.
|
|
183
|
+
*/
|
|
184
|
+
export function parseHarnessWitness(input) {
|
|
185
|
+
if (input.stdin === null || input.stdin === undefined)
|
|
186
|
+
return null;
|
|
187
|
+
const block = input.stdin.context_window;
|
|
188
|
+
// `null` AS WELL AS `undefined` (repair cycle 3). The key being present with
|
|
189
|
+
// no value is how a JSON producer spells "there is no context block", and the
|
|
190
|
+
// harness — not peaks-loop — decides this field's value. Guarding only
|
|
191
|
+
// `undefined` let `null` through to `block.context_window_size` and threw
|
|
192
|
+
// straight out of the statusline render: measured 2026-09-14 against the real
|
|
193
|
+
// CLI, `PEAKS_STATUSLINE_STDIN='{"context_window":null}' … statusline` gave
|
|
194
|
+
// exit 1, empty stdout, `UNHANDLED_ERROR` — no line rendered at all, for as
|
|
195
|
+
// long as the harness sends that shape. `block?.used_percentage` below was
|
|
196
|
+
// already null-safe; this pair was the hole.
|
|
197
|
+
const noBlock = block === null || block === undefined;
|
|
198
|
+
const modelWindowTokens = noBlock ? null : finiteNumber(block.context_window_size);
|
|
199
|
+
const usageTokens = noBlock ? null : sumUsageTokens(block.current_usage);
|
|
200
|
+
const tokensPerWindow = usageTokens !== null && modelWindowTokens !== null && modelWindowTokens > 0
|
|
201
|
+
? usageTokens / modelWindowTokens
|
|
202
|
+
: null;
|
|
203
|
+
const percentage = resolvePercentage(block?.used_percentage, tokensPerWindow);
|
|
204
|
+
const sessionId = input.stdin.session_id;
|
|
205
|
+
return {
|
|
206
|
+
schemaVersion: WITNESS_SCHEMA_VERSION,
|
|
207
|
+
capturedAt: new Date(input.nowMs).toISOString(),
|
|
208
|
+
usedPercentage: percentage.value,
|
|
209
|
+
usedPercentageRaw: percentage.raw,
|
|
210
|
+
usedPercentageUnit: percentage.unit,
|
|
211
|
+
modelWindowTokens,
|
|
212
|
+
usageTokens,
|
|
213
|
+
outerSessionId: typeof sessionId === 'string' && sessionId.length > 0 ? sessionId : null
|
|
214
|
+
};
|
|
215
|
+
}
|
|
216
|
+
export function harnessWitnessPath(projectRoot, sessionId) {
|
|
217
|
+
return `${getSessionDir(projectRoot, sessionId)}/${HARNESS_CONTEXT_WITNESS_FILE}`;
|
|
218
|
+
}
|
|
219
|
+
/**
|
|
220
|
+
* Write this render's record. Returns whether a file was written.
|
|
221
|
+
*
|
|
222
|
+
* This is the ONLY side effect on the statusline render path. It is bounded
|
|
223
|
+
* (one small file, inside peaks-loop's own session directory), it cannot
|
|
224
|
+
* influence any decision peaks-loop makes (nothing except the diagnostic in
|
|
225
|
+
* `peaks code context-now` reads it), and it never throws — a statusline that
|
|
226
|
+
* fails to render because an observability file could not be written would be
|
|
227
|
+
* a worse failure than a missing observation. A failed write is not swallowed:
|
|
228
|
+
* the next probe reports `absent`, which is the visible symptom.
|
|
229
|
+
*
|
|
230
|
+
* No `mkdir`: the session directory is created by the session layer, and a
|
|
231
|
+
* statusline render is the wrong place to be creating directories. Its absence
|
|
232
|
+
* is one of the honest reasons the witness can be missing.
|
|
233
|
+
*
|
|
234
|
+
* Written via temp + rename, not in place. A reader runs in ANOTHER process
|
|
235
|
+
* (the probe) and treats unreadable JSON as "no witness" — so a torn read
|
|
236
|
+
* would not be an error, it would be a SILENT loss of the comparison, which is
|
|
237
|
+
* the failure mode this whole slice exists to remove. The rename makes the
|
|
238
|
+
* partial state unobservable. Same shape as `atomicWriteJson` in
|
|
239
|
+
* statusline-settings-service.ts.
|
|
240
|
+
*
|
|
241
|
+
* The rename is not always available: on Windows it throws EPERM while another
|
|
242
|
+
* process holds the target open, which is exactly the reader this guard is
|
|
243
|
+
* written for. Dropping the sample then would be the silent loss the temp
|
|
244
|
+
* rename was introduced to remove, so the write falls back to in-place. The
|
|
245
|
+
* fallback gives up atomicity for that one write — the reader may see a
|
|
246
|
+
* half-written file and call it "no witness" — which is a narrower failure
|
|
247
|
+
* than never recording the sample at all, and the temp path stays the normal
|
|
248
|
+
* one.
|
|
249
|
+
*
|
|
250
|
+
* A LOWER-INFORMATION RECORD DOES NOT REPLACE A HIGHER-INFORMATION ONE (repair
|
|
251
|
+
* cycle 3). The render that carries no readable percentage is still recorded
|
|
252
|
+
* when there is nothing better on disk — that is what tells "rendered, nothing
|
|
253
|
+
* usable" apart from "never rendered" (§17-B) — but it does NOT overwrite a
|
|
254
|
+
* record whose percentage IS readable. Overwriting silences a real comparison:
|
|
255
|
+
* measured 2026-09-14, a witness giving a real 3.3% window gap reported
|
|
256
|
+
* `disagree` with its sentence, and one render whose payload lacked
|
|
257
|
+
* `context_window` turned the same comparison into `unverifiable` with the
|
|
258
|
+
* sentence suppressed. A render that says less must not delete a sample that
|
|
259
|
+
* says more.
|
|
260
|
+
*
|
|
261
|
+
* AND THE PARSE IS INSIDE A TRY (repair cycle 3). The module's contract at the
|
|
262
|
+
* top of this comment — "it never throws" — was true only by inspection: the
|
|
263
|
+
* `parseHarnessWitness` call sat above the only `try`, and a payload shape the
|
|
264
|
+
* guards did not anticipate escaped the function and killed the render (see
|
|
265
|
+
* `parseHarnessWitness` for the measured case). A nested `try` rather than a
|
|
266
|
+
* wider one, because the outer catch says something different: it is the
|
|
267
|
+
* temp+rename FALLBACK, and a parse failure has no JSON to fall back TO.
|
|
268
|
+
* A payload this module cannot read is a missing observation, which is the
|
|
269
|
+
* outcome the contract already prefers.
|
|
270
|
+
*/
|
|
271
|
+
export function writeHarnessWitness(input) {
|
|
272
|
+
if (input.projectRoot === null || input.sessionId === null)
|
|
273
|
+
return false;
|
|
274
|
+
let witness;
|
|
275
|
+
try {
|
|
276
|
+
witness = parseHarnessWitness({ stdin: input.stdin, nowMs: input.nowMs });
|
|
277
|
+
}
|
|
278
|
+
catch {
|
|
279
|
+
return false;
|
|
280
|
+
}
|
|
281
|
+
if (witness === null)
|
|
282
|
+
return false;
|
|
283
|
+
const path = harnessWitnessPath(input.projectRoot, input.sessionId);
|
|
284
|
+
if (!existsSync(dirname(path)))
|
|
285
|
+
return false;
|
|
286
|
+
if (witness.usedPercentage === null) {
|
|
287
|
+
const existing = readHarnessWitness({ projectRoot: input.projectRoot, sessionId: input.sessionId });
|
|
288
|
+
if (existing.kind === 'valid' && existing.witness.usedPercentage !== null)
|
|
289
|
+
return false;
|
|
290
|
+
}
|
|
291
|
+
const json = `${JSON.stringify(witness, null, 2)}\n`;
|
|
292
|
+
const tempPath = `${path}.tmp-${process.pid}`;
|
|
293
|
+
try {
|
|
294
|
+
writeFileSync(tempPath, json, 'utf8');
|
|
295
|
+
renameSync(tempPath, path);
|
|
296
|
+
return true;
|
|
297
|
+
}
|
|
298
|
+
catch {
|
|
299
|
+
rmSync(tempPath, { force: true });
|
|
300
|
+
try {
|
|
301
|
+
writeFileSync(path, json, 'utf8');
|
|
302
|
+
return true;
|
|
303
|
+
}
|
|
304
|
+
catch {
|
|
305
|
+
return false;
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
/**
|
|
310
|
+
* Read the record for a session. Anything that exists but cannot be used as a
|
|
311
|
+
* record reads as `invalid` — never as `missing`, which would erase the one
|
|
312
|
+
* fact the file's existence carries: a render happened here.
|
|
313
|
+
*/
|
|
314
|
+
export function readHarnessWitness(input) {
|
|
315
|
+
const path = harnessWitnessPath(input.projectRoot, input.sessionId);
|
|
316
|
+
if (!existsSync(path))
|
|
317
|
+
return { kind: 'missing' };
|
|
318
|
+
try {
|
|
319
|
+
const parsed = JSON.parse(readFileSync(path, 'utf8'));
|
|
320
|
+
if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed))
|
|
321
|
+
return { kind: 'invalid' };
|
|
322
|
+
const record = parsed;
|
|
323
|
+
const unit = record['usedPercentageUnit'];
|
|
324
|
+
return {
|
|
325
|
+
kind: 'valid',
|
|
326
|
+
witness: {
|
|
327
|
+
schemaVersion: finiteNumber(record['schemaVersion']) ?? WITNESS_SCHEMA_VERSION,
|
|
328
|
+
capturedAt: typeof record['capturedAt'] === 'string' ? record['capturedAt'] : '',
|
|
329
|
+
usedPercentage: finiteNumber(record['usedPercentage']),
|
|
330
|
+
usedPercentageRaw: finiteNumber(record['usedPercentageRaw']),
|
|
331
|
+
usedPercentageUnit: unit === 'fraction' || unit === 'percent' || unit === 'unestablished' ? unit : null,
|
|
332
|
+
modelWindowTokens: finiteNumber(record['modelWindowTokens']),
|
|
333
|
+
usageTokens: finiteNumber(record['usageTokens']),
|
|
334
|
+
outerSessionId: typeof record['outerSessionId'] === 'string' ? record['outerSessionId'] : null
|
|
335
|
+
}
|
|
336
|
+
};
|
|
337
|
+
}
|
|
338
|
+
catch {
|
|
339
|
+
// Unreadable (permissions, a directory where the file belongs, a torn
|
|
340
|
+
// read) and unparseable both land here. They are one fact for the caller:
|
|
341
|
+
// the file is there and no record can be taken from it.
|
|
342
|
+
return { kind: 'invalid' };
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
/**
|
|
346
|
+
* The budget, in tokens: what a difference between the two ratios can be
|
|
347
|
+
* explained by WITHOUT the two denominators being different, once the
|
|
348
|
+
* sampling skew has been taken out (see `compareHarnessWitness`).
|
|
349
|
+
*
|
|
350
|
+
* rounding 0.005 x window — the harness's percentage is rounded
|
|
351
|
+
* numerator 0.0017 x usedTokens — measured disagreement of the two sums
|
|
352
|
+
*
|
|
353
|
+
* Sample skew is deliberately NOT a term here. It is not a budget at all: it
|
|
354
|
+
* is MEASURED per sample from the two token counts and SUBTRACTED from the
|
|
355
|
+
* deviation, because under equal denominators the token difference and the
|
|
356
|
+
* ratio difference are the same number — see `compareHarnessWitness`.
|
|
357
|
+
*/
|
|
358
|
+
export function witnessToleranceTokens(input) {
|
|
359
|
+
return WITNESS_PERCENT_ROUNDING_FRACTION * input.windowTokens
|
|
360
|
+
+ WITNESS_NUMERATOR_FRACTION * input.usedTokens;
|
|
361
|
+
}
|
|
362
|
+
/**
|
|
363
|
+
* Whether this sample can support an `agree`.
|
|
364
|
+
*
|
|
365
|
+
* `agree` is a claim that the two denominators MATCH. A window difference `d`
|
|
366
|
+
* shows up in the residual as `-d x harnessPct/(1+d)` — it carries the
|
|
367
|
+
* WITNESS's ratio, not peaks-loop's, because the residual is a difference of
|
|
368
|
+
* two ratios and therefore scales with how much of the harness's window was in
|
|
369
|
+
* use when the witness was captured. The observation also carries an error of
|
|
370
|
+
* up to one budget (`tolerance`): the harness's percentage is rounded, and the
|
|
371
|
+
* two sides do not sum exactly the same token quantities. So the smallest
|
|
372
|
+
* difference this guard claims to detect has to leave a residual bigger than
|
|
373
|
+
* twice the budget, or it hides inside the budget's own uncertainty and the
|
|
374
|
+
* strongest answer goes to a real gap.
|
|
375
|
+
*
|
|
376
|
+
* Gating on `peaksRatio` instead — which is what this guard did until repair
|
|
377
|
+
* cycle 2 — is blind to exactly that: it is a function of a number the signal
|
|
378
|
+
* does not contain. Measured (2026-09-13): with the budget and the gate as they
|
|
379
|
+
* were, a real 3.3% gap was reported `agree` at every ratio below ~0.61, 2,436
|
|
380
|
+
* of 95,475 swept samples across ratios 0.18-0.99 reported `agree` over a real
|
|
381
|
+
* gap of 3-8%, and `unverifiable` was unreachable above 0.18 of the window —
|
|
382
|
+
* the one answer that was honest there could not be produced there.
|
|
383
|
+
*/
|
|
384
|
+
function sampleSupportsAgreement(harnessPct, tolerance) {
|
|
385
|
+
const smallestGapSignal = (MIN_DETECTABLE_WINDOW_DIFFERENCE * Math.abs(harnessPct)) / (1 + MIN_DETECTABLE_WINDOW_DIFFERENCE);
|
|
386
|
+
return smallestGapSignal > 2 * tolerance;
|
|
387
|
+
}
|
|
388
|
+
/** Which of the three `absent` states applies, in the reader's own words. */
|
|
389
|
+
function absentReason(cause) {
|
|
390
|
+
if (cause === 'session-dir-missing') {
|
|
391
|
+
return 'this session has no runtime directory yet, so nothing has been captured for it';
|
|
392
|
+
}
|
|
393
|
+
if (cause === 'unreadable') {
|
|
394
|
+
return 'a witness file exists for this session but could not be read as a record — the statusline DID '
|
|
395
|
+
+ 'render here, and its record is unreadable, malformed, or not the shape this reader expects';
|
|
396
|
+
}
|
|
397
|
+
return 'no render has been recorded for this session — the statusline has not rendered through peaks-loop here';
|
|
398
|
+
}
|
|
399
|
+
/** Why a recorded render cannot be compared, in the reader's own words. */
|
|
400
|
+
function unusableReason(witness) {
|
|
401
|
+
if (witness.usedPercentageRaw === null) {
|
|
402
|
+
return 'the payload carried no `used_percentage`, so this render recorded nothing to compare';
|
|
403
|
+
}
|
|
404
|
+
if (witness.usedPercentageUnit === 'unestablished') {
|
|
405
|
+
return `the payload reported used_percentage ${witness.usedPercentageRaw} and carried no token snapshot `
|
|
406
|
+
+ 'to settle whether that is a fraction or a percent; the raw value is recorded, and no comparison is '
|
|
407
|
+
+ 'made from it';
|
|
408
|
+
}
|
|
409
|
+
return `the payload reported used_percentage ${witness.usedPercentageRaw}, which is on neither scale `
|
|
410
|
+
+ '(a 0..1 fraction or a 0..100 percent); the raw value is recorded, and no comparison is made from it';
|
|
411
|
+
}
|
|
412
|
+
/**
|
|
413
|
+
* Compare peaks-loop's ratio with the harness's own percentage.
|
|
414
|
+
*
|
|
415
|
+
* Two identifiers must match before any comparison is allowed:
|
|
416
|
+
* 1. the SESSION — a witness written by another harness session on the same
|
|
417
|
+
* project is not evidence about this one. Lenient in the same direction as
|
|
418
|
+
* the compact-event attribution: refused only when BOTH ids resolve and
|
|
419
|
+
* differ, because a guard that turns a missing field into a permanent
|
|
420
|
+
* "cannot tell" is the failure this whole slice exists to remove.
|
|
421
|
+
* 2. the MOMENT — a witness is a snapshot, and the harness documents that its
|
|
422
|
+
* percentage depends on when it was calculated. The token counts are what
|
|
423
|
+
* says whether the two numbers came from the same API response.
|
|
424
|
+
*
|
|
425
|
+
* THE MOMENT IS REMOVED, NOT BUDGETED (repair cycle 1). If the two ratios
|
|
426
|
+
* share a denominator W, the harness's count and peaks-loop's count differ by
|
|
427
|
+
* exactly the sampling skew, so
|
|
428
|
+
*
|
|
429
|
+
* peaksRatio - harnessPct == (peaksTokens - witnessTokens) / W
|
|
430
|
+
*
|
|
431
|
+
* holds identically — it is not a tolerance to be granted, it is an equality
|
|
432
|
+
* to be tested. Budgeting the skew instead (`tolerance += |skew|`) made the
|
|
433
|
+
* budget grow by `x` while the deviation grew by `x(1+g)`, so a real window
|
|
434
|
+
* difference `g` was cancelled for every skew large enough to absorb it: a
|
|
435
|
+
* measured 3.3% denominator gap read `agree` for skew in [12,400, 22,100]
|
|
436
|
+
* tokens, and gaps up to 5.3% never surfaced at all. Subtracting the skew from
|
|
437
|
+
* the deviation and testing the remainder against a budget that contains only
|
|
438
|
+
* rounding and numerator disagreement makes the test invariant to skew by
|
|
439
|
+
* construction, and leaves `-g x harnessPct/(1+g)` — the window difference
|
|
440
|
+
* itself — as the only thing the residual can be.
|
|
441
|
+
*
|
|
442
|
+
* AND THE RESIDUAL'S SIZE IS THE WITNESS'S, NOT PEAKS-LOOP'S (repair cycle 2).
|
|
443
|
+
* `-g x harnessPct/(1+g)` carries the witness's ratio, so a witness captured
|
|
444
|
+
* while the session was small cannot show a window difference that a bigger
|
|
445
|
+
* witness would. The three answers therefore do not share one gate: a residual
|
|
446
|
+
* past the budget is a `disagree` at any witness size, while `agree` needs the
|
|
447
|
+
* sample to be sharp enough to support it (`sampleSupportsAgreement`).
|
|
448
|
+
*/
|
|
449
|
+
export function compareHarnessWitness(input) {
|
|
450
|
+
const base = {
|
|
451
|
+
peaksRatio: input.peaksRatio,
|
|
452
|
+
peaksTokens: input.peaksTokens,
|
|
453
|
+
witnessTokens: input.witness?.usageTokens ?? null,
|
|
454
|
+
witnessedAt: input.witness?.capturedAt ?? null,
|
|
455
|
+
witnessRawPercentage: input.witness?.usedPercentageRaw ?? null,
|
|
456
|
+
witnessPercentageUnit: input.witness?.usedPercentageUnit ?? null
|
|
457
|
+
};
|
|
458
|
+
if (input.witness === null) {
|
|
459
|
+
return {
|
|
460
|
+
...base,
|
|
461
|
+
verdict: 'absent',
|
|
462
|
+
reason: absentReason(input.absentCause ?? 'not-rendered'),
|
|
463
|
+
harnessPct: null,
|
|
464
|
+
deviation: null,
|
|
465
|
+
residual: null,
|
|
466
|
+
tolerance: null
|
|
467
|
+
};
|
|
468
|
+
}
|
|
469
|
+
const witness = input.witness;
|
|
470
|
+
// The version is the record's own statement of which unit rule normalised
|
|
471
|
+
// `usedPercentage`. Reading a record written under an older rule as if it
|
|
472
|
+
// were this one takes an old value under a new rule: a v1 file that recorded
|
|
473
|
+
// the payload's bare `1` as a fraction holds `usedPercentage: 1` (100%), and
|
|
474
|
+
// reading that as the current shape reports a confident disagreement and
|
|
475
|
+
// blames the two denominators for a unit misfire. Those files are already on
|
|
476
|
+
// disk for anyone who ran an earlier revision, and the next render replaces
|
|
477
|
+
// one — so the honest answer here is to abstain, not to migrate a value whose
|
|
478
|
+
// raw form was never recorded.
|
|
479
|
+
if (witness.schemaVersion !== WITNESS_SCHEMA_VERSION) {
|
|
480
|
+
return {
|
|
481
|
+
...base,
|
|
482
|
+
verdict: 'unverifiable',
|
|
483
|
+
reason: `this witness was recorded under schema v${witness.schemaVersion} and this revision reads `
|
|
484
|
+
+ `v${WITNESS_SCHEMA_VERSION}, which do not agree on what the recorded percentage means; it is not `
|
|
485
|
+
+ 'compared, and the next render replaces it',
|
|
486
|
+
harnessPct: null,
|
|
487
|
+
deviation: null,
|
|
488
|
+
residual: null,
|
|
489
|
+
tolerance: null
|
|
490
|
+
};
|
|
491
|
+
}
|
|
492
|
+
if (witness.outerSessionId !== null &&
|
|
493
|
+
input.outerSessionId !== null &&
|
|
494
|
+
witness.outerSessionId !== input.outerSessionId) {
|
|
495
|
+
return {
|
|
496
|
+
...base,
|
|
497
|
+
verdict: 'foreign-session',
|
|
498
|
+
reason: 'the witness names a different harness session, so it says nothing about this one',
|
|
499
|
+
harnessPct: witness.usedPercentage,
|
|
500
|
+
deviation: witness.usedPercentage === null ? null : input.peaksRatio - witness.usedPercentage,
|
|
501
|
+
residual: null,
|
|
502
|
+
tolerance: null
|
|
503
|
+
};
|
|
504
|
+
}
|
|
505
|
+
const harnessPct = witness.usedPercentage;
|
|
506
|
+
const windowTokens = input.peaksWindowTokens;
|
|
507
|
+
const peaksTokens = input.peaksTokens;
|
|
508
|
+
const witnessTokens = witness.usageTokens;
|
|
509
|
+
if (harnessPct === null) {
|
|
510
|
+
return {
|
|
511
|
+
...base,
|
|
512
|
+
verdict: 'unverifiable',
|
|
513
|
+
reason: unusableReason(witness),
|
|
514
|
+
harnessPct: null,
|
|
515
|
+
deviation: null,
|
|
516
|
+
residual: null,
|
|
517
|
+
tolerance: null
|
|
518
|
+
};
|
|
519
|
+
}
|
|
520
|
+
if (windowTokens === null || windowTokens <= 0 || peaksTokens === null || witnessTokens === null) {
|
|
521
|
+
return {
|
|
522
|
+
...base,
|
|
523
|
+
verdict: 'unverifiable',
|
|
524
|
+
reason: 'the probe and the witness do not both carry a token snapshot, so the two numbers cannot be shown to describe the same moment',
|
|
525
|
+
harnessPct,
|
|
526
|
+
deviation: input.peaksRatio - harnessPct,
|
|
527
|
+
residual: null,
|
|
528
|
+
tolerance: null
|
|
529
|
+
};
|
|
530
|
+
}
|
|
531
|
+
const deviation = input.peaksRatio - harnessPct;
|
|
532
|
+
// The skew is measured here, not assumed, and it is subtracted rather than
|
|
533
|
+
// granted: see the doc comment above. Signed, so a witness captured before
|
|
534
|
+
// or after the probe is corrected in the direction it is actually off.
|
|
535
|
+
const skewRatio = (peaksTokens - witnessTokens) / windowTokens;
|
|
536
|
+
const residual = deviation - skewRatio;
|
|
537
|
+
const tolerance = witnessToleranceTokens({ windowTokens, usedTokens: peaksTokens }) / windowTokens;
|
|
538
|
+
// The two answers are not symmetric, so they do not share one gate.
|
|
539
|
+
// `disagree` only claims that SOME difference exists, and a residual past the
|
|
540
|
+
// budget is exactly that claim — sound at any ratio, any witness age.
|
|
541
|
+
if (Math.abs(residual) > tolerance) {
|
|
542
|
+
return { ...base, verdict: 'disagree', reason: null, harnessPct, deviation, residual, tolerance };
|
|
543
|
+
}
|
|
544
|
+
// `agree` claims there is no difference, which needs the sample to be sharp
|
|
545
|
+
// enough to support it — see `sampleSupportsAgreement`.
|
|
546
|
+
if (!sampleSupportsAgreement(harnessPct, tolerance)) {
|
|
547
|
+
return {
|
|
548
|
+
...base,
|
|
549
|
+
verdict: 'unverifiable',
|
|
550
|
+
reason: 'the witness covers too small a share of the harness window for this sample to separate a real window difference from the budget’s own uncertainty',
|
|
551
|
+
harnessPct,
|
|
552
|
+
deviation,
|
|
553
|
+
residual,
|
|
554
|
+
tolerance
|
|
555
|
+
};
|
|
556
|
+
}
|
|
557
|
+
return { ...base, verdict: 'agree', reason: null, harnessPct, deviation, residual, tolerance };
|
|
558
|
+
}
|
|
559
|
+
/**
|
|
560
|
+
* One-way sentence for a disagreeing witness. Advising, never asking: an
|
|
561
|
+
* auto-compact observation must never become an `AskUserQuestion` (see
|
|
562
|
+
* `.peaks/memory/auto-compact-threshold-policy.md`).
|
|
563
|
+
*
|
|
564
|
+
* The sentence states the quantity that decided it (the residual, after the
|
|
565
|
+
* measured skew was removed) and the raw harness value with the reading that
|
|
566
|
+
* was taken from it. Both are there so a reader can tell a real denominator
|
|
567
|
+
* difference from a unit misread without going back to the file.
|
|
568
|
+
*/
|
|
569
|
+
export function describeHarnessWitness(comparison) {
|
|
570
|
+
if (comparison.verdict !== 'disagree')
|
|
571
|
+
return null;
|
|
572
|
+
const pct = (value) => `${(value * 100).toFixed(1)}%`;
|
|
573
|
+
const skew = comparison.deviation === null || comparison.residual === null
|
|
574
|
+
? null
|
|
575
|
+
: comparison.deviation - comparison.residual;
|
|
576
|
+
const readAs = comparison.witnessPercentageUnit === null
|
|
577
|
+
? ''
|
|
578
|
+
: ` (raw ${comparison.witnessRawPercentage}, read as a ${comparison.witnessPercentageUnit})`;
|
|
579
|
+
return `Harness context witness disagrees: the harness reports ${pct(comparison.harnessPct ?? 0)} used${readAs}, `
|
|
580
|
+
+ `peaks-loop computes ${pct(comparison.peaksRatio)} (deviation ${pct(Math.abs(comparison.deviation ?? 0))}`
|
|
581
|
+
+ `${skew === null ? '' : `, of which the measured sampling skew explains ${pct(Math.abs(skew))}`}, `
|
|
582
|
+
+ `leaving ${pct(Math.abs(comparison.residual ?? 0))} against a budget of ${pct(comparison.tolerance ?? 0)}). `
|
|
583
|
+
+ 'The two ratios do not share a denominator: '
|
|
584
|
+
+ 'peaks-loop\'s configured auto-compact window and the harness\'s effective auto-compact window are '
|
|
585
|
+
+ 'different numbers. Nothing is blocked.';
|
|
586
|
+
}
|
|
587
|
+
/** Convenience for the CLI: read + compare in one call. */
|
|
588
|
+
export function readAndCompareHarnessWitness(input) {
|
|
589
|
+
const read = readHarnessWitness({ projectRoot: input.projectRoot, sessionId: input.sessionId });
|
|
590
|
+
return compareHarnessWitness({
|
|
591
|
+
witness: read.kind === 'valid' ? read.witness : null,
|
|
592
|
+
peaksRatio: input.peaksRatio,
|
|
593
|
+
peaksTokens: input.peaksTokens,
|
|
594
|
+
peaksWindowTokens: input.peaksWindowTokens,
|
|
595
|
+
outerSessionId: input.outerSessionId,
|
|
596
|
+
// Only the reader knows the session directory, so only the reader can tell
|
|
597
|
+
// the two no-file states apart. The third cause is not a directory
|
|
598
|
+
// question — `invalid` IS the fact that a render happened — so it is
|
|
599
|
+
// answered by the read and never falls through to the ternary.
|
|
600
|
+
absentCause: read.kind === 'invalid'
|
|
601
|
+
? 'unreadable'
|
|
602
|
+
: existsSync(getSessionDir(input.projectRoot, input.sessionId))
|
|
603
|
+
? 'not-rendered'
|
|
604
|
+
: 'session-dir-missing'
|
|
605
|
+
});
|
|
606
|
+
}
|