peaks-loop 4.0.46 → 4.0.48

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (207) hide show
  1. package/CHANGELOG.md +54 -0
  2. package/README-en.md +1 -1
  3. package/README.md +1 -1
  4. package/agents/karpathy-reviewer.md +11 -10
  5. package/dist/cli/cli-helpers.d.ts +34 -0
  6. package/dist/cli/cli-helpers.js +57 -0
  7. package/dist/cli/commands/code-job-shape-commands.js +8 -0
  8. package/dist/cli/commands/code-runtime-commands.d.ts +22 -0
  9. package/dist/cli/commands/code-runtime-commands.js +139 -16
  10. package/dist/cli/commands/compact-command.js +241 -1
  11. package/dist/cli/commands/config-commands.js +15 -9
  12. package/dist/cli/commands/container-commands.js +3 -3
  13. package/dist/cli/commands/core/skill-command.js +45 -10
  14. package/dist/cli/commands/cron-commands.js +2 -1
  15. package/dist/cli/commands/dashboard-long-run.js +6 -0
  16. package/dist/cli/commands/dispatch-commands.js +11 -1
  17. package/dist/cli/commands/doctor/invoke-from-code.js +6 -0
  18. package/dist/cli/commands/e2e-verify.js +3 -3
  19. package/dist/cli/commands/governance-classify-contract-commands.js +1 -0
  20. package/dist/cli/commands/hooks-commands.js +14 -5
  21. package/dist/cli/commands/job-commands.js +8 -0
  22. package/dist/cli/commands/loop-commands.js +1 -0
  23. package/dist/cli/commands/loop-eval-commands.js +15 -0
  24. package/dist/cli/commands/perf-audit-commands.js +2 -0
  25. package/dist/cli/commands/playwright-commands.js +14 -1
  26. package/dist/cli/commands/prd-commands.js +1 -1
  27. package/dist/cli/commands/qa-commands.js +22 -0
  28. package/dist/cli/commands/reinject-command.d.ts +72 -0
  29. package/dist/cli/commands/reinject-command.js +174 -0
  30. package/dist/cli/commands/request-commands.js +14 -3
  31. package/dist/cli/commands/scan-commands.js +1 -1
  32. package/dist/cli/commands/security-audit-commands.js +2 -0
  33. package/dist/cli/commands/shadcn-commands.js +1 -0
  34. package/dist/cli/commands/slice-integrate-commands.js +5 -0
  35. package/dist/cli/commands/statusline-commands.js +44 -4
  36. package/dist/cli/commands/sub-agent/detached.d.ts +14 -1
  37. package/dist/cli/commands/sub-agent/detached.js +47 -22
  38. package/dist/cli/commands/sub-agent-shutdown-commands.js +11 -0
  39. package/dist/cli/commands/test-commands.js +2 -1
  40. package/dist/cli/commands/verdict-aggregate-command.js +95 -13
  41. package/dist/cli/commands/vm-commands.js +7 -7
  42. package/dist/cli/commands/workflow-commands.js +1 -1
  43. package/dist/cli/commands/workspace/init-command.js +24 -2
  44. package/dist/cli/commands/worktree-lease-commands.js +4 -4
  45. package/dist/cli/index.js +10 -3
  46. package/dist/cli/program.js +5 -0
  47. package/dist/hooks/pre-tool-use-sub-agent.js +1 -1
  48. package/dist/services/adapter/adapter-registry.js +1 -1
  49. package/dist/services/artifacts/artifact-prerequisites.d.ts +38 -7
  50. package/dist/services/artifacts/artifact-prerequisites.js +130 -65
  51. package/dist/services/artifacts/artifact-service.js +1 -1
  52. package/dist/services/artifacts/request-artifact-service.d.ts +8 -0
  53. package/dist/services/artifacts/request-artifact-service.js +18 -8
  54. package/dist/services/artifacts/request-artifact-state-helpers.d.ts +57 -0
  55. package/dist/services/artifacts/request-artifact-state-helpers.js +91 -10
  56. package/dist/services/audit-independent/perf-audit-service.d.ts +9 -0
  57. package/dist/services/audit-independent/perf-audit-service.js +27 -5
  58. package/dist/services/audit-independent/security-audit-service.d.ts +12 -2
  59. package/dist/services/audit-independent/security-audit-service.js +28 -6
  60. package/dist/services/capability-guard-runner/contracts/J01.js +2 -1
  61. package/dist/services/capability-guard-runner/contracts/J02.js +3 -3
  62. package/dist/services/capability-guard-runner/contracts/J04.js +4 -2
  63. package/dist/services/capability-guard-runner/contracts/J07.js +2 -1
  64. package/dist/services/code/auto-compact-lifecycle.d.ts +130 -1
  65. package/dist/services/code/auto-compact-lifecycle.js +180 -4
  66. package/dist/services/code/auto-compact-orchestrator.d.ts +53 -9
  67. package/dist/services/code/auto-compact-orchestrator.js +166 -34
  68. package/dist/services/code/compact-event-settle.d.ts +122 -0
  69. package/dist/services/code/compact-event-settle.js +219 -0
  70. package/dist/services/code/orchestrator-can-do.d.ts +4 -2
  71. package/dist/services/code/orchestrator-can-do.js +37 -5
  72. package/dist/services/codegraph/codegraph-exclude-reconciler.js +2 -1
  73. package/dist/services/codegraph/codegraph-process-runner.js +3 -2
  74. package/dist/services/compact/request-transition-hook.js +5 -2
  75. package/dist/services/compact-history/compact-history-service.d.ts +75 -0
  76. package/dist/services/compact-history/compact-history-service.js +49 -0
  77. package/dist/services/config/config-restore.d.ts +12 -1
  78. package/dist/services/config/config-restore.js +35 -4
  79. package/dist/services/config/config-rollback.js +6 -1
  80. package/dist/services/config/config-safety.d.ts +52 -0
  81. package/dist/services/config/config-safety.js +75 -1
  82. package/dist/services/context/auto-compact-dispatcher.d.ts +7 -37
  83. package/dist/services/context/auto-compact-dispatcher.js +113 -40
  84. package/dist/services/context/auto-compact-reader.d.ts +68 -28
  85. package/dist/services/context/auto-compact-reader.js +155 -1
  86. package/dist/services/context/auto-compact-types.d.ts +89 -12
  87. package/dist/services/context/auto-compact-types.js +16 -32
  88. package/dist/services/context/harness-context-witness.d.ts +310 -0
  89. package/dist/services/context/harness-context-witness.js +606 -0
  90. package/dist/services/context/harness-window-config.d.ts +412 -0
  91. package/dist/services/context/harness-window-config.js +607 -0
  92. package/dist/services/context/main-session-monitor.d.ts +27 -0
  93. package/dist/services/context/main-session-monitor.js +32 -1
  94. package/dist/services/context/post-compact-reinjection.d.ts +221 -0
  95. package/dist/services/context/post-compact-reinjection.js +491 -0
  96. package/dist/services/dispatch/merge-back-runner.js +5 -5
  97. package/dist/services/dispatch/service-shutdown.js +3 -3
  98. package/dist/services/doc/doc-generator.js +2 -1
  99. package/dist/services/env/shell-probe.js +1 -1
  100. package/dist/services/evidence/evidence-generator.js +86 -49
  101. package/dist/services/final-review/final-review-service.d.ts +9 -0
  102. package/dist/services/final-review/final-review-service.js +36 -12
  103. package/dist/services/fuzzy-matching/fzf-pick-service.js +2 -0
  104. package/dist/services/hooks/auto-compact-hook-install.d.ts +10 -2
  105. package/dist/services/hooks/auto-compact-hook-install.js +8 -0
  106. package/dist/services/ide/adapters/claude-code-adapter.d.ts +107 -3
  107. package/dist/services/ide/adapters/claude-code-adapter.js +154 -7
  108. package/dist/services/ide/ide-registry.d.ts +31 -0
  109. package/dist/services/ide/ide-registry.js +35 -0
  110. package/dist/services/ide/ide-types.d.ts +59 -0
  111. package/dist/services/job/job-state-store.js +7 -0
  112. package/dist/services/lint/detect-eslint.js +2 -2
  113. package/dist/services/lint/eslint-runner.js +3 -1
  114. package/dist/services/loop/evaluator-dispatcher.js +2 -1
  115. package/dist/services/memory/project-memory-service/index/kind-dispatch.js +1 -1
  116. package/dist/services/memory/project-memory-service/store/paths.d.ts +9 -1
  117. package/dist/services/memory/project-memory-service/store/paths.js +15 -6
  118. package/dist/services/polyrepo/polyrepo-dispatcher.js +11 -0
  119. package/dist/services/prd/best-practice-auto-trigger.js +1 -0
  120. package/dist/services/prd/handoff-auto-regen.js +31 -27
  121. package/dist/services/prd/handoff-frontmatter.d.ts +44 -0
  122. package/dist/services/prd/handoff-frontmatter.js +75 -0
  123. package/dist/services/prd/handoff-service.d.ts +41 -2
  124. package/dist/services/prd/handoff-service.js +81 -8
  125. package/dist/services/prd/handoff-types.d.ts +3 -2
  126. package/dist/services/prd/handoff-types.js +3 -2
  127. package/dist/services/qa/qa-business-review-state.js +9 -0
  128. package/dist/services/release/version-precheck-service.d.ts +2 -1
  129. package/dist/services/release/version-precheck-service.js +82 -12
  130. package/dist/services/runtime/vendor-adapter.d.ts +29 -4
  131. package/dist/services/runtime/vendors/claude-code.js +1 -1
  132. package/dist/services/runtime/vendors/codex.js +1 -1
  133. package/dist/services/runtime/vendors/copilot.js +1 -1
  134. package/dist/services/sc/sc-service.js +1 -1
  135. package/dist/services/scan/diff-scope-service.js +2 -2
  136. package/dist/services/scan/file-size-scan.js +2 -2
  137. package/dist/services/scan/karpathy-service.js +2 -2
  138. package/dist/services/scan/orphan-service.js +2 -1
  139. package/dist/services/scan/type-sanity-service.js +2 -2
  140. package/dist/services/session/session-checkpoint-service.js +8 -0
  141. package/dist/services/skill/resume-detector.js +29 -11
  142. package/dist/services/skillhub/tar-runtime.js +1 -0
  143. package/dist/services/skills/hooks-codegate-superpowers.d.ts +14 -0
  144. package/dist/services/skills/hooks-codegate-superpowers.js +99 -3
  145. package/dist/services/skills/hooks-settings-service.d.ts +12 -0
  146. package/dist/services/skills/hooks-settings-service.js +91 -14
  147. package/dist/services/skills/session-start-hook-constants.d.ts +86 -0
  148. package/dist/services/skills/session-start-hook-constants.js +86 -0
  149. package/dist/services/skills/skill-presence-service.js +9 -0
  150. package/dist/services/skills/skill-statusline-service.d.ts +14 -0
  151. package/dist/services/slice/slice-check-service.js +31 -12
  152. package/dist/services/slice/slice-decompose-runners.js +2 -1
  153. package/dist/services/slice/slice-review-state.js +8 -0
  154. package/dist/services/upgrade/upgrade-service.js +1 -0
  155. package/dist/services/workflow/pipeline-verify-gate-support.d.ts +47 -10
  156. package/dist/services/workflow/pipeline-verify-gate-support.js +212 -93
  157. package/dist/services/workflow/pipeline-verify-service.js +24 -23
  158. package/dist/services/workflow/pipeline-verify-types.d.ts +10 -3
  159. package/dist/services/workflow/workflow-skip-service.js +2 -1
  160. package/dist/services/workspace/claude-settings-template.d.ts +56 -8
  161. package/dist/services/workspace/claude-settings-template.js +98 -20
  162. package/dist/services/workspace/migrate-service.js +1 -1
  163. package/dist/services/workspace/workspace-claude-settings-materializer.js +124 -9
  164. package/dist/services/workspace/workspace-service.js +8 -0
  165. package/dist/services/worktree/host-worktree-reconciler.js +1 -0
  166. package/dist/services/worktree/long-path-cleanup.js +3 -2
  167. package/dist/shared/process.js +1 -1
  168. package/package.json +6 -6
  169. package/scripts/install-skills.mjs +1 -0
  170. package/scripts/watch.mjs +3 -1
  171. package/skills/bee/peaks-perf-audit/SKILL.md +1 -1
  172. package/skills/bee/peaks-prd/SKILL.md +8 -6
  173. package/skills/bee/peaks-qa/SKILL.md +7 -7
  174. package/skills/bee/peaks-qa/references/qa-runbook.md +2 -2
  175. package/skills/bee/peaks-qa/references/qa-transition-gates.md +7 -7
  176. package/skills/bee/peaks-rd/SKILL.md +10 -8
  177. package/skills/bee/peaks-rd/references/artifact-per-request.md +2 -2
  178. package/skills/bee/peaks-rd/references/parallel-review-fanout.md +7 -5
  179. package/skills/bee/peaks-rd/references/rd-fanout-contracts.md +13 -13
  180. package/skills/bee/peaks-rd/references/rd-runbook.md +9 -5
  181. package/skills/bee/peaks-rd/references/rd-transition-gates.md +9 -7
  182. package/skills/bee/peaks-rd/references/writing-handoff-frontmatter.md +6 -6
  183. package/skills/bee/peaks-reviewer/SKILL.md +1 -1
  184. package/skills/bee/peaks-sc/SKILL.md +1 -1
  185. package/skills/bee/peaks-security-audit/SKILL.md +1 -1
  186. package/skills/bee/peaks-txt/SKILL.md +1 -1
  187. package/skills/bee/peaks-ui/SKILL.md +1 -1
  188. package/skills/peaks-audit/SKILL.md +1 -1
  189. package/skills/peaks-code/SKILL.md +3 -3
  190. package/skills/peaks-code/references/a2a-artifact-mapping.md +3 -3
  191. package/skills/peaks-code/references/local-artifact-workspace.md +1 -1
  192. package/skills/peaks-code/references/resume-detection.md +13 -7
  193. package/skills/peaks-code/references/runbook.md +3 -2
  194. package/skills/peaks-code/references/session-overload-signal-index.md +2 -1
  195. package/skills/peaks-code/references/sub-agent-dispatch.md +1 -1
  196. package/skills/peaks-code/references/workflow-gates-and-types.md +8 -6
  197. package/skills/peaks-content/SKILL.md +1 -1
  198. package/skills/peaks-doctor/SKILL.md +1 -1
  199. package/skills/peaks-final-review/SKILL.md +1 -1
  200. package/skills/peaks-ide/SKILL.md +1 -1
  201. package/skills/peaks-issue-fix-orchestrator/SKILL.md +1 -1
  202. package/skills/peaks-resume/SKILL.md +1 -1
  203. package/skills/peaks-slice-decompose/SKILL.md +1 -1
  204. package/skills/peaks-solo/SKILL.md +1 -1
  205. package/skills/peaks-sop/SKILL.md +1 -1
  206. package/skills/peaks-status/SKILL.md +1 -1
  207. package/skills/peaks-test/SKILL.md +1 -1
@@ -0,0 +1,606 @@
1
+ /**
2
+ * Harness context witness — peaks-loop's second scale, captured from the
3
+ * harness instead of re-derived by peaks-loop.
4
+ *
5
+ * The harness pipes a JSON session payload on the statusline command's stdin,
6
+ * and that payload is the ONLY supported programming channel that carries the
7
+ * harness's OWN context numbers (`context_window.used_percentage` and
8
+ * `context_window.context_window_size`). peaks-loop's own ratio is derived
9
+ * separately (transcript estimate) and divides by a window peaks-loop writes
10
+ * into the harness's settings. Two derivations, two authors — and until now
11
+ * nothing compared them. This module is the comparison.
12
+ *
13
+ * WHAT IS NOT COMPARED, and why — read before changing anything here:
14
+ * `context_window_size` is the MODEL window. peaks-loop's denominator is the
15
+ * AUTO-COMPACT window it configures. They are semantically different objects
16
+ * that happen to be equal on the machine this slice was written on. The
17
+ * comparison is therefore `peaks ratio` vs `used_percentage` — two fractions
18
+ * "used / some window" — and NEVER a comparison of the two windows as if they
19
+ * were one field. A guard that compared the windows would report agreement on
20
+ * the one machine where the two numbers coincide and would have no idea why.
21
+ *
22
+ * `context_window_size` IS read as the denominator of the harness's OWN
23
+ * `used_percentage`, to recover the unit of a payload that does not state one
24
+ * (`readPercentageUnit`). That is reading the harness's field against the
25
+ * harness's own window; it is not a denominator for the comparison above.
26
+ *
27
+ * WHY A RESIDUAL, AND WHY THIS BUDGET (`witnessToleranceTokens` + the residual
28
+ * in `compareHarnessWitness`).
29
+ *
30
+ * WHY A THIRD ANSWER. This guard has three outcomes, not two: after "they
31
+ * agree" and "they disagree" there is "this sample cannot tell". A guard with
32
+ * only two answers cannot distinguish "correctly silent" from "broken", which
33
+ * is the failure mode this repository has a memory about. `unverifiable` is
34
+ * that third answer, and it is the honest one whenever the two numbers are not
35
+ * known to describe the same moment — and also whenever they describe the same
36
+ * moment but the sample is too coarse to separate a real window difference from
37
+ * the budget's own uncertainty, which is the state a small or stale witness
38
+ * puts this guard in.
39
+ *
40
+ * THE FILE IS A RENDER LEDGER, NOT ONLY A WITNESS. It is written on every
41
+ * render that receives a payload, including a render whose payload carried no
42
+ * usable percentage. That is what makes `absent` diagnosable: "no file" means
43
+ * nothing rendered here, and "a file with no percentage" means one did and its
44
+ * payload had nothing to read. See `.peaks/_runtime/<sid>/rd/tech-doc.md` §17.
45
+ *
46
+ * ...BUT A LOWER-INFORMATION RENDER DOES NOT OVERWRITE A HIGHER-INFORMATION
47
+ * ONE (repair cycle 3). The ledger rule is what makes the FIRST render of a
48
+ * session land; the no-downgrade rule is what stops a later contextless render
49
+ * from silencing a comparison the earlier one could still make. `capturedAt`
50
+ * keeps naming the render that produced the surviving record, so nothing on
51
+ * disk claims to be fresher than it is. See `writeHarnessWitness`.
52
+ *
53
+ * COST (the statusline runs this on every render): one small file write on the
54
+ * render path, one small file read on the probe path. No network, no directory
55
+ * walk, no transcript read, no new dependency.
56
+ */
57
+ import { existsSync, readFileSync, renameSync, rmSync, writeFileSync } from 'node:fs';
58
+ import { dirname } from 'node:path';
59
+ import { getSessionDir } from '../session/getSessionDir.js';
60
+ export const HARNESS_CONTEXT_WITNESS_FILE = 'harness-context-witness.json';
61
+ export const WITNESS_SCHEMA_VERSION = 2;
62
+ /**
63
+ * Contributor 1 of the budget: the harness reports a rounded percentage.
64
+ * "Pre-calculated percentage of context window used" is not documented as
65
+ * fractional, so the conservative reading is an integer percent, i.e. up to
66
+ * ±0.5 percentage points = ±0.005 of the window. If the real payload turns out
67
+ * to carry decimals this term shrinks tenfold and the guard gets sharper; the
68
+ * assumption is visible in the first witness file, where
69
+ * `usageTokens / modelWindowTokens` and `usedPercentage` must agree.
70
+ */
71
+ export const WITNESS_PERCENT_ROUNDING_FRACTION = 0.005;
72
+ /**
73
+ * Contributor 2: the two sides do not sum exactly the same token quantities.
74
+ * MEASURED, not guessed — on 2026-09-13 the harness's own pre-compact count
75
+ * (963,306 tokens) exceeded peaks-loop's transcript estimate (961,658) by
76
+ * 1,648 tokens = 0.171%. Used as a fraction of the tokens, not of the window.
77
+ */
78
+ export const WITNESS_NUMERATOR_FRACTION = 0.0017;
79
+ /**
80
+ * The smallest window difference this guard claims to detect, and therefore
81
+ * the yardstick for "is this sample sharp enough to answer at all".
82
+ *
83
+ * 3% is not arbitrary: the harness compacts a native-1M model at ~967,000 by
84
+ * default while peaks-loop writes the model ceiling, so 1,000,000 vs 967,000 —
85
+ * a 3.3% difference — is the smallest real-world disagreement between the two
86
+ * denominators today. Anything the guard reports as "agree" while its own
87
+ * budget is wider than that difference is a claim it cannot support.
88
+ *
89
+ * NOTE (repair cycle 2): the constant is a floor on the SAMPLE, and the sample
90
+ * it floors is the WITNESS's, not peaks-loop's — a gap of this size leaves a
91
+ * residual proportional to the witness's ratio, so a stale (or low) witness
92
+ * shrinks the signal while the budget's rounding term does not shrink with it.
93
+ * See `sampleSupportsAgreement`, which is where this constant is applied. At
94
+ * zero skew it was already correctly calibrated (measured onset 0.176 of the
95
+ * window against a predicted 0.177); the fault was that only zero skew was
96
+ * correctly calibrated.
97
+ */
98
+ export const MIN_DETECTABLE_WINDOW_DIFFERENCE = 0.03;
99
+ /**
100
+ * The largest raw value a FRACTION reading can carry. Above it the fraction
101
+ * reading is out of range, so the value has exactly ONE in-range reading —
102
+ * percent — and the unit needs no other evidence. (The repo's env / statusline
103
+ * readers use 1.5; that threshold calls values in (1, 1.5] fractions by fiat
104
+ * and then has to throw them away, which is why this reader does not reuse it.)
105
+ */
106
+ const FRACTION_MAX = 1;
107
+ /** Upper bound of a sane percentage scale; above this the payload is not a percent. */
108
+ const PERCENT_SCALE_MAX = 100;
109
+ /** The three prompt-side components peaks-loop's own `rawTokens` sums. */
110
+ const USAGE_TOKEN_KEYS = ['input_tokens', 'cache_read_input_tokens', 'cache_creation_input_tokens'];
111
+ function finiteNumber(value) {
112
+ return typeof value === 'number' && Number.isFinite(value) ? value : null;
113
+ }
114
+ /**
115
+ * Pick the reading of a raw `used_percentage` whose unit the payload does not
116
+ * state.
117
+ *
118
+ * Above `FRACTION_MAX` only the percent reading is in range, so the value
119
+ * decides. At or below it both readings are in range and the value cannot
120
+ * decide, so the witness's OWN token snapshot does: the two readings differ by
121
+ * exactly 100x, which puts their geometric midpoint at `raw / 10`.
122
+ * `unestablished` when the payload did not carry enough to compute that ratio:
123
+ * the value is in range on both scales and nothing in the payload says which
124
+ * one the harness meant.
125
+ *
126
+ * NOTE: this reads the harness's field against the harness's own window. It is
127
+ * not the comparison's denominator (see the module header).
128
+ */
129
+ function readPercentageUnit(raw, tokensPerWindow) {
130
+ if (raw > FRACTION_MAX)
131
+ return 'percent';
132
+ if (tokensPerWindow === null)
133
+ return 'unestablished';
134
+ return tokensPerWindow < raw / 10 ? 'percent' : 'fraction';
135
+ }
136
+ /**
137
+ * Normalise a harness percentage to 0..1, keeping the raw value either way.
138
+ *
139
+ * A value on neither scale (negative, or above 100) is refused rather than
140
+ * clamped — a witness nobody can interpret must not be compared — but it is
141
+ * still RETURNED with its raw value attached, because discarding it is what
142
+ * left a refused payload indistinguishable from a render that never happened.
143
+ *
144
+ * A value that is in range on BOTH scales, with no snapshot to settle which,
145
+ * is refused the same way — no reading is taken. Applying the repo's
146
+ * fraction-first convention here instead is what read a "used 1%" payload as
147
+ * 100% and turned a unit misfire into a confident disagreement that blamed the
148
+ * two denominators (see `unusableReason`).
149
+ */
150
+ function resolvePercentage(raw, tokensPerWindow) {
151
+ const value = finiteNumber(raw);
152
+ if (value === null || value < 0 || value > PERCENT_SCALE_MAX) {
153
+ return { value: null, raw: value, unit: null };
154
+ }
155
+ const unit = readPercentageUnit(value, tokensPerWindow);
156
+ if (unit === 'unestablished')
157
+ return { value: null, raw: value, unit };
158
+ return { value: unit === 'percent' ? value / PERCENT_SCALE_MAX : value, raw: value, unit };
159
+ }
160
+ /** Sum the prompt-side usage components, or `null` when none is present. */
161
+ function sumUsageTokens(currentUsage) {
162
+ if (currentUsage === null || typeof currentUsage !== 'object' || Array.isArray(currentUsage)) {
163
+ return null;
164
+ }
165
+ const record = currentUsage;
166
+ let sum = 0;
167
+ let seen = false;
168
+ for (const key of USAGE_TOKEN_KEYS) {
169
+ const value = finiteNumber(record[key]);
170
+ if (value !== null) {
171
+ sum += value;
172
+ seen = true;
173
+ }
174
+ }
175
+ return seen ? sum : null;
176
+ }
177
+ /**
178
+ * Read the harness's context block out of a parsed statusline stdin payload.
179
+ * Returns `null` only when there was no payload at all (a render on a TTY, or
180
+ * a manual `peaks statusline`) — in which case there is nothing to record and
181
+ * nothing to say. A payload that arrived but carried no usable percentage is
182
+ * recorded, not dropped.
183
+ */
184
+ export function parseHarnessWitness(input) {
185
+ if (input.stdin === null || input.stdin === undefined)
186
+ return null;
187
+ const block = input.stdin.context_window;
188
+ // `null` AS WELL AS `undefined` (repair cycle 3). The key being present with
189
+ // no value is how a JSON producer spells "there is no context block", and the
190
+ // harness — not peaks-loop — decides this field's value. Guarding only
191
+ // `undefined` let `null` through to `block.context_window_size` and threw
192
+ // straight out of the statusline render: measured 2026-09-14 against the real
193
+ // CLI, `PEAKS_STATUSLINE_STDIN='{"context_window":null}' … statusline` gave
194
+ // exit 1, empty stdout, `UNHANDLED_ERROR` — no line rendered at all, for as
195
+ // long as the harness sends that shape. `block?.used_percentage` below was
196
+ // already null-safe; this pair was the hole.
197
+ const noBlock = block === null || block === undefined;
198
+ const modelWindowTokens = noBlock ? null : finiteNumber(block.context_window_size);
199
+ const usageTokens = noBlock ? null : sumUsageTokens(block.current_usage);
200
+ const tokensPerWindow = usageTokens !== null && modelWindowTokens !== null && modelWindowTokens > 0
201
+ ? usageTokens / modelWindowTokens
202
+ : null;
203
+ const percentage = resolvePercentage(block?.used_percentage, tokensPerWindow);
204
+ const sessionId = input.stdin.session_id;
205
+ return {
206
+ schemaVersion: WITNESS_SCHEMA_VERSION,
207
+ capturedAt: new Date(input.nowMs).toISOString(),
208
+ usedPercentage: percentage.value,
209
+ usedPercentageRaw: percentage.raw,
210
+ usedPercentageUnit: percentage.unit,
211
+ modelWindowTokens,
212
+ usageTokens,
213
+ outerSessionId: typeof sessionId === 'string' && sessionId.length > 0 ? sessionId : null
214
+ };
215
+ }
216
+ export function harnessWitnessPath(projectRoot, sessionId) {
217
+ return `${getSessionDir(projectRoot, sessionId)}/${HARNESS_CONTEXT_WITNESS_FILE}`;
218
+ }
219
+ /**
220
+ * Write this render's record. Returns whether a file was written.
221
+ *
222
+ * This is the ONLY side effect on the statusline render path. It is bounded
223
+ * (one small file, inside peaks-loop's own session directory), it cannot
224
+ * influence any decision peaks-loop makes (nothing except the diagnostic in
225
+ * `peaks code context-now` reads it), and it never throws — a statusline that
226
+ * fails to render because an observability file could not be written would be
227
+ * a worse failure than a missing observation. A failed write is not swallowed:
228
+ * the next probe reports `absent`, which is the visible symptom.
229
+ *
230
+ * No `mkdir`: the session directory is created by the session layer, and a
231
+ * statusline render is the wrong place to be creating directories. Its absence
232
+ * is one of the honest reasons the witness can be missing.
233
+ *
234
+ * Written via temp + rename, not in place. A reader runs in ANOTHER process
235
+ * (the probe) and treats unreadable JSON as "no witness" — so a torn read
236
+ * would not be an error, it would be a SILENT loss of the comparison, which is
237
+ * the failure mode this whole slice exists to remove. The rename makes the
238
+ * partial state unobservable. Same shape as `atomicWriteJson` in
239
+ * statusline-settings-service.ts.
240
+ *
241
+ * The rename is not always available: on Windows it throws EPERM while another
242
+ * process holds the target open, which is exactly the reader this guard is
243
+ * written for. Dropping the sample then would be the silent loss the temp
244
+ * rename was introduced to remove, so the write falls back to in-place. The
245
+ * fallback gives up atomicity for that one write — the reader may see a
246
+ * half-written file and call it "no witness" — which is a narrower failure
247
+ * than never recording the sample at all, and the temp path stays the normal
248
+ * one.
249
+ *
250
+ * A LOWER-INFORMATION RECORD DOES NOT REPLACE A HIGHER-INFORMATION ONE (repair
251
+ * cycle 3). The render that carries no readable percentage is still recorded
252
+ * when there is nothing better on disk — that is what tells "rendered, nothing
253
+ * usable" apart from "never rendered" (§17-B) — but it does NOT overwrite a
254
+ * record whose percentage IS readable. Overwriting silences a real comparison:
255
+ * measured 2026-09-14, a witness giving a real 3.3% window gap reported
256
+ * `disagree` with its sentence, and one render whose payload lacked
257
+ * `context_window` turned the same comparison into `unverifiable` with the
258
+ * sentence suppressed. A render that says less must not delete a sample that
259
+ * says more.
260
+ *
261
+ * AND THE PARSE IS INSIDE A TRY (repair cycle 3). The module's contract at the
262
+ * top of this comment — "it never throws" — was true only by inspection: the
263
+ * `parseHarnessWitness` call sat above the only `try`, and a payload shape the
264
+ * guards did not anticipate escaped the function and killed the render (see
265
+ * `parseHarnessWitness` for the measured case). A nested `try` rather than a
266
+ * wider one, because the outer catch says something different: it is the
267
+ * temp+rename FALLBACK, and a parse failure has no JSON to fall back TO.
268
+ * A payload this module cannot read is a missing observation, which is the
269
+ * outcome the contract already prefers.
270
+ */
271
+ export function writeHarnessWitness(input) {
272
+ if (input.projectRoot === null || input.sessionId === null)
273
+ return false;
274
+ let witness;
275
+ try {
276
+ witness = parseHarnessWitness({ stdin: input.stdin, nowMs: input.nowMs });
277
+ }
278
+ catch {
279
+ return false;
280
+ }
281
+ if (witness === null)
282
+ return false;
283
+ const path = harnessWitnessPath(input.projectRoot, input.sessionId);
284
+ if (!existsSync(dirname(path)))
285
+ return false;
286
+ if (witness.usedPercentage === null) {
287
+ const existing = readHarnessWitness({ projectRoot: input.projectRoot, sessionId: input.sessionId });
288
+ if (existing.kind === 'valid' && existing.witness.usedPercentage !== null)
289
+ return false;
290
+ }
291
+ const json = `${JSON.stringify(witness, null, 2)}\n`;
292
+ const tempPath = `${path}.tmp-${process.pid}`;
293
+ try {
294
+ writeFileSync(tempPath, json, 'utf8');
295
+ renameSync(tempPath, path);
296
+ return true;
297
+ }
298
+ catch {
299
+ rmSync(tempPath, { force: true });
300
+ try {
301
+ writeFileSync(path, json, 'utf8');
302
+ return true;
303
+ }
304
+ catch {
305
+ return false;
306
+ }
307
+ }
308
+ }
309
+ /**
310
+ * Read the record for a session. Anything that exists but cannot be used as a
311
+ * record reads as `invalid` — never as `missing`, which would erase the one
312
+ * fact the file's existence carries: a render happened here.
313
+ */
314
+ export function readHarnessWitness(input) {
315
+ const path = harnessWitnessPath(input.projectRoot, input.sessionId);
316
+ if (!existsSync(path))
317
+ return { kind: 'missing' };
318
+ try {
319
+ const parsed = JSON.parse(readFileSync(path, 'utf8'));
320
+ if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed))
321
+ return { kind: 'invalid' };
322
+ const record = parsed;
323
+ const unit = record['usedPercentageUnit'];
324
+ return {
325
+ kind: 'valid',
326
+ witness: {
327
+ schemaVersion: finiteNumber(record['schemaVersion']) ?? WITNESS_SCHEMA_VERSION,
328
+ capturedAt: typeof record['capturedAt'] === 'string' ? record['capturedAt'] : '',
329
+ usedPercentage: finiteNumber(record['usedPercentage']),
330
+ usedPercentageRaw: finiteNumber(record['usedPercentageRaw']),
331
+ usedPercentageUnit: unit === 'fraction' || unit === 'percent' || unit === 'unestablished' ? unit : null,
332
+ modelWindowTokens: finiteNumber(record['modelWindowTokens']),
333
+ usageTokens: finiteNumber(record['usageTokens']),
334
+ outerSessionId: typeof record['outerSessionId'] === 'string' ? record['outerSessionId'] : null
335
+ }
336
+ };
337
+ }
338
+ catch {
339
+ // Unreadable (permissions, a directory where the file belongs, a torn
340
+ // read) and unparseable both land here. They are one fact for the caller:
341
+ // the file is there and no record can be taken from it.
342
+ return { kind: 'invalid' };
343
+ }
344
+ }
345
+ /**
346
+ * The budget, in tokens: what a difference between the two ratios can be
347
+ * explained by WITHOUT the two denominators being different, once the
348
+ * sampling skew has been taken out (see `compareHarnessWitness`).
349
+ *
350
+ * rounding 0.005 x window — the harness's percentage is rounded
351
+ * numerator 0.0017 x usedTokens — measured disagreement of the two sums
352
+ *
353
+ * Sample skew is deliberately NOT a term here. It is not a budget at all: it
354
+ * is MEASURED per sample from the two token counts and SUBTRACTED from the
355
+ * deviation, because under equal denominators the token difference and the
356
+ * ratio difference are the same number — see `compareHarnessWitness`.
357
+ */
358
+ export function witnessToleranceTokens(input) {
359
+ return WITNESS_PERCENT_ROUNDING_FRACTION * input.windowTokens
360
+ + WITNESS_NUMERATOR_FRACTION * input.usedTokens;
361
+ }
362
+ /**
363
+ * Whether this sample can support an `agree`.
364
+ *
365
+ * `agree` is a claim that the two denominators MATCH. A window difference `d`
366
+ * shows up in the residual as `-d x harnessPct/(1+d)` — it carries the
367
+ * WITNESS's ratio, not peaks-loop's, because the residual is a difference of
368
+ * two ratios and therefore scales with how much of the harness's window was in
369
+ * use when the witness was captured. The observation also carries an error of
370
+ * up to one budget (`tolerance`): the harness's percentage is rounded, and the
371
+ * two sides do not sum exactly the same token quantities. So the smallest
372
+ * difference this guard claims to detect has to leave a residual bigger than
373
+ * twice the budget, or it hides inside the budget's own uncertainty and the
374
+ * strongest answer goes to a real gap.
375
+ *
376
+ * Gating on `peaksRatio` instead — which is what this guard did until repair
377
+ * cycle 2 — is blind to exactly that: it is a function of a number the signal
378
+ * does not contain. Measured (2026-09-13): with the budget and the gate as they
379
+ * were, a real 3.3% gap was reported `agree` at every ratio below ~0.61, 2,436
380
+ * of 95,475 swept samples across ratios 0.18-0.99 reported `agree` over a real
381
+ * gap of 3-8%, and `unverifiable` was unreachable above 0.18 of the window —
382
+ * the one answer that was honest there could not be produced there.
383
+ */
384
+ function sampleSupportsAgreement(harnessPct, tolerance) {
385
+ const smallestGapSignal = (MIN_DETECTABLE_WINDOW_DIFFERENCE * Math.abs(harnessPct)) / (1 + MIN_DETECTABLE_WINDOW_DIFFERENCE);
386
+ return smallestGapSignal > 2 * tolerance;
387
+ }
388
+ /** Which of the three `absent` states applies, in the reader's own words. */
389
+ function absentReason(cause) {
390
+ if (cause === 'session-dir-missing') {
391
+ return 'this session has no runtime directory yet, so nothing has been captured for it';
392
+ }
393
+ if (cause === 'unreadable') {
394
+ return 'a witness file exists for this session but could not be read as a record — the statusline DID '
395
+ + 'render here, and its record is unreadable, malformed, or not the shape this reader expects';
396
+ }
397
+ return 'no render has been recorded for this session — the statusline has not rendered through peaks-loop here';
398
+ }
399
+ /** Why a recorded render cannot be compared, in the reader's own words. */
400
+ function unusableReason(witness) {
401
+ if (witness.usedPercentageRaw === null) {
402
+ return 'the payload carried no `used_percentage`, so this render recorded nothing to compare';
403
+ }
404
+ if (witness.usedPercentageUnit === 'unestablished') {
405
+ return `the payload reported used_percentage ${witness.usedPercentageRaw} and carried no token snapshot `
406
+ + 'to settle whether that is a fraction or a percent; the raw value is recorded, and no comparison is '
407
+ + 'made from it';
408
+ }
409
+ return `the payload reported used_percentage ${witness.usedPercentageRaw}, which is on neither scale `
410
+ + '(a 0..1 fraction or a 0..100 percent); the raw value is recorded, and no comparison is made from it';
411
+ }
412
+ /**
413
+ * Compare peaks-loop's ratio with the harness's own percentage.
414
+ *
415
+ * Two identifiers must match before any comparison is allowed:
416
+ * 1. the SESSION — a witness written by another harness session on the same
417
+ * project is not evidence about this one. Lenient in the same direction as
418
+ * the compact-event attribution: refused only when BOTH ids resolve and
419
+ * differ, because a guard that turns a missing field into a permanent
420
+ * "cannot tell" is the failure this whole slice exists to remove.
421
+ * 2. the MOMENT — a witness is a snapshot, and the harness documents that its
422
+ * percentage depends on when it was calculated. The token counts are what
423
+ * says whether the two numbers came from the same API response.
424
+ *
425
+ * THE MOMENT IS REMOVED, NOT BUDGETED (repair cycle 1). If the two ratios
426
+ * share a denominator W, the harness's count and peaks-loop's count differ by
427
+ * exactly the sampling skew, so
428
+ *
429
+ * peaksRatio - harnessPct == (peaksTokens - witnessTokens) / W
430
+ *
431
+ * holds identically — it is not a tolerance to be granted, it is an equality
432
+ * to be tested. Budgeting the skew instead (`tolerance += |skew|`) made the
433
+ * budget grow by `x` while the deviation grew by `x(1+g)`, so a real window
434
+ * difference `g` was cancelled for every skew large enough to absorb it: a
435
+ * measured 3.3% denominator gap read `agree` for skew in [12,400, 22,100]
436
+ * tokens, and gaps up to 5.3% never surfaced at all. Subtracting the skew from
437
+ * the deviation and testing the remainder against a budget that contains only
438
+ * rounding and numerator disagreement makes the test invariant to skew by
439
+ * construction, and leaves `-g x harnessPct/(1+g)` — the window difference
440
+ * itself — as the only thing the residual can be.
441
+ *
442
+ * AND THE RESIDUAL'S SIZE IS THE WITNESS'S, NOT PEAKS-LOOP'S (repair cycle 2).
443
+ * `-g x harnessPct/(1+g)` carries the witness's ratio, so a witness captured
444
+ * while the session was small cannot show a window difference that a bigger
445
+ * witness would. The three answers therefore do not share one gate: a residual
446
+ * past the budget is a `disagree` at any witness size, while `agree` needs the
447
+ * sample to be sharp enough to support it (`sampleSupportsAgreement`).
448
+ */
449
+ export function compareHarnessWitness(input) {
450
+ const base = {
451
+ peaksRatio: input.peaksRatio,
452
+ peaksTokens: input.peaksTokens,
453
+ witnessTokens: input.witness?.usageTokens ?? null,
454
+ witnessedAt: input.witness?.capturedAt ?? null,
455
+ witnessRawPercentage: input.witness?.usedPercentageRaw ?? null,
456
+ witnessPercentageUnit: input.witness?.usedPercentageUnit ?? null
457
+ };
458
+ if (input.witness === null) {
459
+ return {
460
+ ...base,
461
+ verdict: 'absent',
462
+ reason: absentReason(input.absentCause ?? 'not-rendered'),
463
+ harnessPct: null,
464
+ deviation: null,
465
+ residual: null,
466
+ tolerance: null
467
+ };
468
+ }
469
+ const witness = input.witness;
470
+ // The version is the record's own statement of which unit rule normalised
471
+ // `usedPercentage`. Reading a record written under an older rule as if it
472
+ // were this one takes an old value under a new rule: a v1 file that recorded
473
+ // the payload's bare `1` as a fraction holds `usedPercentage: 1` (100%), and
474
+ // reading that as the current shape reports a confident disagreement and
475
+ // blames the two denominators for a unit misfire. Those files are already on
476
+ // disk for anyone who ran an earlier revision, and the next render replaces
477
+ // one — so the honest answer here is to abstain, not to migrate a value whose
478
+ // raw form was never recorded.
479
+ if (witness.schemaVersion !== WITNESS_SCHEMA_VERSION) {
480
+ return {
481
+ ...base,
482
+ verdict: 'unverifiable',
483
+ reason: `this witness was recorded under schema v${witness.schemaVersion} and this revision reads `
484
+ + `v${WITNESS_SCHEMA_VERSION}, which do not agree on what the recorded percentage means; it is not `
485
+ + 'compared, and the next render replaces it',
486
+ harnessPct: null,
487
+ deviation: null,
488
+ residual: null,
489
+ tolerance: null
490
+ };
491
+ }
492
+ if (witness.outerSessionId !== null &&
493
+ input.outerSessionId !== null &&
494
+ witness.outerSessionId !== input.outerSessionId) {
495
+ return {
496
+ ...base,
497
+ verdict: 'foreign-session',
498
+ reason: 'the witness names a different harness session, so it says nothing about this one',
499
+ harnessPct: witness.usedPercentage,
500
+ deviation: witness.usedPercentage === null ? null : input.peaksRatio - witness.usedPercentage,
501
+ residual: null,
502
+ tolerance: null
503
+ };
504
+ }
505
+ const harnessPct = witness.usedPercentage;
506
+ const windowTokens = input.peaksWindowTokens;
507
+ const peaksTokens = input.peaksTokens;
508
+ const witnessTokens = witness.usageTokens;
509
+ if (harnessPct === null) {
510
+ return {
511
+ ...base,
512
+ verdict: 'unverifiable',
513
+ reason: unusableReason(witness),
514
+ harnessPct: null,
515
+ deviation: null,
516
+ residual: null,
517
+ tolerance: null
518
+ };
519
+ }
520
+ if (windowTokens === null || windowTokens <= 0 || peaksTokens === null || witnessTokens === null) {
521
+ return {
522
+ ...base,
523
+ verdict: 'unverifiable',
524
+ reason: 'the probe and the witness do not both carry a token snapshot, so the two numbers cannot be shown to describe the same moment',
525
+ harnessPct,
526
+ deviation: input.peaksRatio - harnessPct,
527
+ residual: null,
528
+ tolerance: null
529
+ };
530
+ }
531
+ const deviation = input.peaksRatio - harnessPct;
532
+ // The skew is measured here, not assumed, and it is subtracted rather than
533
+ // granted: see the doc comment above. Signed, so a witness captured before
534
+ // or after the probe is corrected in the direction it is actually off.
535
+ const skewRatio = (peaksTokens - witnessTokens) / windowTokens;
536
+ const residual = deviation - skewRatio;
537
+ const tolerance = witnessToleranceTokens({ windowTokens, usedTokens: peaksTokens }) / windowTokens;
538
+ // The two answers are not symmetric, so they do not share one gate.
539
+ // `disagree` only claims that SOME difference exists, and a residual past the
540
+ // budget is exactly that claim — sound at any ratio, any witness age.
541
+ if (Math.abs(residual) > tolerance) {
542
+ return { ...base, verdict: 'disagree', reason: null, harnessPct, deviation, residual, tolerance };
543
+ }
544
+ // `agree` claims there is no difference, which needs the sample to be sharp
545
+ // enough to support it — see `sampleSupportsAgreement`.
546
+ if (!sampleSupportsAgreement(harnessPct, tolerance)) {
547
+ return {
548
+ ...base,
549
+ verdict: 'unverifiable',
550
+ reason: 'the witness covers too small a share of the harness window for this sample to separate a real window difference from the budget’s own uncertainty',
551
+ harnessPct,
552
+ deviation,
553
+ residual,
554
+ tolerance
555
+ };
556
+ }
557
+ return { ...base, verdict: 'agree', reason: null, harnessPct, deviation, residual, tolerance };
558
+ }
559
+ /**
560
+ * One-way sentence for a disagreeing witness. Advising, never asking: an
561
+ * auto-compact observation must never become an `AskUserQuestion` (see
562
+ * `.peaks/memory/auto-compact-threshold-policy.md`).
563
+ *
564
+ * The sentence states the quantity that decided it (the residual, after the
565
+ * measured skew was removed) and the raw harness value with the reading that
566
+ * was taken from it. Both are there so a reader can tell a real denominator
567
+ * difference from a unit misread without going back to the file.
568
+ */
569
+ export function describeHarnessWitness(comparison) {
570
+ if (comparison.verdict !== 'disagree')
571
+ return null;
572
+ const pct = (value) => `${(value * 100).toFixed(1)}%`;
573
+ const skew = comparison.deviation === null || comparison.residual === null
574
+ ? null
575
+ : comparison.deviation - comparison.residual;
576
+ const readAs = comparison.witnessPercentageUnit === null
577
+ ? ''
578
+ : ` (raw ${comparison.witnessRawPercentage}, read as a ${comparison.witnessPercentageUnit})`;
579
+ return `Harness context witness disagrees: the harness reports ${pct(comparison.harnessPct ?? 0)} used${readAs}, `
580
+ + `peaks-loop computes ${pct(comparison.peaksRatio)} (deviation ${pct(Math.abs(comparison.deviation ?? 0))}`
581
+ + `${skew === null ? '' : `, of which the measured sampling skew explains ${pct(Math.abs(skew))}`}, `
582
+ + `leaving ${pct(Math.abs(comparison.residual ?? 0))} against a budget of ${pct(comparison.tolerance ?? 0)}). `
583
+ + 'The two ratios do not share a denominator: '
584
+ + 'peaks-loop\'s configured auto-compact window and the harness\'s effective auto-compact window are '
585
+ + 'different numbers. Nothing is blocked.';
586
+ }
587
+ /** Convenience for the CLI: read + compare in one call. */
588
+ export function readAndCompareHarnessWitness(input) {
589
+ const read = readHarnessWitness({ projectRoot: input.projectRoot, sessionId: input.sessionId });
590
+ return compareHarnessWitness({
591
+ witness: read.kind === 'valid' ? read.witness : null,
592
+ peaksRatio: input.peaksRatio,
593
+ peaksTokens: input.peaksTokens,
594
+ peaksWindowTokens: input.peaksWindowTokens,
595
+ outerSessionId: input.outerSessionId,
596
+ // Only the reader knows the session directory, so only the reader can tell
597
+ // the two no-file states apart. The third cause is not a directory
598
+ // question — `invalid` IS the fact that a render happened — so it is
599
+ // answered by the read and never falls through to the ternary.
600
+ absentCause: read.kind === 'invalid'
601
+ ? 'unreadable'
602
+ : existsSync(getSessionDir(input.projectRoot, input.sessionId))
603
+ ? 'not-rendered'
604
+ : 'session-dir-missing'
605
+ });
606
+ }