agentwrangler 0.1.1-next.2e8c141 → 0.1.1-next.2f6c1df

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/README.md +82 -162
  2. package/dist/apply/jobs.js +6 -2
  3. package/dist/daemon/router.js +1 -1
  4. package/dist/detector/detectors/d10_catalog_footprint.js +2 -2
  5. package/dist/detector/detectors/d1_ctx_always_loaded.js +9 -9
  6. package/dist/detector/detectors/d2_session_long_full_context.js +2 -2
  7. package/dist/detector/detectors/d4_model_mismatch.js +3 -3
  8. package/dist/detector/detectors/d6_tool_result_bloat.js +1 -1
  9. package/dist/detector/detectors/d7_loop_retry_waste.js +1 -1
  10. package/dist/detector/detectors/d8_cache_write_churn.js +2 -2
  11. package/dist/detector/detectors/d9_idle_background_session.js +1 -1
  12. package/dist/hook/precompact-checkpoint-hook.mjs +26 -13
  13. package/dist/query/api/cost-per-success.js +13 -1
  14. package/dist/query/api/delivery.js +25 -3
  15. package/dist/query/api/effectiveness.js +30 -5
  16. package/dist/query/api/efficiency-headroom.js +12 -14
  17. package/dist/query/api/overview.js +77 -3
  18. package/dist/query/spend.js +64 -2
  19. package/dist/ui/assets/{BarChart-BwFZrcLn.js → BarChart-BC-ciS9d.js} +1 -1
  20. package/dist/ui/assets/BriefsPage-Dls6npt7.js +2 -0
  21. package/dist/ui/assets/{CacheWriteSpikesChart-zJIJWO41.js → CacheWriteSpikesChart-ncVwVoud.js} +1 -1
  22. package/dist/ui/assets/{CartesianChart-qApXl3Ak.js → CartesianChart-CRV6h2et.js} +1 -1
  23. package/dist/ui/assets/Chip-EizRm13x.js +1 -0
  24. package/dist/ui/assets/{ComposedChart-DYf1wymW.js → ComposedChart-CrOxp6Ym.js} +1 -1
  25. package/dist/ui/assets/{EmptyState-DL8gAVnJ.js → EmptyState-9kcoimjj.js} +1 -1
  26. package/dist/ui/assets/{FlavorDecomposition-CvJW78M9.js → FlavorDecomposition-DRZxH0vQ.js} +1 -1
  27. package/dist/ui/assets/FrictionCell-BGYB9UKh.js +2 -0
  28. package/dist/ui/assets/GlossaryPage-CRyQb1Hn.js +1 -0
  29. package/dist/ui/assets/HotSessionsPage-DfE90Kxi.js +1 -0
  30. package/dist/ui/assets/{InfoTip-Ckc9_LpJ.js → InfoTip-CmcYFNQt.js} +1 -1
  31. package/dist/ui/assets/{Legend-C_YMp0Rv.js → Legend-BCXsSaEM.js} +1 -1
  32. package/dist/ui/assets/{Line-DU4UvEBO.js → Line-2rUzs15v.js} +1 -1
  33. package/dist/ui/assets/OverviewPage-Bm5KvXqa.js +2 -0
  34. package/dist/ui/assets/RecommendationsPage-CvSl3lzp.js +2 -0
  35. package/dist/ui/assets/{Scatter-CNetVLuX.js → Scatter-DpIZt_y5.js} +1 -1
  36. package/dist/ui/assets/SessionDetailPage-w_lSgite.js +1 -0
  37. package/dist/ui/assets/SettingsPage-CfUoP8lB.js +25 -0
  38. package/dist/ui/assets/{Skeleton-BO9mXDXU.js → Skeleton-DjuxPm_D.js} +1 -1
  39. package/dist/ui/assets/SpendPercentileChip-DYun7Wzh.js +1 -0
  40. package/dist/ui/assets/{TrendChart-DKy9aH5G.js → TrendChart-gyhHK9H_.js} +1 -1
  41. package/dist/ui/assets/WorkspaceDetailPage-DacPKaKA.js +1 -0
  42. package/dist/ui/assets/WorkspacesPage-DiNQaAYG.js +1 -0
  43. package/dist/ui/assets/{chart-theme-BPPMjbVX.js → chart-theme-DnpNuyRt.js} +1 -1
  44. package/dist/ui/assets/{graphicalItemSelectors-DgKF1Dg2.js → graphicalItemSelectors-CFhpQjLz.js} +1 -1
  45. package/dist/ui/assets/{index-SrfNUeBZ.css → index-D_IJsvkJ.css} +1 -1
  46. package/dist/ui/assets/{index-CUyiomzU.js → index-hbaLYHUX.js} +3 -3
  47. package/dist/ui/assets/{prompt-templates-DhR4Qoy9.js → prompt-templates-CVzMex9L.js} +4 -4
  48. package/dist/ui/assets/rec-sessions-ggz9MYgP.js +1 -0
  49. package/dist/ui/assets/{useExperimentalActions-BeDSmwBu.js → useExperimentalActions-B7J8gIkb.js} +1 -1
  50. package/dist/ui/index.html +2 -2
  51. package/package.json +1 -1
  52. package/dist/ui/assets/BriefsPage-CtJwHLOh.js +0 -2
  53. package/dist/ui/assets/Chip-Cpt_fEc9.js +0 -1
  54. package/dist/ui/assets/FrictionCell-DciiSd3L.js +0 -2
  55. package/dist/ui/assets/GlossaryPage-BMkZGt2D.js +0 -1
  56. package/dist/ui/assets/HotSessionsPage-Ejn385bb.js +0 -1
  57. package/dist/ui/assets/OverviewPage-BcEKdWZj.js +0 -2
  58. package/dist/ui/assets/RecommendationsPage-CHfuNP_W.js +0 -2
  59. package/dist/ui/assets/SessionDetailPage-CYm5Gxo0.js +0 -1
  60. package/dist/ui/assets/SettingsPage-DjgGZj8N.js +0 -25
  61. package/dist/ui/assets/SpendPercentileChip-BvaZ8Lzi.js +0 -1
  62. package/dist/ui/assets/WorkspaceDetailPage-dutc4EpN.js +0 -1
  63. package/dist/ui/assets/WorkspacesPage-isSML2fh.js +0 -1
package/README.md CHANGED
@@ -1,184 +1,104 @@
1
1
  <p align="center">
2
- <img src="docs/assets/logo.png" alt="AgentWrangler logo" width="260">
2
+ <img src="https://raw.githubusercontent.com/Doogit/AgentWrangler/main/docs/assets/logo.png" alt="AgentWrangler logo" width="200">
3
3
  </p>
4
4
 
5
- <h1 align="center">AgentWrangler</h1>
5
+ # AgentWrangler
6
6
 
7
- <p align="center">
8
- <b>See where your Claude Code tokens go — and whether the work actually shipped.</b><br>
9
- Local-first observability for Claude Code: token spend, session outcomes, waste detection,
10
- and installable guardrails. Runs entirely on your machine.
11
- </p>
12
-
13
- <p align="center">
14
- <a href="https://www.npmjs.com/package/agentwrangler"><img src="https://img.shields.io/npm/v/agentwrangler" alt="npm version"></a>
15
- <a href="https://github.com/Doogit/AgentWrangler/actions/workflows/ci.yml"><img src="https://github.com/Doogit/AgentWrangler/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
16
- <a href="https://img.shields.io/node/v/agentwrangler"><img src="https://img.shields.io/node/v/agentwrangler" alt="node version"></a>
17
- <a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache--2.0-blue" alt="license"></a>
18
- </p>
19
-
20
- ![AgentWrangler dashboard — spend verdict, model mix, recommendations, and per-repo efficiency](docs/assets/dashboard.gif)
21
-
22
- <!--
23
- All screenshots and the demo GIF in this README are captured from a SANITIZED instance
24
- (Vite test-mode fixtures — anonymized names, no live data) so no real workspace/repository
25
- names or paths are committed (SEC-101). Regenerate against `npx vite --mode test`.
26
- -->
27
-
28
- ## Why
29
-
30
- Claude Code tells you almost nothing about where your token budget goes. Sessions balloon,
31
- caches miss, background agents idle — and you find out when you hit the rate limit.
32
- AgentWrangler reads the transcripts Claude Code already writes to your disk and answers three
33
- questions:
34
-
35
- - **Where did the tokens go?** Per-model, per-workspace, per-session spend with cache economics.
36
- - **Did the work ship?** Sessions are linked to the pull requests they merged or closed.
37
- - **What should I change?** Ranked waste-source detectors with modeled savings — and one-click
38
- guardrail hooks that warn *inside* Claude Code before waste happens.
39
-
40
- No cloud backend, no telemetry, no account. A daemon on `127.0.0.1`, a dashboard in your
41
- browser, and a SQLite file in your home directory.
7
+ See where your Claude Code tokens go, inspect costly sessions, and track whether a change helped.
8
+ AgentWrangler reads local Claude Code transcripts and serves a dashboard on your machine.
9
+ No account or cloud backend is required.
42
10
 
43
11
  ## Quick start
44
12
 
13
+ Requires **Node 22-24 and npm**, plus Claude Code transcripts for populated charts.
14
+
45
15
  ```sh
46
16
  npx agentwrangler@latest
47
17
  ```
48
18
 
49
- That's it — requires Node **22–24**. The daemon starts on `http://127.0.0.1:47821`, opens your
50
- browser, and scans your `~/.claude/projects` transcripts in the background; the dashboard
51
- appears immediately and fills in as the scan completes.
52
-
53
- More options (install from source, GitHub outcomes sync, environment variables):
54
- **[Getting started →](docs/getting-started.md)**
55
-
56
- ## Features
57
-
58
- ### Overview — verdict first, details on demand
59
-
60
- One screen answers "how bad is it this week": a spend verdict with trend, your top waste
61
- source with a copyable fix prompt, live rate-limit gauges (5-hour and 7-day), a burn forecast
62
- against your calibrated weekly limit, hot sessions, cache efficiency, and per-model
63
- context-per-turn tiles.
64
-
65
- ![Overview tab — at-a-glance verdict, rate limits, burn forecast, hot sessions](docs/assets/overview.png)
66
-
67
- ### Recommendations — waste-source detectors, ranked by impact
68
-
69
- Ten detector families watch your sessions for the patterns that actually burn tokens: cache
70
- misses (the biggest single lever), session hygiene, retry/redundant-read loops, tool-result
71
- bloat, model routing, idle background sessions, and more. Each recommendation shows modeled
72
- weekly savings, a confidence tier, and a concrete action — install a hook, copy a config
73
- snippet, or copy a guided prompt straight into Claude Code. Adopted changes flow into an
74
- **impact ledger** that tracks the measured effect, and modeled savings are never counted as
75
- achieved.
76
-
77
- ![Recommendations tab — ranked detector families with modeled savings and one-click actions](docs/assets/recommendations.png)
78
-
79
- ### Installable guardrails — local checks inside Claude Code, before the waste
80
-
81
- Five small hooks you can install from the dashboard (directly, or via a copyable prompt that
82
- Claude Code applies itself):
83
-
84
- | Guardrail | What it does |
85
- |---|---|
86
- | **Context-budget warning** | Warns when a session's context crosses your soft/hard thresholds |
87
- | **Loop guard** | Flags repeated identical tool failures before they spiral |
88
- | **Burn alert** | Catches idle sessions still burning tokens in the background |
89
- | **Pre-compaction checkpoint** | Copies the raw local transcript before an automatic compaction, subject to a local retention cap |
90
- | **Dangerous-command guard** | Asks before risky shell commands and denies a small catastrophe list |
91
-
92
- The context-budget and burn hooks warn. The loop guard warns before it denies repeated identical
93
- failures, and the dangerous-command guard can ask or deny. Direct install enables all five hooks;
94
- the copied install prompt enables the context-budget, loop, and burn hooks only. Thresholds are
95
- tunable from Settings, and direct uninstall removes every AgentWrangler hook.
96
-
97
- ### Sessions — who spent it, and on what
98
-
99
- The highest-cost sessions ranked with their output-to-context split, model, friction band
100
- (API errors, tool failures, compactions, interrupts), and a "top X% by spend" self-percentile
101
- chip. Drill into any session for a turn-by-turn timeline, its cost drivers (which detectors
102
- fired and how hard), and a guided fix prompt built only from measured numbers.
103
-
104
- ![Sessions tab — highest-cost sessions with friction bands and spend percentiles](docs/assets/sessions.png)
105
-
106
- <details>
107
- <summary>Session detail view</summary>
108
-
109
- ![Session detail — per-session KPIs, cost drivers, and a measured-context fix prompt](docs/assets/session-detail.png)
110
- </details>
111
-
112
- ### Workspaces — spend efficiency by repository
113
-
114
- Every repo you run Claude Code in, with spend share, trend, context-per-turn, cache-write
115
- share, Opus share, and $/turn. With a GitHub token configured, sessions are linked to the PRs
116
- and commits they produced — so you can see cost-per-merged-PR, not just cost.
117
-
118
- ![Workspaces tab — per-repository spend, efficiency, and outcome linkage](docs/assets/workspaces.png)
119
-
120
- <details>
121
- <summary>Workspace detail view</summary>
122
-
123
- ![Workspace detail — top sessions, context composition, and outcomes](docs/assets/workspace-detail.png)
124
- </details>
125
-
126
- ### Weekly brief — one page, three decisions
127
-
128
- The week in one screen: spend verdict, what changed vs. last week, the top actions to take —
129
- with a **Copy as Markdown** button so the whole brief drops into a standup note or a message.
19
+ Open **http://127.0.0.1:47821** if the browser does not open automatically. Keep the terminal
20
+ running; press **Ctrl+C** to stop. The first scan runs in the background. An empty history is
21
+ valid: use Claude Code, then return after ingestion. If a daemon is already running, stop it
22
+ before starting another on the same port.
130
23
 
131
- ![Briefs tab — weekly verdict, week-over-week deltas, and top actions](docs/assets/briefs.png)
24
+ [Install from source, configure scan roots, or troubleshoot](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md).
132
25
 
133
- ### Honest numbers, labeled as such
26
+ ## Your first five minutes
134
27
 
135
- Every metric carries an honesty-tier chip — `EXACT`, `LIST_EQUIV`, `MODELED`, `PROXY`,
136
- `DIRECTIONAL`, `EXPERIMENTAL` — so you always know what is measured versus estimated. Dollar
137
- figures are list-price *equivalents* (subscription plans aren't billed per token; tokens drive
138
- rate limits), and the built-in glossary ("How to read this dashboard") defines every
139
- key metric in plain language. The full tour: **[Dashboard tour →](docs/dashboard-tour.md)**
28
+ 1. **Check the scan.** In Overview, read the onboarding status. A completed scan with no
29
+ recommendations is a valid result. For an unexpected empty history, inspect scan roots
30
+ and parser health in Settings; saved scan-root changes require a daemon restart.
31
+ 2. **Find one expensive session.** Select a date window, open Workspaces, choose a workspace,
32
+ then open one of its sessions. Compare context, cache writes, and output before deciding
33
+ what to change.
34
+ 3. **Inspect one recommendation.** Open Recommendations and **Show details** on an instance.
35
+ Read its evidence and caveats. A modeled amount is a projection; directional advice may
36
+ have no dollar estimate. If nothing fires, there is nothing to adopt just to finish setup.
37
+ 4. **Make one deliberate change.** A copied prompt is an artifact to review and run yourself.
38
+ It does not edit files. After completing a supported change, use **I completed the change**
39
+ when offered, then **Track this change** to record a baseline.
40
+ 5. **Return to the Impact ledger.** Tracking starts observation, not a savings claim. Eligible
41
+ signals use a 14-day observation window and may finish inconclusive. You can finish this
42
+ first visit without a GitHub token, calibration, or installed hooks.
140
43
 
141
- ## Privacy — local-only by design
44
+ Dollar figures are **list-price equivalents**, not your subscription bill. Modeled savings
45
+ are not achieved savings, and observed improvement does not prove the change caused it.
46
+ GitHub linkage adds outcome metadata; calibration enables the burn forecast. Local spend,
47
+ session inspection, and supported recommendation tracking work without either.
142
48
 
143
- - The daemon binds to **`127.0.0.1`** only. There is no cloud backend, telemetry, or account.
144
- - Most dashboard data is **aggregates, ids, counts, and structural anchors**. Local command text
145
- and filesystem paths can also be retained in SQLite; treat the database as sensitive. The optional
146
- direct-install PreCompact hook can separately copy full raw transcripts locally.
147
- - The optional GitHub token is read locally, never logged, never persisted to the DB.
148
- - Usage refreshes can call Anthropic with an existing Claude Code sign-in. GitHub outcomes sync
149
- calls GitHub only when a token is configured; calibration and G2 judging are separate opt-ins.
49
+ [Worked example: trim always-loaded context and inspect its effect](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#example-trim-always-loaded-context).
150
50
 
151
- Full details, including exactly what is and isn't stored: **[Privacy model →](docs/privacy.md)**
51
+ ## Dashboard preview
152
52
 
153
- ## Configuration
53
+ ![Synthetic Overview: spend for the selected window, model mix, and links to sessions and recommendations](https://raw.githubusercontent.com/Doogit/AgentWrangler/main/docs/assets/overview.png)
154
54
 
155
- Everything is optional with sensible defaults — port, DB path, scan roots, GitHub token, and
156
- more are environment variables documented in [Getting started](docs/getting-started.md#configuration)
157
- and [`.env.example`](.env.example).
55
+ This static preview uses synthetic data. [The dashboard tour](https://github.com/Doogit/AgentWrangler/blob/main/docs/dashboard-tour.md)
56
+ contains the other views and an optional animated preview.
158
57
 
159
- ## Documentation
58
+ ## Pick a question
160
59
 
161
- | Page | What's in it |
60
+ | I want to... | Start here |
162
61
  |---|---|
163
- | [Getting started](docs/getting-started.md) | Install paths, optional setup, configuration, troubleshooting |
164
- | [Dashboard tour](docs/dashboard-tour.md) | Every tab in depth, plus the metric vocabulary |
165
- | [Privacy model](docs/privacy.md) | Local storage, raw-checkpoint exception, and network integrations |
166
- | [Architecture](docs/planning/AgentWrangler_Technical_Architecture_v4_5_0.md) | Daemon, ingestion, detector, and query design |
167
- | [Data model & metrics](docs/planning/AgentWrangler_Data_Model_and_Metrics_v2.md) | SQLite schema and metric definitions |
168
- | [Contributing](.github/CONTRIBUTING.md) | Dev setup, checks, PR expectations |
169
- | [Security policy](.github/SECURITY.md) | Threat model and how to report a vulnerability |
170
-
171
- ## Limitations
172
-
173
- - Reads Claude Code's JSONL transcript format; an upstream format change can require an
174
- ingestion update.
175
- - Tested on Windows; macOS/Linux are believed working — reports welcome.
176
- - Single local user by design — no multi-user or team-aggregation mode.
177
- - Outcome linkage needs a read-only GitHub token; without one the feature stays inert (and
178
- Settings says so — nothing fails silently).
179
-
180
- ## License
181
-
182
- [Apache 2.0](LICENSE) © 2026 AgentWrangler contributors.
183
- See [CONTRIBUTING.md](.github/CONTRIBUTING.md), [SECURITY.md](.github/SECURITY.md), and
184
- [CODE_OF_CONDUCT.md](.github/CODE_OF_CONDUCT.md).
62
+ | Find where my tokens went | [Workspaces and Sessions](https://github.com/Doogit/AgentWrangler/blob/main/docs/dashboard-tour.md#workspaces) |
63
+ | Turn a recommendation into a change | [Worked example](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#example-trim-always-loaded-context) |
64
+ | Understand the dollars and verdicts | [Metric vocabulary](https://github.com/Doogit/AgentWrangler/blob/main/docs/dashboard-tour.md#glossary-how-to-read-this-dashboard) |
65
+ | Fix an empty dashboard | [First launch](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#first-launch) |
66
+ | Connect GitHub or calibrate limits | [Optional setup](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#optional-setup) |
67
+ | Read a weekly summary | [Briefs](https://github.com/Doogit/AgentWrangler/blob/main/docs/dashboard-tour.md#briefs) |
68
+
69
+ ## Installable guardrails — local checks inside Claude Code, before the waste
70
+
71
+ Hooks are optional and require an explicit install. Settings **Install directly** installs
72
+ five hooks; **Copy install prompt** prepares instructions for three. Copying alone installs nothing.
73
+
74
+ | Hook | Behavior | Install path |
75
+ |---|---|---|
76
+ | Context-budget | Warns when context crosses configured thresholds | Direct or copied prompt |
77
+ | Loop guard | Warns, then can deny repeated identical failures | Direct or copied prompt |
78
+ | Burn alert | Warns about session budget consumption | Direct or copied prompt |
79
+ | Dangerous-command | Can ask or deny risky commands | Direct only |
80
+ | PreCompact checkpoint | Copies raw transcripts locally, with retention limits | Direct only |
81
+
82
+ [Installation, removal, and the checkpoint privacy details](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#in-session-guardrails).
83
+
84
+ ## Privacy and limits
85
+
86
+ - The daemon binds to **127.0.0.1**. There is no telemetry or hosted product backend.
87
+ - SQLite contains aggregates and structural data, and can also contain local command text
88
+ and filesystem paths. Treat it as sensitive. The optional PreCompact hook makes separate
89
+ raw transcript copies on your machine.
90
+ - Usage refresh can contact Anthropic using your existing Claude Code sign-in. GitHub
91
+ outcomes sync requires a configured token. [Privacy details](https://github.com/Doogit/AgentWrangler/blob/main/docs/privacy.md).
92
+ - Claude Code format changes can require parser updates. This is a single-user tool.
93
+ - Windows local validation and Linux/macOS CI smoke coverage do not establish full native
94
+ accessibility or credential-store compatibility on every platform.
95
+
96
+ ## Deeper documentation
97
+
98
+ [Configuration](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#configuration) |
99
+ [Architecture](https://github.com/Doogit/AgentWrangler/blob/main/docs/planning/AgentWrangler_Technical_Architecture_v4_5_0.md) |
100
+ [Data model and metrics](https://github.com/Doogit/AgentWrangler/blob/main/docs/planning/AgentWrangler_Data_Model_and_Metrics_v2.md) |
101
+ [Contributing](https://github.com/Doogit/AgentWrangler/blob/main/.github/CONTRIBUTING.md) |
102
+ [Security policy](https://github.com/Doogit/AgentWrangler/blob/main/.github/SECURITY.md)
103
+
104
+ [Apache 2.0](https://github.com/Doogit/AgentWrangler/blob/main/LICENSE).
@@ -200,6 +200,7 @@ function runClaudePhase(db, job, rec, phase, settingsPath, rt) {
200
200
  const changedPaths = [];
201
201
  let finalized = false;
202
202
  let fatalExitMessage = null;
203
+ let killTimer;
203
204
  const cleanup = () => {
204
205
  fs.unlink(settingsPath, () => { });
205
206
  };
@@ -251,9 +252,11 @@ function runClaudePhase(db, job, rec, phase, settingsPath, rt) {
251
252
  }
252
253
  });
253
254
  const timer = setTimeout(() => {
255
+ // Keep the job exclusive until the child has actually closed. Publishing
256
+ // FAILED here lets a retry (or workspace cleanup) race the dying process.
257
+ fatalExitMessage ??= "job timed out";
254
258
  proc.kill("SIGTERM");
255
- setTimeout(() => proc.kill("SIGKILL"), 2000);
256
- markFailed("job timed out");
259
+ killTimer = setTimeout(() => proc.kill("SIGKILL"), 2000);
257
260
  }, rt.timeoutMs);
258
261
  proc.on("error", (err) => {
259
262
  clearTimeout(timer);
@@ -261,6 +264,7 @@ function runClaudePhase(db, job, rec, phase, settingsPath, rt) {
261
264
  });
262
265
  proc.on("close", (code) => {
263
266
  clearTimeout(timer);
267
+ clearTimeout(killTimer);
264
268
  if (finalized)
265
269
  return;
266
270
  finalized = true;
@@ -264,7 +264,7 @@ export function handleApiRequest(_db, req, res, method, url) {
264
264
  sendJson(res, 200, getPractices(_db, { from, to }));
265
265
  return;
266
266
  }
267
- // GET /api/efficiency-headroom (BM2 modeled savings vs trailing-window spend)
267
+ // GET /api/efficiency-headroom (individual weekly estimates; separate selected-window spend)
268
268
  if (method === "GET" && pathname === "/api/efficiency-headroom") {
269
269
  const { from, to } = resolveWindow(parseWindowFilter(url));
270
270
  sendJson(res, 200, getEfficiencyHeadroom(_db, { from, to }));
@@ -95,7 +95,7 @@ export const d10Detector = {
95
95
  const latestProbedAt = rows.reduce((latest, row) => (latest === null || row.probed_at > latest ? row.probed_at : latest), null);
96
96
  const refs = rows.map((row) => row.file_ref);
97
97
  const evidence = {
98
- title: `Review ${state.effective_catalog_state} tool catalog: ${Math.round(catalogTokens / 1000)}K tokens`,
98
+ title: `Review tool catalog: ${Math.round(catalogTokens / 1000)}K estimated tokens`,
99
99
  component: "MCP_SCHEMAS",
100
100
  file_ref: refs.length === 1 ? refs[0] : null,
101
101
  file_refs: refs,
@@ -120,7 +120,7 @@ export const d10Detector = {
120
120
  scopeKey: "D10|global|MCP_SCHEMAS",
121
121
  category: "TOOLING",
122
122
  scope_workspace_id: null,
123
- lever: "Too many connected tools, plugins, and skills",
123
+ lever: "The tool catalog exceeds its size target; actual loaded context is not measured",
124
124
  target_metric: "catalog_context_tokens",
125
125
  // R11 is required before catalog size can become a freed-headroom claim.
126
126
  modeled_savings_u_per_wk: null,
@@ -29,16 +29,16 @@ function stepsFor(component, fileRef) {
29
29
  case "CLAUDE_MD":
30
30
  return [
31
31
  `Open ${fileRef}`,
32
- "Move changelog/history/rationale prose to a linked doc",
33
- "Keep current-state rules + pointers only",
34
- "Re-measure: probe re-sizes on next daemon pass",
32
+ "Move changelogs, history, and background explanations to a linked document",
33
+ "Keep only current rules and links here",
34
+ "Check the size again after the next local scan",
35
35
  ];
36
36
  case "MEMORY":
37
37
  return [
38
38
  `Review memory files under ${fileRef}`,
39
- "Delete stale or duplicate memories",
40
- "Consolidate overlapping facts into concise entries",
41
- "Re-measure: probe re-sizes on next daemon pass",
39
+ "Remove old or duplicate memories",
40
+ "Combine overlapping facts into concise entries",
41
+ "Check the size again after the next local scan",
42
42
  ];
43
43
  case "MCP_SCHEMAS":
44
44
  return [
@@ -62,13 +62,13 @@ function titleFor(component, tokens, target) {
62
62
  function leverFor(component) {
63
63
  switch (component) {
64
64
  case "CLAUDE_MD":
65
- return "Move changelog/history prose out of CLAUDE.md; keep current-state + pointers.";
65
+ return "Move changelogs and history out of CLAUDE.md; keep current rules and links.";
66
66
  case "MEMORY":
67
- return "Prune stale/duplicate memories; consolidate overlapping facts.";
67
+ return "Remove old or duplicate memories and combine overlapping facts.";
68
68
  case "MCP_SCHEMAS":
69
69
  return "Identify rarely-used skills/plugins; extract to on-demand or disable.";
70
70
  default:
71
- return "Trim always-loaded context source to the per-source target.";
71
+ return "Shorten this content that is loaded into every conversation.";
72
72
  }
73
73
  }
74
74
  export const d1Detector = {
@@ -81,12 +81,12 @@ export const d2Detector = {
81
81
  scopeKey: `D2|global|${formula.model}`,
82
82
  category: "CONTEXT",
83
83
  scope_workspace_id: null,
84
- lever: "/clear between unrelated tasks; split long work; avoid mid-task /compact.",
84
+ lever: "Use /clear between unrelated tasks, split long work into separate sessions, and avoid automatic compaction while you are working.",
85
85
  target_metric: "avg_context_per_turn",
86
86
  modeled_savings_u_per_wk: savingsU,
87
87
  modeled_formula: formula,
88
88
  evidence: {
89
- title: `Shorten sessions: ${n} long-context run${n === 1 ? "" : "s"} this week`,
89
+ title: `Split ${n} long session${n === 1 ? "" : "s"} this week`,
90
90
  qualifying_session_count: qualifying.length,
91
91
  qualifying_turn_count: qualifyingTurnCount,
92
92
  session_ids: sessionIds,
@@ -192,7 +192,7 @@ export const d4Detector = {
192
192
  ? {
193
193
  withheld: true,
194
194
  withheld_reason: `Sonnet weekly cap is the binding constraint (Sonnet util ${bindingSonnet.utilization} >= all-models ${perModelSnapshot.seven_day_util}) — routing Opus->Sonnet would worsen it`,
195
- title: `[withheld] Route Opus→Sonnet: ${mismatchPct}% of turns are high-context low-output`,
195
+ title: `Model change is not recommended: ${mismatchPct}% of Opus turns have large context and little output`,
196
196
  }
197
197
  : perModelSnapshot && sonnetEntries && sonnetEntries.length > 0
198
198
  ? { cap_attribution: "all_models_or_opus_binds" }
@@ -202,7 +202,7 @@ export const d4Detector = {
202
202
  category: "MODEL",
203
203
  scope_workspace_id: workspace_id,
204
204
  // Advisory gate (W0.3): which cap binds is NOT inferable from JSONL. Conditional lever.
205
- lever: "If your all-models / Opus / 5h cap is the one binding — check /usage — these high-context low-output Opus turns are Sonnet-movable. This does NOT help, and can hurt, if your Sonnet-specific weekly cap is the binding constraint.",
205
+ lever: "Check /usage first. If your overall, Opus, or 5-hour limit is filling, these large-context Opus turns with little output may be suitable for Sonnet. Do not switch if your Sonnet weekly limit is the one filling.",
206
206
  target_metric: "model_mix_opus_fraction",
207
207
  // Advisory gate: suppress the crisp $/wk headline until live /usage cap-attribution exists.
208
208
  modeled_savings_u_per_wk: null,
@@ -210,7 +210,7 @@ export const d4Detector = {
210
210
  // Destructure out result_usd_per_wk so the advisory formula carries no crisp $/wk figure.
211
211
  modeled_formula: (({ result_usd_per_wk: _, ...rest }) => ({ ...rest, kind: "ADVISORY" }))(formula),
212
212
  evidence: {
213
- title: `Route Opus→Sonnet: ${mismatchPct}% of turns are high-context low-output`,
213
+ title: `Review model choice: ${mismatchPct}% of Opus turns have large context and little output`,
214
214
  workspace_id,
215
215
  total_opus_turns_per_week: totalOpus,
216
216
  mismatch_turns_per_week: mismatchCount,
@@ -262,7 +262,7 @@ export const d6Detector = {
262
262
  modeled_savings_u_per_wk: modeledSavingsU,
263
263
  modeled_formula: formula,
264
264
  evidence: {
265
- title: `Trim tool output: ${Math.round(bloatShare * 100)}% bloat share in session`,
265
+ title: `Reduce large tool results: ${Math.round(bloatShare * 100)}% of this session's context`,
266
266
  session_id: row.session_id,
267
267
  workspace_id: row.workspace_id,
268
268
  tool_result_bytes: row.tool_result_bytes,
@@ -299,7 +299,7 @@ export const d7Detector = {
299
299
  expression: "cap-weighted exposure of turns owning repeat-excess events; not an avoidable-token or USD savings estimate",
300
300
  },
301
301
  evidence: {
302
- title: `Break retry loops: ${flaggedTurns.size} flagged turn${flaggedTurns.size === 1 ? "" : "s"} in session`,
302
+ title: `Stop repeated attempts: ${flaggedTurns.size} affected turn${flaggedTurns.size === 1 ? "" : "s"} in this session`,
303
303
  session_id: sessionId,
304
304
  workspace_id: session.workspaceId,
305
305
  loop_flagged_event_count: flaggedEventIds.size,
@@ -125,7 +125,7 @@ export const d8Detector = {
125
125
  // TTL-regime facet: is this session dominated by 5m-tier creation?
126
126
  const fiveMinCreation = events.reduce((s, e) => s + e.cache_write_5m, 0);
127
127
  const regime5m = totalCreation > 0 && fiveMinCreation / totalCreation >= D8_REGIME_5M_SHARE;
128
- const baseLever = "Use /clear (or resume-from-summary) before idling past the cache TTL, and batch prefix/CLAUDE.md edits to a session boundary so they don't invalidate the warm cache mid-session.";
128
+ const baseLever = "Before a long break, use /clear or resume from a summary. Make instruction-file changes between sessions so they do not rebuild the cache while you work.";
129
129
  const lever = regime5m
130
130
  ? `${baseLever} This session's creation is mostly 5m-tier — enable the 1h cache regime (ENABLE_PROMPT_CACHING_1H) where long pauses are unavoidable.`
131
131
  : baseLever;
@@ -138,7 +138,7 @@ export const d8Detector = {
138
138
  modeled_savings_u_per_wk: savingsU,
139
139
  modeled_formula: formula,
140
140
  evidence: {
141
- title: `Reduce cache-write churn: ${events.length} re-write${events.length === 1 ? "" : "s"} in session`,
141
+ title: `Avoid repeated cache rebuilds: ${events.length} after a long pause`,
142
142
  session_id: sessionId,
143
143
  workspace_id,
144
144
  churn_event_count: events.length,
@@ -62,7 +62,7 @@ export const d9Detector = {
62
62
  modeled_savings_u_per_wk: null,
63
63
  modeled_formula: d9Formula(sidechainCap, D9_UNPRODUCTIVE_FRACTION),
64
64
  evidence: {
65
- title: `Review background fan-out: ${Math.round(share * 100)}% cap-weighted sidechain`,
65
+ title: `Review background-agent work: ${Math.round(share * 100)}% of estimated limit use`,
66
66
  workspace_id: row.workspace_id,
67
67
  sidechain_cap_weighted_tokens: sidechainCap,
68
68
  total_cap_weighted_tokens: totalCap,
@@ -11,6 +11,7 @@
11
11
  * committed). The hook writes no transcript content to stdout/stderr.
12
12
  */
13
13
 
14
+ import { randomBytes } from "node:crypto";
14
15
  import * as fs from "node:fs";
15
16
  import * as os from "node:os";
16
17
  import * as path from "node:path";
@@ -19,9 +20,11 @@ import { fileURLToPath } from "node:url";
19
20
 
20
21
  const MAX_COUNT = 20;
21
22
  const MAX_BYTES = 500 * 1024 * 1024; // 500 MiB across all snapshots
23
+ const MAX_COLLISION_ATTEMPTS = 16;
22
24
 
23
- // Monotonic counter so two firings in the same millisecond get distinct filenames
24
- // that still sort in write order (the timestamp dominates; this breaks same-ms ties).
25
+ // A per-process counter preserves local write ordering. The random token prevents
26
+ // separately spawned hooks with the same session and millisecond from choosing the
27
+ // same name; exclusive creation below is the final no-overwrite guard.
25
28
  let snapshotCounter = 0;
26
29
 
27
30
  /** The directory snapshots are written to (overridable for tests). */
@@ -33,7 +36,8 @@ export function checkpointDir() {
33
36
  export function snapshotName(sessionId, now) {
34
37
  snapshotCounter = (snapshotCounter + 1) % 1_000_000;
35
38
  const stamp = now.toISOString().replace(/[:.]/g, "-");
36
- return `${encodeURIComponent(sessionId)}-${stamp}-${String(snapshotCounter).padStart(6, "0")}.jsonl`;
39
+ const counter = String(snapshotCounter).padStart(6, "0");
40
+ return `${encodeURIComponent(sessionId)}-${stamp}-${counter}-${randomBytes(6).toString("hex")}.jsonl`;
37
41
  }
38
42
 
39
43
  /**
@@ -43,14 +47,22 @@ export function snapshotName(sessionId, now) {
43
47
  export function writeCheckpoint(transcriptPath, sessionId, dir, now = new Date()) {
44
48
  if (typeof transcriptPath !== "string" || !fs.existsSync(transcriptPath)) return null;
45
49
  fs.mkdirSync(dir, { recursive: true });
46
- const dest = path.join(dir, snapshotName(sessionId, now));
47
- fs.copyFileSync(transcriptPath, dest);
48
- try {
49
- fs.chmodSync(dest, 0o600);
50
- } catch {
51
- // chmod is a no-op boundary on Windows; the user-profile ACL is the equivalent guard.
50
+ for (let attempt = 0; attempt < MAX_COLLISION_ATTEMPTS; attempt += 1) {
51
+ const dest = path.join(dir, snapshotName(sessionId, now));
52
+ try {
53
+ fs.copyFileSync(transcriptPath, dest, fs.constants.COPYFILE_EXCL);
54
+ } catch (error) {
55
+ if (error?.code === "EEXIST") continue;
56
+ throw error;
57
+ }
58
+ try {
59
+ fs.chmodSync(dest, 0o600);
60
+ } catch {
61
+ // chmod is a no-op boundary on Windows; the user-profile ACL is the equivalent guard.
62
+ }
63
+ return dest;
52
64
  }
53
- return dest;
65
+ return null;
54
66
  }
55
67
 
56
68
  /**
@@ -58,9 +70,10 @@ export function writeCheckpoint(transcriptPath, sessionId, dir, now = new Date()
58
70
  * not create. Malformed .jsonl files are deliberately left alone by retention.
59
71
  */
60
72
  function snapshotTimestamp(name) {
61
- const match = /-(\d{4})-(\d{2})-(\d{2})T(\d{2})-(\d{2})-(\d{2})-(\d{3})Z-\d{6}\.jsonl$/.exec(
62
- name,
63
- );
73
+ const match =
74
+ /-(\d{4})-(\d{2})-(\d{2})T(\d{2})-(\d{2})-(\d{2})-(\d{3})Z-\d{6}(?:-[a-f0-9]{12})?\.jsonl$/.exec(
75
+ name,
76
+ );
64
77
  if (!match) return null;
65
78
  const [, year, month, day, hour, minute, second, millisecond] = match;
66
79
  const timestamp = Date.parse(
@@ -21,6 +21,7 @@ export function getCostPerSuccess(db, workspaceId, from, to) {
21
21
  to,
22
22
  workspaceId,
23
23
  workspaceId,
24
+ workspaceId,
24
25
  from,
25
26
  to,
26
27
  workspaceId,
@@ -46,6 +47,14 @@ export function getCostPerSuccess(db, workspaceId, from, to) {
46
47
  JOIN merged_work_items mwi ON mwi.work_item_id = swl.work_item_id
47
48
  WHERE 1 = 1${sessionFilter}
48
49
  ),
50
+ merged_session_links AS (
51
+ SELECT swl.session_id, COUNT(DISTINCT swl.work_item_id) AS merged_pr_links
52
+ FROM session_work_links swl
53
+ JOIN merged_work_items mwi ON mwi.work_item_id = swl.work_item_id
54
+ JOIN sessions s ON s.session_id = swl.session_id
55
+ WHERE 1 = 1${sessionFilter}
56
+ GROUP BY swl.session_id
57
+ ),
49
58
  in_window_sessions AS (
50
59
  SELECT s.session_id, s.cost_equiv_u
51
60
  FROM sessions s
@@ -64,6 +73,8 @@ export function getCostPerSuccess(db, workspaceId, from, to) {
64
73
  SELECT
65
74
  (SELECT COUNT(*) FROM merged_work_items) AS merged_pr_count,
66
75
  (SELECT COUNT(*) FROM closed_unmerged_work_items) AS closed_unmerged_count,
76
+ (SELECT COUNT(*) FROM merged_session_links) AS unique_linked_session_count,
77
+ (SELECT COUNT(*) FROM merged_session_links WHERE merged_pr_links > 1) AS shared_linked_session_count,
67
78
  CASE WHEN (SELECT COUNT(*) FROM merged_work_items) = 0 THEN NULL
68
79
  ELSE (SELECT COALESCE(SUM(cost_equiv_u), 0) FROM merged_session_costs) * 1.0
69
80
  / (SELECT COUNT(*) FROM merged_work_items)
@@ -91,13 +102,14 @@ export function getCostPerSuccess(db, workspaceId, from, to) {
91
102
  };
92
103
  return buildResponse(data, {
93
104
  claim_kind: "OBS_PROXY",
105
+ metric_definition_version: "esf-1",
94
106
  n: data.merged_pr_count,
95
107
  window: { from, to },
96
108
  qualification: {
97
109
  provisional_excluded: false,
98
110
  unpriced_turns: 0,
99
111
  claim_kinds_count: 1,
100
- note: "Directional (OBS_PROXY): survivorship bias (heavy-spend sessions that never open a PR are invisible); reviewer-dependence (merge is a human decision, not a quality guarantee); linkage-coverage cap (only linkage_coverage_pct% of in-window sessions are linked to a PR, so unlinked spend is excluded). cost_per_merged_pr_u uses lifecycle attribution: each merged PR carries the full cost of every linked session whenever it ran, so narrowing the window changes the PR population, not the per-PR cost.",
112
+ note: "Directional (OBS_PROXY): Unique linked-session cost / merged PRs. The terminal-date PR cohort counts each linked session once, even when it links to multiple merged PRs; shared sessions are coverage, not per-PR allocation. Full linked-session lifecycle cost is used, while linkage_coverage_pct is a separate session-start coverage observation. Unknown pricing is not free. Survivorship bias and reviewer dependence remain.",
101
113
  },
102
114
  ...(workspaceId === null ? {} : { drilldown_ids: { workspace_id: workspaceId } }),
103
115
  });