agentwrangler 0.1.1-next.2e8c141 → 0.1.1-next.2f6c1df
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +82 -162
- package/dist/apply/jobs.js +6 -2
- package/dist/daemon/router.js +1 -1
- package/dist/detector/detectors/d10_catalog_footprint.js +2 -2
- package/dist/detector/detectors/d1_ctx_always_loaded.js +9 -9
- package/dist/detector/detectors/d2_session_long_full_context.js +2 -2
- package/dist/detector/detectors/d4_model_mismatch.js +3 -3
- package/dist/detector/detectors/d6_tool_result_bloat.js +1 -1
- package/dist/detector/detectors/d7_loop_retry_waste.js +1 -1
- package/dist/detector/detectors/d8_cache_write_churn.js +2 -2
- package/dist/detector/detectors/d9_idle_background_session.js +1 -1
- package/dist/hook/precompact-checkpoint-hook.mjs +26 -13
- package/dist/query/api/cost-per-success.js +13 -1
- package/dist/query/api/delivery.js +25 -3
- package/dist/query/api/effectiveness.js +30 -5
- package/dist/query/api/efficiency-headroom.js +12 -14
- package/dist/query/api/overview.js +77 -3
- package/dist/query/spend.js +64 -2
- package/dist/ui/assets/{BarChart-BwFZrcLn.js → BarChart-BC-ciS9d.js} +1 -1
- package/dist/ui/assets/BriefsPage-Dls6npt7.js +2 -0
- package/dist/ui/assets/{CacheWriteSpikesChart-zJIJWO41.js → CacheWriteSpikesChart-ncVwVoud.js} +1 -1
- package/dist/ui/assets/{CartesianChart-qApXl3Ak.js → CartesianChart-CRV6h2et.js} +1 -1
- package/dist/ui/assets/Chip-EizRm13x.js +1 -0
- package/dist/ui/assets/{ComposedChart-DYf1wymW.js → ComposedChart-CrOxp6Ym.js} +1 -1
- package/dist/ui/assets/{EmptyState-DL8gAVnJ.js → EmptyState-9kcoimjj.js} +1 -1
- package/dist/ui/assets/{FlavorDecomposition-CvJW78M9.js → FlavorDecomposition-DRZxH0vQ.js} +1 -1
- package/dist/ui/assets/FrictionCell-BGYB9UKh.js +2 -0
- package/dist/ui/assets/GlossaryPage-CRyQb1Hn.js +1 -0
- package/dist/ui/assets/HotSessionsPage-DfE90Kxi.js +1 -0
- package/dist/ui/assets/{InfoTip-Ckc9_LpJ.js → InfoTip-CmcYFNQt.js} +1 -1
- package/dist/ui/assets/{Legend-C_YMp0Rv.js → Legend-BCXsSaEM.js} +1 -1
- package/dist/ui/assets/{Line-DU4UvEBO.js → Line-2rUzs15v.js} +1 -1
- package/dist/ui/assets/OverviewPage-Bm5KvXqa.js +2 -0
- package/dist/ui/assets/RecommendationsPage-CvSl3lzp.js +2 -0
- package/dist/ui/assets/{Scatter-CNetVLuX.js → Scatter-DpIZt_y5.js} +1 -1
- package/dist/ui/assets/SessionDetailPage-w_lSgite.js +1 -0
- package/dist/ui/assets/SettingsPage-CfUoP8lB.js +25 -0
- package/dist/ui/assets/{Skeleton-BO9mXDXU.js → Skeleton-DjuxPm_D.js} +1 -1
- package/dist/ui/assets/SpendPercentileChip-DYun7Wzh.js +1 -0
- package/dist/ui/assets/{TrendChart-DKy9aH5G.js → TrendChart-gyhHK9H_.js} +1 -1
- package/dist/ui/assets/WorkspaceDetailPage-DacPKaKA.js +1 -0
- package/dist/ui/assets/WorkspacesPage-DiNQaAYG.js +1 -0
- package/dist/ui/assets/{chart-theme-BPPMjbVX.js → chart-theme-DnpNuyRt.js} +1 -1
- package/dist/ui/assets/{graphicalItemSelectors-DgKF1Dg2.js → graphicalItemSelectors-CFhpQjLz.js} +1 -1
- package/dist/ui/assets/{index-SrfNUeBZ.css → index-D_IJsvkJ.css} +1 -1
- package/dist/ui/assets/{index-CUyiomzU.js → index-hbaLYHUX.js} +3 -3
- package/dist/ui/assets/{prompt-templates-DhR4Qoy9.js → prompt-templates-CVzMex9L.js} +4 -4
- package/dist/ui/assets/rec-sessions-ggz9MYgP.js +1 -0
- package/dist/ui/assets/{useExperimentalActions-BeDSmwBu.js → useExperimentalActions-B7J8gIkb.js} +1 -1
- package/dist/ui/index.html +2 -2
- package/package.json +1 -1
- package/dist/ui/assets/BriefsPage-CtJwHLOh.js +0 -2
- package/dist/ui/assets/Chip-Cpt_fEc9.js +0 -1
- package/dist/ui/assets/FrictionCell-DciiSd3L.js +0 -2
- package/dist/ui/assets/GlossaryPage-BMkZGt2D.js +0 -1
- package/dist/ui/assets/HotSessionsPage-Ejn385bb.js +0 -1
- package/dist/ui/assets/OverviewPage-BcEKdWZj.js +0 -2
- package/dist/ui/assets/RecommendationsPage-CHfuNP_W.js +0 -2
- package/dist/ui/assets/SessionDetailPage-CYm5Gxo0.js +0 -1
- package/dist/ui/assets/SettingsPage-DjgGZj8N.js +0 -25
- package/dist/ui/assets/SpendPercentileChip-BvaZ8Lzi.js +0 -1
- package/dist/ui/assets/WorkspaceDetailPage-dutc4EpN.js +0 -1
- package/dist/ui/assets/WorkspacesPage-isSML2fh.js +0 -1
package/README.md
CHANGED
|
@@ -1,184 +1,104 @@
|
|
|
1
1
|
<p align="center">
|
|
2
|
-
<img src="docs/assets/logo.png" alt="AgentWrangler logo" width="
|
|
2
|
+
<img src="https://raw.githubusercontent.com/Doogit/AgentWrangler/main/docs/assets/logo.png" alt="AgentWrangler logo" width="200">
|
|
3
3
|
</p>
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
# AgentWrangler
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
and installable guardrails. Runs entirely on your machine.
|
|
11
|
-
</p>
|
|
12
|
-
|
|
13
|
-
<p align="center">
|
|
14
|
-
<a href="https://www.npmjs.com/package/agentwrangler"><img src="https://img.shields.io/npm/v/agentwrangler" alt="npm version"></a>
|
|
15
|
-
<a href="https://github.com/Doogit/AgentWrangler/actions/workflows/ci.yml"><img src="https://github.com/Doogit/AgentWrangler/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
|
16
|
-
<a href="https://img.shields.io/node/v/agentwrangler"><img src="https://img.shields.io/node/v/agentwrangler" alt="node version"></a>
|
|
17
|
-
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache--2.0-blue" alt="license"></a>
|
|
18
|
-
</p>
|
|
19
|
-
|
|
20
|
-

|
|
21
|
-
|
|
22
|
-
<!--
|
|
23
|
-
All screenshots and the demo GIF in this README are captured from a SANITIZED instance
|
|
24
|
-
(Vite test-mode fixtures — anonymized names, no live data) so no real workspace/repository
|
|
25
|
-
names or paths are committed (SEC-101). Regenerate against `npx vite --mode test`.
|
|
26
|
-
-->
|
|
27
|
-
|
|
28
|
-
## Why
|
|
29
|
-
|
|
30
|
-
Claude Code tells you almost nothing about where your token budget goes. Sessions balloon,
|
|
31
|
-
caches miss, background agents idle — and you find out when you hit the rate limit.
|
|
32
|
-
AgentWrangler reads the transcripts Claude Code already writes to your disk and answers three
|
|
33
|
-
questions:
|
|
34
|
-
|
|
35
|
-
- **Where did the tokens go?** Per-model, per-workspace, per-session spend with cache economics.
|
|
36
|
-
- **Did the work ship?** Sessions are linked to the pull requests they merged or closed.
|
|
37
|
-
- **What should I change?** Ranked waste-source detectors with modeled savings — and one-click
|
|
38
|
-
guardrail hooks that warn *inside* Claude Code before waste happens.
|
|
39
|
-
|
|
40
|
-
No cloud backend, no telemetry, no account. A daemon on `127.0.0.1`, a dashboard in your
|
|
41
|
-
browser, and a SQLite file in your home directory.
|
|
7
|
+
See where your Claude Code tokens go, inspect costly sessions, and track whether a change helped.
|
|
8
|
+
AgentWrangler reads local Claude Code transcripts and serves a dashboard on your machine.
|
|
9
|
+
No account or cloud backend is required.
|
|
42
10
|
|
|
43
11
|
## Quick start
|
|
44
12
|
|
|
13
|
+
Requires **Node 22-24 and npm**, plus Claude Code transcripts for populated charts.
|
|
14
|
+
|
|
45
15
|
```sh
|
|
46
16
|
npx agentwrangler@latest
|
|
47
17
|
```
|
|
48
18
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
More options (install from source, GitHub outcomes sync, environment variables):
|
|
54
|
-
**[Getting started →](docs/getting-started.md)**
|
|
55
|
-
|
|
56
|
-
## Features
|
|
57
|
-
|
|
58
|
-
### Overview — verdict first, details on demand
|
|
59
|
-
|
|
60
|
-
One screen answers "how bad is it this week": a spend verdict with trend, your top waste
|
|
61
|
-
source with a copyable fix prompt, live rate-limit gauges (5-hour and 7-day), a burn forecast
|
|
62
|
-
against your calibrated weekly limit, hot sessions, cache efficiency, and per-model
|
|
63
|
-
context-per-turn tiles.
|
|
64
|
-
|
|
65
|
-

|
|
66
|
-
|
|
67
|
-
### Recommendations — waste-source detectors, ranked by impact
|
|
68
|
-
|
|
69
|
-
Ten detector families watch your sessions for the patterns that actually burn tokens: cache
|
|
70
|
-
misses (the biggest single lever), session hygiene, retry/redundant-read loops, tool-result
|
|
71
|
-
bloat, model routing, idle background sessions, and more. Each recommendation shows modeled
|
|
72
|
-
weekly savings, a confidence tier, and a concrete action — install a hook, copy a config
|
|
73
|
-
snippet, or copy a guided prompt straight into Claude Code. Adopted changes flow into an
|
|
74
|
-
**impact ledger** that tracks the measured effect, and modeled savings are never counted as
|
|
75
|
-
achieved.
|
|
76
|
-
|
|
77
|
-

|
|
78
|
-
|
|
79
|
-
### Installable guardrails — local checks inside Claude Code, before the waste
|
|
80
|
-
|
|
81
|
-
Five small hooks you can install from the dashboard (directly, or via a copyable prompt that
|
|
82
|
-
Claude Code applies itself):
|
|
83
|
-
|
|
84
|
-
| Guardrail | What it does |
|
|
85
|
-
|---|---|
|
|
86
|
-
| **Context-budget warning** | Warns when a session's context crosses your soft/hard thresholds |
|
|
87
|
-
| **Loop guard** | Flags repeated identical tool failures before they spiral |
|
|
88
|
-
| **Burn alert** | Catches idle sessions still burning tokens in the background |
|
|
89
|
-
| **Pre-compaction checkpoint** | Copies the raw local transcript before an automatic compaction, subject to a local retention cap |
|
|
90
|
-
| **Dangerous-command guard** | Asks before risky shell commands and denies a small catastrophe list |
|
|
91
|
-
|
|
92
|
-
The context-budget and burn hooks warn. The loop guard warns before it denies repeated identical
|
|
93
|
-
failures, and the dangerous-command guard can ask or deny. Direct install enables all five hooks;
|
|
94
|
-
the copied install prompt enables the context-budget, loop, and burn hooks only. Thresholds are
|
|
95
|
-
tunable from Settings, and direct uninstall removes every AgentWrangler hook.
|
|
96
|
-
|
|
97
|
-
### Sessions — who spent it, and on what
|
|
98
|
-
|
|
99
|
-
The highest-cost sessions ranked with their output-to-context split, model, friction band
|
|
100
|
-
(API errors, tool failures, compactions, interrupts), and a "top X% by spend" self-percentile
|
|
101
|
-
chip. Drill into any session for a turn-by-turn timeline, its cost drivers (which detectors
|
|
102
|
-
fired and how hard), and a guided fix prompt built only from measured numbers.
|
|
103
|
-
|
|
104
|
-

|
|
105
|
-
|
|
106
|
-
<details>
|
|
107
|
-
<summary>Session detail view</summary>
|
|
108
|
-
|
|
109
|
-

|
|
110
|
-
</details>
|
|
111
|
-
|
|
112
|
-
### Workspaces — spend efficiency by repository
|
|
113
|
-
|
|
114
|
-
Every repo you run Claude Code in, with spend share, trend, context-per-turn, cache-write
|
|
115
|
-
share, Opus share, and $/turn. With a GitHub token configured, sessions are linked to the PRs
|
|
116
|
-
and commits they produced — so you can see cost-per-merged-PR, not just cost.
|
|
117
|
-
|
|
118
|
-

|
|
119
|
-
|
|
120
|
-
<details>
|
|
121
|
-
<summary>Workspace detail view</summary>
|
|
122
|
-
|
|
123
|
-

|
|
124
|
-
</details>
|
|
125
|
-
|
|
126
|
-
### Weekly brief — one page, three decisions
|
|
127
|
-
|
|
128
|
-
The week in one screen: spend verdict, what changed vs. last week, the top actions to take —
|
|
129
|
-
with a **Copy as Markdown** button so the whole brief drops into a standup note or a message.
|
|
19
|
+
Open **http://127.0.0.1:47821** if the browser does not open automatically. Keep the terminal
|
|
20
|
+
running; press **Ctrl+C** to stop. The first scan runs in the background. An empty history is
|
|
21
|
+
valid: use Claude Code, then return after ingestion. If a daemon is already running, stop it
|
|
22
|
+
before starting another on the same port.
|
|
130
23
|
|
|
131
|
-
|
|
24
|
+
[Install from source, configure scan roots, or troubleshoot](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md).
|
|
132
25
|
|
|
133
|
-
|
|
26
|
+
## Your first five minutes
|
|
134
27
|
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
28
|
+
1. **Check the scan.** In Overview, read the onboarding status. A completed scan with no
|
|
29
|
+
recommendations is a valid result. For an unexpected empty history, inspect scan roots
|
|
30
|
+
and parser health in Settings; saved scan-root changes require a daemon restart.
|
|
31
|
+
2. **Find one expensive session.** Select a date window, open Workspaces, choose a workspace,
|
|
32
|
+
then open one of its sessions. Compare context, cache writes, and output before deciding
|
|
33
|
+
what to change.
|
|
34
|
+
3. **Inspect one recommendation.** Open Recommendations and **Show details** on an instance.
|
|
35
|
+
Read its evidence and caveats. A modeled amount is a projection; directional advice may
|
|
36
|
+
have no dollar estimate. If nothing fires, there is nothing to adopt just to finish setup.
|
|
37
|
+
4. **Make one deliberate change.** A copied prompt is an artifact to review and run yourself.
|
|
38
|
+
It does not edit files. After completing a supported change, use **I completed the change**
|
|
39
|
+
when offered, then **Track this change** to record a baseline.
|
|
40
|
+
5. **Return to the Impact ledger.** Tracking starts observation, not a savings claim. Eligible
|
|
41
|
+
signals use a 14-day observation window and may finish inconclusive. You can finish this
|
|
42
|
+
first visit without a GitHub token, calibration, or installed hooks.
|
|
140
43
|
|
|
141
|
-
|
|
44
|
+
Dollar figures are **list-price equivalents**, not your subscription bill. Modeled savings
|
|
45
|
+
are not achieved savings, and observed improvement does not prove the change caused it.
|
|
46
|
+
GitHub linkage adds outcome metadata; calibration enables the burn forecast. Local spend,
|
|
47
|
+
session inspection, and supported recommendation tracking work without either.
|
|
142
48
|
|
|
143
|
-
-
|
|
144
|
-
- Most dashboard data is **aggregates, ids, counts, and structural anchors**. Local command text
|
|
145
|
-
and filesystem paths can also be retained in SQLite; treat the database as sensitive. The optional
|
|
146
|
-
direct-install PreCompact hook can separately copy full raw transcripts locally.
|
|
147
|
-
- The optional GitHub token is read locally, never logged, never persisted to the DB.
|
|
148
|
-
- Usage refreshes can call Anthropic with an existing Claude Code sign-in. GitHub outcomes sync
|
|
149
|
-
calls GitHub only when a token is configured; calibration and G2 judging are separate opt-ins.
|
|
49
|
+
[Worked example: trim always-loaded context and inspect its effect](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#example-trim-always-loaded-context).
|
|
150
50
|
|
|
151
|
-
|
|
51
|
+
## Dashboard preview
|
|
152
52
|
|
|
153
|
-
|
|
53
|
+

|
|
154
54
|
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
and [`.env.example`](.env.example).
|
|
55
|
+
This static preview uses synthetic data. [The dashboard tour](https://github.com/Doogit/AgentWrangler/blob/main/docs/dashboard-tour.md)
|
|
56
|
+
contains the other views and an optional animated preview.
|
|
158
57
|
|
|
159
|
-
##
|
|
58
|
+
## Pick a question
|
|
160
59
|
|
|
161
|
-
|
|
|
60
|
+
| I want to... | Start here |
|
|
162
61
|
|---|---|
|
|
163
|
-
| [
|
|
164
|
-
| [
|
|
165
|
-
| [
|
|
166
|
-
| [
|
|
167
|
-
|
|
|
168
|
-
| [
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
62
|
+
| Find where my tokens went | [Workspaces and Sessions](https://github.com/Doogit/AgentWrangler/blob/main/docs/dashboard-tour.md#workspaces) |
|
|
63
|
+
| Turn a recommendation into a change | [Worked example](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#example-trim-always-loaded-context) |
|
|
64
|
+
| Understand the dollars and verdicts | [Metric vocabulary](https://github.com/Doogit/AgentWrangler/blob/main/docs/dashboard-tour.md#glossary-how-to-read-this-dashboard) |
|
|
65
|
+
| Fix an empty dashboard | [First launch](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#first-launch) |
|
|
66
|
+
| Connect GitHub or calibrate limits | [Optional setup](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#optional-setup) |
|
|
67
|
+
| Read a weekly summary | [Briefs](https://github.com/Doogit/AgentWrangler/blob/main/docs/dashboard-tour.md#briefs) |
|
|
68
|
+
|
|
69
|
+
## Installable guardrails — local checks inside Claude Code, before the waste
|
|
70
|
+
|
|
71
|
+
Hooks are optional and require an explicit install. Settings **Install directly** installs
|
|
72
|
+
five hooks; **Copy install prompt** prepares instructions for three. Copying alone installs nothing.
|
|
73
|
+
|
|
74
|
+
| Hook | Behavior | Install path |
|
|
75
|
+
|---|---|---|
|
|
76
|
+
| Context-budget | Warns when context crosses configured thresholds | Direct or copied prompt |
|
|
77
|
+
| Loop guard | Warns, then can deny repeated identical failures | Direct or copied prompt |
|
|
78
|
+
| Burn alert | Warns about session budget consumption | Direct or copied prompt |
|
|
79
|
+
| Dangerous-command | Can ask or deny risky commands | Direct only |
|
|
80
|
+
| PreCompact checkpoint | Copies raw transcripts locally, with retention limits | Direct only |
|
|
81
|
+
|
|
82
|
+
[Installation, removal, and the checkpoint privacy details](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#in-session-guardrails).
|
|
83
|
+
|
|
84
|
+
## Privacy and limits
|
|
85
|
+
|
|
86
|
+
- The daemon binds to **127.0.0.1**. There is no telemetry or hosted product backend.
|
|
87
|
+
- SQLite contains aggregates and structural data, and can also contain local command text
|
|
88
|
+
and filesystem paths. Treat it as sensitive. The optional PreCompact hook makes separate
|
|
89
|
+
raw transcript copies on your machine.
|
|
90
|
+
- Usage refresh can contact Anthropic using your existing Claude Code sign-in. GitHub
|
|
91
|
+
outcomes sync requires a configured token. [Privacy details](https://github.com/Doogit/AgentWrangler/blob/main/docs/privacy.md).
|
|
92
|
+
- Claude Code format changes can require parser updates. This is a single-user tool.
|
|
93
|
+
- Windows local validation and Linux/macOS CI smoke coverage do not establish full native
|
|
94
|
+
accessibility or credential-store compatibility on every platform.
|
|
95
|
+
|
|
96
|
+
## Deeper documentation
|
|
97
|
+
|
|
98
|
+
[Configuration](https://github.com/Doogit/AgentWrangler/blob/main/docs/getting-started.md#configuration) |
|
|
99
|
+
[Architecture](https://github.com/Doogit/AgentWrangler/blob/main/docs/planning/AgentWrangler_Technical_Architecture_v4_5_0.md) |
|
|
100
|
+
[Data model and metrics](https://github.com/Doogit/AgentWrangler/blob/main/docs/planning/AgentWrangler_Data_Model_and_Metrics_v2.md) |
|
|
101
|
+
[Contributing](https://github.com/Doogit/AgentWrangler/blob/main/.github/CONTRIBUTING.md) |
|
|
102
|
+
[Security policy](https://github.com/Doogit/AgentWrangler/blob/main/.github/SECURITY.md)
|
|
103
|
+
|
|
104
|
+
[Apache 2.0](https://github.com/Doogit/AgentWrangler/blob/main/LICENSE).
|
package/dist/apply/jobs.js
CHANGED
|
@@ -200,6 +200,7 @@ function runClaudePhase(db, job, rec, phase, settingsPath, rt) {
|
|
|
200
200
|
const changedPaths = [];
|
|
201
201
|
let finalized = false;
|
|
202
202
|
let fatalExitMessage = null;
|
|
203
|
+
let killTimer;
|
|
203
204
|
const cleanup = () => {
|
|
204
205
|
fs.unlink(settingsPath, () => { });
|
|
205
206
|
};
|
|
@@ -251,9 +252,11 @@ function runClaudePhase(db, job, rec, phase, settingsPath, rt) {
|
|
|
251
252
|
}
|
|
252
253
|
});
|
|
253
254
|
const timer = setTimeout(() => {
|
|
255
|
+
// Keep the job exclusive until the child has actually closed. Publishing
|
|
256
|
+
// FAILED here lets a retry (or workspace cleanup) race the dying process.
|
|
257
|
+
fatalExitMessage ??= "job timed out";
|
|
254
258
|
proc.kill("SIGTERM");
|
|
255
|
-
setTimeout(() => proc.kill("SIGKILL"), 2000);
|
|
256
|
-
markFailed("job timed out");
|
|
259
|
+
killTimer = setTimeout(() => proc.kill("SIGKILL"), 2000);
|
|
257
260
|
}, rt.timeoutMs);
|
|
258
261
|
proc.on("error", (err) => {
|
|
259
262
|
clearTimeout(timer);
|
|
@@ -261,6 +264,7 @@ function runClaudePhase(db, job, rec, phase, settingsPath, rt) {
|
|
|
261
264
|
});
|
|
262
265
|
proc.on("close", (code) => {
|
|
263
266
|
clearTimeout(timer);
|
|
267
|
+
clearTimeout(killTimer);
|
|
264
268
|
if (finalized)
|
|
265
269
|
return;
|
|
266
270
|
finalized = true;
|
package/dist/daemon/router.js
CHANGED
|
@@ -264,7 +264,7 @@ export function handleApiRequest(_db, req, res, method, url) {
|
|
|
264
264
|
sendJson(res, 200, getPractices(_db, { from, to }));
|
|
265
265
|
return;
|
|
266
266
|
}
|
|
267
|
-
// GET /api/efficiency-headroom (
|
|
267
|
+
// GET /api/efficiency-headroom (individual weekly estimates; separate selected-window spend)
|
|
268
268
|
if (method === "GET" && pathname === "/api/efficiency-headroom") {
|
|
269
269
|
const { from, to } = resolveWindow(parseWindowFilter(url));
|
|
270
270
|
sendJson(res, 200, getEfficiencyHeadroom(_db, { from, to }));
|
|
@@ -95,7 +95,7 @@ export const d10Detector = {
|
|
|
95
95
|
const latestProbedAt = rows.reduce((latest, row) => (latest === null || row.probed_at > latest ? row.probed_at : latest), null);
|
|
96
96
|
const refs = rows.map((row) => row.file_ref);
|
|
97
97
|
const evidence = {
|
|
98
|
-
title: `Review
|
|
98
|
+
title: `Review tool catalog: ${Math.round(catalogTokens / 1000)}K estimated tokens`,
|
|
99
99
|
component: "MCP_SCHEMAS",
|
|
100
100
|
file_ref: refs.length === 1 ? refs[0] : null,
|
|
101
101
|
file_refs: refs,
|
|
@@ -120,7 +120,7 @@ export const d10Detector = {
|
|
|
120
120
|
scopeKey: "D10|global|MCP_SCHEMAS",
|
|
121
121
|
category: "TOOLING",
|
|
122
122
|
scope_workspace_id: null,
|
|
123
|
-
lever: "
|
|
123
|
+
lever: "The tool catalog exceeds its size target; actual loaded context is not measured",
|
|
124
124
|
target_metric: "catalog_context_tokens",
|
|
125
125
|
// R11 is required before catalog size can become a freed-headroom claim.
|
|
126
126
|
modeled_savings_u_per_wk: null,
|
|
@@ -29,16 +29,16 @@ function stepsFor(component, fileRef) {
|
|
|
29
29
|
case "CLAUDE_MD":
|
|
30
30
|
return [
|
|
31
31
|
`Open ${fileRef}`,
|
|
32
|
-
"Move
|
|
33
|
-
"Keep current
|
|
34
|
-
"
|
|
32
|
+
"Move changelogs, history, and background explanations to a linked document",
|
|
33
|
+
"Keep only current rules and links here",
|
|
34
|
+
"Check the size again after the next local scan",
|
|
35
35
|
];
|
|
36
36
|
case "MEMORY":
|
|
37
37
|
return [
|
|
38
38
|
`Review memory files under ${fileRef}`,
|
|
39
|
-
"
|
|
40
|
-
"
|
|
41
|
-
"
|
|
39
|
+
"Remove old or duplicate memories",
|
|
40
|
+
"Combine overlapping facts into concise entries",
|
|
41
|
+
"Check the size again after the next local scan",
|
|
42
42
|
];
|
|
43
43
|
case "MCP_SCHEMAS":
|
|
44
44
|
return [
|
|
@@ -62,13 +62,13 @@ function titleFor(component, tokens, target) {
|
|
|
62
62
|
function leverFor(component) {
|
|
63
63
|
switch (component) {
|
|
64
64
|
case "CLAUDE_MD":
|
|
65
|
-
return "Move
|
|
65
|
+
return "Move changelogs and history out of CLAUDE.md; keep current rules and links.";
|
|
66
66
|
case "MEMORY":
|
|
67
|
-
return "
|
|
67
|
+
return "Remove old or duplicate memories and combine overlapping facts.";
|
|
68
68
|
case "MCP_SCHEMAS":
|
|
69
69
|
return "Identify rarely-used skills/plugins; extract to on-demand or disable.";
|
|
70
70
|
default:
|
|
71
|
-
return "
|
|
71
|
+
return "Shorten this content that is loaded into every conversation.";
|
|
72
72
|
}
|
|
73
73
|
}
|
|
74
74
|
export const d1Detector = {
|
|
@@ -81,12 +81,12 @@ export const d2Detector = {
|
|
|
81
81
|
scopeKey: `D2|global|${formula.model}`,
|
|
82
82
|
category: "CONTEXT",
|
|
83
83
|
scope_workspace_id: null,
|
|
84
|
-
lever: "/clear between unrelated tasks
|
|
84
|
+
lever: "Use /clear between unrelated tasks, split long work into separate sessions, and avoid automatic compaction while you are working.",
|
|
85
85
|
target_metric: "avg_context_per_turn",
|
|
86
86
|
modeled_savings_u_per_wk: savingsU,
|
|
87
87
|
modeled_formula: formula,
|
|
88
88
|
evidence: {
|
|
89
|
-
title: `
|
|
89
|
+
title: `Split ${n} long session${n === 1 ? "" : "s"} this week`,
|
|
90
90
|
qualifying_session_count: qualifying.length,
|
|
91
91
|
qualifying_turn_count: qualifyingTurnCount,
|
|
92
92
|
session_ids: sessionIds,
|
|
@@ -192,7 +192,7 @@ export const d4Detector = {
|
|
|
192
192
|
? {
|
|
193
193
|
withheld: true,
|
|
194
194
|
withheld_reason: `Sonnet weekly cap is the binding constraint (Sonnet util ${bindingSonnet.utilization} >= all-models ${perModelSnapshot.seven_day_util}) — routing Opus->Sonnet would worsen it`,
|
|
195
|
-
title: `
|
|
195
|
+
title: `Model change is not recommended: ${mismatchPct}% of Opus turns have large context and little output`,
|
|
196
196
|
}
|
|
197
197
|
: perModelSnapshot && sonnetEntries && sonnetEntries.length > 0
|
|
198
198
|
? { cap_attribution: "all_models_or_opus_binds" }
|
|
@@ -202,7 +202,7 @@ export const d4Detector = {
|
|
|
202
202
|
category: "MODEL",
|
|
203
203
|
scope_workspace_id: workspace_id,
|
|
204
204
|
// Advisory gate (W0.3): which cap binds is NOT inferable from JSONL. Conditional lever.
|
|
205
|
-
lever: "If your
|
|
205
|
+
lever: "Check /usage first. If your overall, Opus, or 5-hour limit is filling, these large-context Opus turns with little output may be suitable for Sonnet. Do not switch if your Sonnet weekly limit is the one filling.",
|
|
206
206
|
target_metric: "model_mix_opus_fraction",
|
|
207
207
|
// Advisory gate: suppress the crisp $/wk headline until live /usage cap-attribution exists.
|
|
208
208
|
modeled_savings_u_per_wk: null,
|
|
@@ -210,7 +210,7 @@ export const d4Detector = {
|
|
|
210
210
|
// Destructure out result_usd_per_wk so the advisory formula carries no crisp $/wk figure.
|
|
211
211
|
modeled_formula: (({ result_usd_per_wk: _, ...rest }) => ({ ...rest, kind: "ADVISORY" }))(formula),
|
|
212
212
|
evidence: {
|
|
213
|
-
title: `
|
|
213
|
+
title: `Review model choice: ${mismatchPct}% of Opus turns have large context and little output`,
|
|
214
214
|
workspace_id,
|
|
215
215
|
total_opus_turns_per_week: totalOpus,
|
|
216
216
|
mismatch_turns_per_week: mismatchCount,
|
|
@@ -262,7 +262,7 @@ export const d6Detector = {
|
|
|
262
262
|
modeled_savings_u_per_wk: modeledSavingsU,
|
|
263
263
|
modeled_formula: formula,
|
|
264
264
|
evidence: {
|
|
265
|
-
title: `
|
|
265
|
+
title: `Reduce large tool results: ${Math.round(bloatShare * 100)}% of this session's context`,
|
|
266
266
|
session_id: row.session_id,
|
|
267
267
|
workspace_id: row.workspace_id,
|
|
268
268
|
tool_result_bytes: row.tool_result_bytes,
|
|
@@ -299,7 +299,7 @@ export const d7Detector = {
|
|
|
299
299
|
expression: "cap-weighted exposure of turns owning repeat-excess events; not an avoidable-token or USD savings estimate",
|
|
300
300
|
},
|
|
301
301
|
evidence: {
|
|
302
|
-
title: `
|
|
302
|
+
title: `Stop repeated attempts: ${flaggedTurns.size} affected turn${flaggedTurns.size === 1 ? "" : "s"} in this session`,
|
|
303
303
|
session_id: sessionId,
|
|
304
304
|
workspace_id: session.workspaceId,
|
|
305
305
|
loop_flagged_event_count: flaggedEventIds.size,
|
|
@@ -125,7 +125,7 @@ export const d8Detector = {
|
|
|
125
125
|
// TTL-regime facet: is this session dominated by 5m-tier creation?
|
|
126
126
|
const fiveMinCreation = events.reduce((s, e) => s + e.cache_write_5m, 0);
|
|
127
127
|
const regime5m = totalCreation > 0 && fiveMinCreation / totalCreation >= D8_REGIME_5M_SHARE;
|
|
128
|
-
const baseLever = "
|
|
128
|
+
const baseLever = "Before a long break, use /clear or resume from a summary. Make instruction-file changes between sessions so they do not rebuild the cache while you work.";
|
|
129
129
|
const lever = regime5m
|
|
130
130
|
? `${baseLever} This session's creation is mostly 5m-tier — enable the 1h cache regime (ENABLE_PROMPT_CACHING_1H) where long pauses are unavoidable.`
|
|
131
131
|
: baseLever;
|
|
@@ -138,7 +138,7 @@ export const d8Detector = {
|
|
|
138
138
|
modeled_savings_u_per_wk: savingsU,
|
|
139
139
|
modeled_formula: formula,
|
|
140
140
|
evidence: {
|
|
141
|
-
title: `
|
|
141
|
+
title: `Avoid repeated cache rebuilds: ${events.length} after a long pause`,
|
|
142
142
|
session_id: sessionId,
|
|
143
143
|
workspace_id,
|
|
144
144
|
churn_event_count: events.length,
|
|
@@ -62,7 +62,7 @@ export const d9Detector = {
|
|
|
62
62
|
modeled_savings_u_per_wk: null,
|
|
63
63
|
modeled_formula: d9Formula(sidechainCap, D9_UNPRODUCTIVE_FRACTION),
|
|
64
64
|
evidence: {
|
|
65
|
-
title: `Review background
|
|
65
|
+
title: `Review background-agent work: ${Math.round(share * 100)}% of estimated limit use`,
|
|
66
66
|
workspace_id: row.workspace_id,
|
|
67
67
|
sidechain_cap_weighted_tokens: sidechainCap,
|
|
68
68
|
total_cap_weighted_tokens: totalCap,
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
* committed). The hook writes no transcript content to stdout/stderr.
|
|
12
12
|
*/
|
|
13
13
|
|
|
14
|
+
import { randomBytes } from "node:crypto";
|
|
14
15
|
import * as fs from "node:fs";
|
|
15
16
|
import * as os from "node:os";
|
|
16
17
|
import * as path from "node:path";
|
|
@@ -19,9 +20,11 @@ import { fileURLToPath } from "node:url";
|
|
|
19
20
|
|
|
20
21
|
const MAX_COUNT = 20;
|
|
21
22
|
const MAX_BYTES = 500 * 1024 * 1024; // 500 MiB across all snapshots
|
|
23
|
+
const MAX_COLLISION_ATTEMPTS = 16;
|
|
22
24
|
|
|
23
|
-
//
|
|
24
|
-
//
|
|
25
|
+
// A per-process counter preserves local write ordering. The random token prevents
|
|
26
|
+
// separately spawned hooks with the same session and millisecond from choosing the
|
|
27
|
+
// same name; exclusive creation below is the final no-overwrite guard.
|
|
25
28
|
let snapshotCounter = 0;
|
|
26
29
|
|
|
27
30
|
/** The directory snapshots are written to (overridable for tests). */
|
|
@@ -33,7 +36,8 @@ export function checkpointDir() {
|
|
|
33
36
|
export function snapshotName(sessionId, now) {
|
|
34
37
|
snapshotCounter = (snapshotCounter + 1) % 1_000_000;
|
|
35
38
|
const stamp = now.toISOString().replace(/[:.]/g, "-");
|
|
36
|
-
|
|
39
|
+
const counter = String(snapshotCounter).padStart(6, "0");
|
|
40
|
+
return `${encodeURIComponent(sessionId)}-${stamp}-${counter}-${randomBytes(6).toString("hex")}.jsonl`;
|
|
37
41
|
}
|
|
38
42
|
|
|
39
43
|
/**
|
|
@@ -43,14 +47,22 @@ export function snapshotName(sessionId, now) {
|
|
|
43
47
|
export function writeCheckpoint(transcriptPath, sessionId, dir, now = new Date()) {
|
|
44
48
|
if (typeof transcriptPath !== "string" || !fs.existsSync(transcriptPath)) return null;
|
|
45
49
|
fs.mkdirSync(dir, { recursive: true });
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
50
|
+
for (let attempt = 0; attempt < MAX_COLLISION_ATTEMPTS; attempt += 1) {
|
|
51
|
+
const dest = path.join(dir, snapshotName(sessionId, now));
|
|
52
|
+
try {
|
|
53
|
+
fs.copyFileSync(transcriptPath, dest, fs.constants.COPYFILE_EXCL);
|
|
54
|
+
} catch (error) {
|
|
55
|
+
if (error?.code === "EEXIST") continue;
|
|
56
|
+
throw error;
|
|
57
|
+
}
|
|
58
|
+
try {
|
|
59
|
+
fs.chmodSync(dest, 0o600);
|
|
60
|
+
} catch {
|
|
61
|
+
// chmod is a no-op boundary on Windows; the user-profile ACL is the equivalent guard.
|
|
62
|
+
}
|
|
63
|
+
return dest;
|
|
52
64
|
}
|
|
53
|
-
return
|
|
65
|
+
return null;
|
|
54
66
|
}
|
|
55
67
|
|
|
56
68
|
/**
|
|
@@ -58,9 +70,10 @@ export function writeCheckpoint(transcriptPath, sessionId, dir, now = new Date()
|
|
|
58
70
|
* not create. Malformed .jsonl files are deliberately left alone by retention.
|
|
59
71
|
*/
|
|
60
72
|
function snapshotTimestamp(name) {
|
|
61
|
-
const match =
|
|
62
|
-
|
|
63
|
-
|
|
73
|
+
const match =
|
|
74
|
+
/-(\d{4})-(\d{2})-(\d{2})T(\d{2})-(\d{2})-(\d{2})-(\d{3})Z-\d{6}(?:-[a-f0-9]{12})?\.jsonl$/.exec(
|
|
75
|
+
name,
|
|
76
|
+
);
|
|
64
77
|
if (!match) return null;
|
|
65
78
|
const [, year, month, day, hour, minute, second, millisecond] = match;
|
|
66
79
|
const timestamp = Date.parse(
|
|
@@ -21,6 +21,7 @@ export function getCostPerSuccess(db, workspaceId, from, to) {
|
|
|
21
21
|
to,
|
|
22
22
|
workspaceId,
|
|
23
23
|
workspaceId,
|
|
24
|
+
workspaceId,
|
|
24
25
|
from,
|
|
25
26
|
to,
|
|
26
27
|
workspaceId,
|
|
@@ -46,6 +47,14 @@ export function getCostPerSuccess(db, workspaceId, from, to) {
|
|
|
46
47
|
JOIN merged_work_items mwi ON mwi.work_item_id = swl.work_item_id
|
|
47
48
|
WHERE 1 = 1${sessionFilter}
|
|
48
49
|
),
|
|
50
|
+
merged_session_links AS (
|
|
51
|
+
SELECT swl.session_id, COUNT(DISTINCT swl.work_item_id) AS merged_pr_links
|
|
52
|
+
FROM session_work_links swl
|
|
53
|
+
JOIN merged_work_items mwi ON mwi.work_item_id = swl.work_item_id
|
|
54
|
+
JOIN sessions s ON s.session_id = swl.session_id
|
|
55
|
+
WHERE 1 = 1${sessionFilter}
|
|
56
|
+
GROUP BY swl.session_id
|
|
57
|
+
),
|
|
49
58
|
in_window_sessions AS (
|
|
50
59
|
SELECT s.session_id, s.cost_equiv_u
|
|
51
60
|
FROM sessions s
|
|
@@ -64,6 +73,8 @@ export function getCostPerSuccess(db, workspaceId, from, to) {
|
|
|
64
73
|
SELECT
|
|
65
74
|
(SELECT COUNT(*) FROM merged_work_items) AS merged_pr_count,
|
|
66
75
|
(SELECT COUNT(*) FROM closed_unmerged_work_items) AS closed_unmerged_count,
|
|
76
|
+
(SELECT COUNT(*) FROM merged_session_links) AS unique_linked_session_count,
|
|
77
|
+
(SELECT COUNT(*) FROM merged_session_links WHERE merged_pr_links > 1) AS shared_linked_session_count,
|
|
67
78
|
CASE WHEN (SELECT COUNT(*) FROM merged_work_items) = 0 THEN NULL
|
|
68
79
|
ELSE (SELECT COALESCE(SUM(cost_equiv_u), 0) FROM merged_session_costs) * 1.0
|
|
69
80
|
/ (SELECT COUNT(*) FROM merged_work_items)
|
|
@@ -91,13 +102,14 @@ export function getCostPerSuccess(db, workspaceId, from, to) {
|
|
|
91
102
|
};
|
|
92
103
|
return buildResponse(data, {
|
|
93
104
|
claim_kind: "OBS_PROXY",
|
|
105
|
+
metric_definition_version: "esf-1",
|
|
94
106
|
n: data.merged_pr_count,
|
|
95
107
|
window: { from, to },
|
|
96
108
|
qualification: {
|
|
97
109
|
provisional_excluded: false,
|
|
98
110
|
unpriced_turns: 0,
|
|
99
111
|
claim_kinds_count: 1,
|
|
100
|
-
note: "Directional (OBS_PROXY):
|
|
112
|
+
note: "Directional (OBS_PROXY): Unique linked-session cost / merged PRs. The terminal-date PR cohort counts each linked session once, even when it links to multiple merged PRs; shared sessions are coverage, not per-PR allocation. Full linked-session lifecycle cost is used, while linkage_coverage_pct is a separate session-start coverage observation. Unknown pricing is not free. Survivorship bias and reviewer dependence remain.",
|
|
101
113
|
},
|
|
102
114
|
...(workspaceId === null ? {} : { drilldown_ids: { workspace_id: workspaceId } }),
|
|
103
115
|
});
|