context-doctor 0.13.5 → 0.14.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +75 -3
- package/dist/calibration.d.ts +28 -0
- package/dist/calibration.js +75 -0
- package/dist/cli.js +71 -4
- package/dist/config.d.ts +2 -0
- package/dist/config.js +2 -2
- package/dist/doctor.js +19 -0
- package/dist/experiment.d.ts +58 -0
- package/dist/experiment.js +214 -0
- package/dist/install.d.ts +10 -1
- package/dist/install.js +67 -4
- package/dist/mcp.js +1 -1
- package/dist/optimize.d.ts +9 -0
- package/dist/optimize.js +53 -7
- package/dist/profile.d.ts +8 -0
- package/dist/profile.js +5 -1
- package/dist/proxy.js +39 -2
- package/dist/report.js +5 -0
- package/dist/session.d.ts +12 -0
- package/dist/session.js +1 -1
- package/dist/statusline.d.ts +44 -0
- package/dist/statusline.js +104 -0
- package/dist/subagents.d.ts +50 -0
- package/dist/subagents.js +175 -0
- package/package.json +2 -2
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `context-doctor experiment` — the fresh-vs-existing session harness.
|
|
3
|
+
*
|
|
4
|
+
* Everything else in this tool measures what is IN the context. None of it can
|
|
5
|
+
* say whether the task succeeded, so a smaller transcript can be a cheaper
|
|
6
|
+
* failure. This runs the same task twice against the same starting commit —
|
|
7
|
+
* once in a fresh session, once forked from an existing one — with the same
|
|
8
|
+
* model and tools, records what each was billed and how long it took, runs the
|
|
9
|
+
* same check command against each result, and puts the two side by side.
|
|
10
|
+
*
|
|
11
|
+
* It spends the user's Claude budget, so it refuses to start on a dirty tree,
|
|
12
|
+
* caps spend per arm, forks the existing session rather than mutating it, and
|
|
13
|
+
* resets the tree between arms only because it verified the tree was clean.
|
|
14
|
+
*/
|
|
15
|
+
import { execFileSync, spawnSync } from "node:child_process";
|
|
16
|
+
import { existsSync } from "node:fs";
|
|
17
|
+
import { homedir } from "node:os";
|
|
18
|
+
import { join } from "node:path";
|
|
19
|
+
import { parseSessionFile } from "./session.js";
|
|
20
|
+
import { formatTokens } from "./tokens.js";
|
|
21
|
+
import { formatUsd } from "./pricing.js";
|
|
22
|
+
/** Where a test can point at a stub instead of the real CLI. */
|
|
23
|
+
function claudeBinary() {
|
|
24
|
+
return process.env.CONTEXT_DOCTOR_CLAUDE_BIN ?? "claude";
|
|
25
|
+
}
|
|
26
|
+
function git(cwd, ...args) {
|
|
27
|
+
return execFileSync("git", args, { cwd, encoding: "utf8", stdio: ["ignore", "pipe", "ignore"] }).trim();
|
|
28
|
+
}
|
|
29
|
+
function num(v) {
|
|
30
|
+
return typeof v === "number" && Number.isFinite(v) ? v : 0;
|
|
31
|
+
}
|
|
32
|
+
/** Build the argv for one arm — exported so the dry run and the tests see exactly what runs. */
|
|
33
|
+
export function claudeArgs(opts, arm) {
|
|
34
|
+
const args = ["-p", opts.task, "--output-format", "json", "--permission-mode", "acceptEdits"];
|
|
35
|
+
if (opts.model)
|
|
36
|
+
args.push("--model", opts.model);
|
|
37
|
+
args.push("--max-budget-usd", String(opts.budgetUsd ?? 1));
|
|
38
|
+
if (arm === "existing" && opts.existing)
|
|
39
|
+
args.push("--resume", opts.existing, "--fork-session");
|
|
40
|
+
return args;
|
|
41
|
+
}
|
|
42
|
+
function findTranscript(sessionId, cwd) {
|
|
43
|
+
// Claude Code keys project dirs by the cwd with separators replaced.
|
|
44
|
+
const projectDir = join(homedir(), ".claude", "projects", cwd.replace(/[\\/:]/g, "-"));
|
|
45
|
+
const candidate = join(projectDir, `${sessionId}.jsonl`);
|
|
46
|
+
return existsSync(candidate) ? candidate : undefined;
|
|
47
|
+
}
|
|
48
|
+
function runArm(opts, arm, cwd) {
|
|
49
|
+
const started = Date.now();
|
|
50
|
+
const proc = spawnSync(claudeBinary(), claudeArgs(opts, arm), {
|
|
51
|
+
cwd,
|
|
52
|
+
encoding: "utf8",
|
|
53
|
+
env: { ...process.env, CLAUDECODE: undefined },
|
|
54
|
+
maxBuffer: 64 * 1024 * 1024,
|
|
55
|
+
});
|
|
56
|
+
const base = {
|
|
57
|
+
arm, costUsd: 0, durationMs: Date.now() - started, turns: 0,
|
|
58
|
+
inputTokens: 0, cacheRead: 0, cacheWrite: 0, outputTokens: 0,
|
|
59
|
+
};
|
|
60
|
+
if (proc.error)
|
|
61
|
+
return { ...base, error: `could not run ${claudeBinary()}: ${proc.error.message}` };
|
|
62
|
+
let parsed;
|
|
63
|
+
try {
|
|
64
|
+
parsed = JSON.parse(proc.stdout);
|
|
65
|
+
}
|
|
66
|
+
catch {
|
|
67
|
+
return { ...base, error: `claude did not return JSON (exit ${proc.status}): ${(proc.stderr || proc.stdout).slice(0, 300)}` };
|
|
68
|
+
}
|
|
69
|
+
const usage = parsed.usage ?? {};
|
|
70
|
+
const result = {
|
|
71
|
+
...base,
|
|
72
|
+
sessionId: parsed.session_id,
|
|
73
|
+
costUsd: num(parsed.total_cost_usd),
|
|
74
|
+
durationMs: num(parsed.duration_ms) || base.durationMs,
|
|
75
|
+
turns: num(parsed.num_turns),
|
|
76
|
+
inputTokens: num(usage.input_tokens),
|
|
77
|
+
cacheRead: num(usage.cache_read_input_tokens),
|
|
78
|
+
cacheWrite: num(usage.cache_creation_input_tokens),
|
|
79
|
+
outputTokens: num(usage.output_tokens),
|
|
80
|
+
error: parsed.is_error ? `claude reported an error: ${String(parsed.result ?? "").slice(0, 300)}` : undefined,
|
|
81
|
+
};
|
|
82
|
+
if (parsed.session_id) {
|
|
83
|
+
const transcript = findTranscript(parsed.session_id, cwd);
|
|
84
|
+
if (transcript) {
|
|
85
|
+
try {
|
|
86
|
+
result.liveContextTokens = parseSessionFile(transcript).reportedInputTokens;
|
|
87
|
+
}
|
|
88
|
+
catch {
|
|
89
|
+
/* the numbers above stand on their own */
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
if (opts.check) {
|
|
94
|
+
const check = spawnSync(opts.check, { cwd, shell: true, encoding: "utf8", stdio: ["ignore", "pipe", "pipe"] });
|
|
95
|
+
result.check = { command: opts.check, passed: check.status === 0, exitCode: check.status ?? -1 };
|
|
96
|
+
}
|
|
97
|
+
try {
|
|
98
|
+
result.diffStat = git(cwd, "diff", "--stat") || "(no changes)";
|
|
99
|
+
}
|
|
100
|
+
catch {
|
|
101
|
+
/* not a git repo after all; the arm still ran */
|
|
102
|
+
}
|
|
103
|
+
return result;
|
|
104
|
+
}
|
|
105
|
+
export function runExperiment(opts) {
|
|
106
|
+
const cwd = opts.cwd ?? process.cwd();
|
|
107
|
+
if (process.env.CLAUDECODE && !opts.dryRun) {
|
|
108
|
+
return { arms: [], refused: "This runs the claude CLI, which refuses to start inside another Claude Code session. Run the experiment from a regular terminal." };
|
|
109
|
+
}
|
|
110
|
+
let startCommit;
|
|
111
|
+
let clean = false;
|
|
112
|
+
try {
|
|
113
|
+
startCommit = git(cwd, "rev-parse", "HEAD");
|
|
114
|
+
clean = git(cwd, "status", "--porcelain") === "";
|
|
115
|
+
}
|
|
116
|
+
catch {
|
|
117
|
+
return { arms: [], refused: `${cwd} is not a git repository. The experiment needs a commit to reset to between arms.` };
|
|
118
|
+
}
|
|
119
|
+
// A dry run touches nothing, so a dirty tree only needs mentioning.
|
|
120
|
+
if (opts.dryRun)
|
|
121
|
+
return { arms: [], startCommit, treeClean: clean };
|
|
122
|
+
if (!clean && !opts.allowDirty) {
|
|
123
|
+
return { arms: [], refused: "Working tree has uncommitted changes. Both arms must start from the same commit, and the tree is reset between them — commit or stash first (or pass --allow-dirty to accept losing those changes)." };
|
|
124
|
+
}
|
|
125
|
+
const arms = [];
|
|
126
|
+
const reset = () => {
|
|
127
|
+
// Safe only because the tree was verified clean (or the user opted in).
|
|
128
|
+
git(cwd, "reset", "--hard", startCommit);
|
|
129
|
+
git(cwd, "clean", "-fd");
|
|
130
|
+
};
|
|
131
|
+
arms.push(runArm(opts, "fresh", cwd));
|
|
132
|
+
reset();
|
|
133
|
+
if (opts.existing) {
|
|
134
|
+
arms.push(runArm(opts, "existing", cwd));
|
|
135
|
+
reset();
|
|
136
|
+
}
|
|
137
|
+
return { arms, startCommit };
|
|
138
|
+
}
|
|
139
|
+
export function renderExperiment(result, opts) {
|
|
140
|
+
const lines = [];
|
|
141
|
+
lines.push("CONTEXT DOCTOR — fresh vs existing session");
|
|
142
|
+
lines.push("═".repeat(56));
|
|
143
|
+
if (result.refused) {
|
|
144
|
+
lines.push(`Refused: ${result.refused}`);
|
|
145
|
+
return lines.join("\n");
|
|
146
|
+
}
|
|
147
|
+
lines.push(`task: ${opts.task}`);
|
|
148
|
+
if (result.startCommit)
|
|
149
|
+
lines.push(`commit: ${result.startCommit.slice(0, 12)} (both arms start here)`);
|
|
150
|
+
if (opts.model)
|
|
151
|
+
lines.push(`model: ${opts.model}`);
|
|
152
|
+
if (opts.check)
|
|
153
|
+
lines.push(`check: ${opts.check}`);
|
|
154
|
+
lines.push(`budget: ${formatUsd(opts.budgetUsd ?? 1)} per arm`);
|
|
155
|
+
if (opts.dryRun) {
|
|
156
|
+
lines.push("");
|
|
157
|
+
lines.push("Dry run. Commands that would execute:");
|
|
158
|
+
lines.push(` fresh: claude ${claudeArgs(opts, "fresh").map((a) => (a.includes(" ") ? JSON.stringify(a) : a)).join(" ")}`);
|
|
159
|
+
if (opts.existing)
|
|
160
|
+
lines.push(` existing: claude ${claudeArgs(opts, "existing").map((a) => (a.includes(" ") ? JSON.stringify(a) : a)).join(" ")}`);
|
|
161
|
+
lines.push(" (tree is reset to the start commit after each arm)");
|
|
162
|
+
if (result.treeClean === false)
|
|
163
|
+
lines.push(" NOTE: the tree is dirty right now; a real run would refuse unless you commit, stash, or pass --allow-dirty.");
|
|
164
|
+
return lines.join("\n");
|
|
165
|
+
}
|
|
166
|
+
lines.push("");
|
|
167
|
+
const col = (s, w = 14) => s.padStart(w);
|
|
168
|
+
const header = `${"".padEnd(22)}${col("fresh")}${opts.existing ? col("existing") : ""}`;
|
|
169
|
+
lines.push(header);
|
|
170
|
+
lines.push("─".repeat(header.length));
|
|
171
|
+
const row = (label, pick) => {
|
|
172
|
+
lines.push(`${label.padEnd(22)}${result.arms.map((a) => col(pick(a))).join("")}`);
|
|
173
|
+
};
|
|
174
|
+
row("billed input", (a) => formatTokens(a.inputTokens + a.cacheRead + a.cacheWrite));
|
|
175
|
+
row(" of which cache read", (a) => formatTokens(a.cacheRead));
|
|
176
|
+
row(" of which cache write", (a) => formatTokens(a.cacheWrite));
|
|
177
|
+
row("output tokens", (a) => formatTokens(a.outputTokens));
|
|
178
|
+
row("cost", (a) => formatUsd(a.costUsd));
|
|
179
|
+
row("wall clock", (a) => `${(a.durationMs / 1000).toFixed(0)}s`);
|
|
180
|
+
row("turns", (a) => String(a.turns));
|
|
181
|
+
if (result.arms.some((a) => a.liveContextTokens))
|
|
182
|
+
row("live context (last)", (a) => (a.liveContextTokens ? formatTokens(a.liveContextTokens) : "—"));
|
|
183
|
+
if (opts.check)
|
|
184
|
+
row("check", (a) => (a.check ? (a.check.passed ? "PASS" : `FAIL (${a.check.exitCode})`) : "—"));
|
|
185
|
+
lines.push("");
|
|
186
|
+
for (const a of result.arms) {
|
|
187
|
+
if (a.error)
|
|
188
|
+
lines.push(`${a.arm}: ${a.error}`);
|
|
189
|
+
if (a.diffStat && a.diffStat !== "(no changes)")
|
|
190
|
+
lines.push(`${a.arm} changed:\n${a.diffStat.split("\n").map((l) => " " + l).join("\n")}`);
|
|
191
|
+
}
|
|
192
|
+
// The verdict is the whole point: cheaper is only better if it also passed.
|
|
193
|
+
if (opts.existing && result.arms.length === 2 && opts.check) {
|
|
194
|
+
const [fresh, existing] = result.arms;
|
|
195
|
+
const cheaper = fresh.costUsd <= existing.costUsd ? fresh : existing;
|
|
196
|
+
const other = cheaper === fresh ? existing : fresh;
|
|
197
|
+
if (cheaper.check?.passed && !other.check?.passed) {
|
|
198
|
+
lines.push(`Verdict: ${cheaper.arm} was cheaper AND passed; ${other.arm} failed the check. Clear win for ${cheaper.arm}.`);
|
|
199
|
+
}
|
|
200
|
+
else if (cheaper.check?.passed && other.check?.passed) {
|
|
201
|
+
lines.push(`Verdict: both passed; ${cheaper.arm} was cheaper by ${formatUsd(Math.abs(fresh.costUsd - existing.costUsd))}.`);
|
|
202
|
+
}
|
|
203
|
+
else if (!cheaper.check?.passed && other.check?.passed) {
|
|
204
|
+
lines.push(`Verdict: ${cheaper.arm} was cheaper but FAILED the check. A smaller bill for a wrong answer is not a saving; ${other.arm} wins.`);
|
|
205
|
+
}
|
|
206
|
+
else {
|
|
207
|
+
lines.push("Verdict: neither arm passed the check. Cost comparison is moot until one does.");
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
else if (opts.check) {
|
|
211
|
+
lines.push("One arm only. Pass --existing <session-id> to compare against a forked existing session.");
|
|
212
|
+
}
|
|
213
|
+
return lines.join("\n");
|
|
214
|
+
}
|
package/dist/install.d.ts
CHANGED
|
@@ -20,6 +20,13 @@ export declare function npxLauncher(platformName: string): {
|
|
|
20
20
|
command: string;
|
|
21
21
|
args: string[];
|
|
22
22
|
};
|
|
23
|
+
/**
|
|
24
|
+
* Claude Code's status bar: a `statusLine` command whose stdout is shown while
|
|
25
|
+
* the user types. Opt-in, because there is only one status line and it may
|
|
26
|
+
* already be someone's own — this never overwrites a statusLine that is not
|
|
27
|
+
* ours. Returns what happened so install can print the truth.
|
|
28
|
+
*/
|
|
29
|
+
export declare function installStatusLine(): "installed" | "already" | "kept-foreign" | "no-claude-code";
|
|
23
30
|
/** Outcome of an install run, so the CLI can set a truthful exit code. */
|
|
24
31
|
export interface InstallResult {
|
|
25
32
|
/** Detected targets that could not be configured, with the reason. */
|
|
@@ -35,5 +42,7 @@ export interface InstallResult {
|
|
|
35
42
|
* failed target is a lie that surfaces later as "the tools never showed up".
|
|
36
43
|
* So: keep going, summarize, and return the failures for a non-zero exit.
|
|
37
44
|
*/
|
|
38
|
-
export declare function runInstall(
|
|
45
|
+
export declare function runInstall(options?: {
|
|
46
|
+
statusLine?: boolean;
|
|
47
|
+
}): InstallResult;
|
|
39
48
|
export declare function runUninstall(): void;
|
package/dist/install.js
CHANGED
|
@@ -122,14 +122,54 @@ function isEphemeralPath(path) {
|
|
|
122
122
|
* prompt. `node` (not process.execPath) keeps it alive across Node upgrades.
|
|
123
123
|
*/
|
|
124
124
|
function hookCommand() {
|
|
125
|
+
return cliCommand("hook");
|
|
126
|
+
}
|
|
127
|
+
/** The same durable command resolution, for any subcommand Claude Code runs for us. */
|
|
128
|
+
function cliCommand(subcommand) {
|
|
125
129
|
const selfDir = dirname(fileURLToPath(import.meta.url));
|
|
126
130
|
const localCli = join(selfDir, "cli.js");
|
|
127
131
|
if (!isEphemeralPath(selfDir + sep) && existsSync(localCli))
|
|
128
|
-
return `node "${localCli}"
|
|
132
|
+
return `node "${localCli}" ${subcommand}`;
|
|
129
133
|
const global = binOnPath("context-doctor");
|
|
130
134
|
if (global)
|
|
131
|
-
return `"${global}"
|
|
132
|
-
return
|
|
135
|
+
return `"${global}" ${subcommand}`;
|
|
136
|
+
return `npx -y context-doctor ${subcommand}`;
|
|
137
|
+
}
|
|
138
|
+
/**
|
|
139
|
+
* Claude Code's status bar: a `statusLine` command whose stdout is shown while
|
|
140
|
+
* the user types. Opt-in, because there is only one status line and it may
|
|
141
|
+
* already be someone's own — this never overwrites a statusLine that is not
|
|
142
|
+
* ours. Returns what happened so install can print the truth.
|
|
143
|
+
*/
|
|
144
|
+
export function installStatusLine() {
|
|
145
|
+
const settingsPath = join(homedir(), ".claude", "settings.json");
|
|
146
|
+
if (!existsSync(join(homedir(), ".claude")))
|
|
147
|
+
return "no-claude-code";
|
|
148
|
+
const settings = readJson(settingsPath);
|
|
149
|
+
const want = cliCommand("statusline");
|
|
150
|
+
const current = settings.statusLine?.command;
|
|
151
|
+
if (current === want)
|
|
152
|
+
return "already";
|
|
153
|
+
if (current && !isOurStatusLine(current))
|
|
154
|
+
return "kept-foreign";
|
|
155
|
+
settings.statusLine = { type: "command", command: want };
|
|
156
|
+
writeJsonWithBackup(settingsPath, settings);
|
|
157
|
+
return "installed";
|
|
158
|
+
}
|
|
159
|
+
function isOurStatusLine(command) {
|
|
160
|
+
return /context-doctor|cli\.js"?\s+statusline\s*$/.test(command);
|
|
161
|
+
}
|
|
162
|
+
function uninstallStatusLine() {
|
|
163
|
+
const settingsPath = join(homedir(), ".claude", "settings.json");
|
|
164
|
+
if (!existsSync(settingsPath))
|
|
165
|
+
return;
|
|
166
|
+
const settings = readJson(settingsPath);
|
|
167
|
+
const current = settings.statusLine?.command;
|
|
168
|
+
if (current && isOurStatusLine(current)) {
|
|
169
|
+
delete settings.statusLine;
|
|
170
|
+
writeJsonWithBackup(settingsPath, settings);
|
|
171
|
+
console.log("✓ Claude Code status line removed");
|
|
172
|
+
}
|
|
133
173
|
}
|
|
134
174
|
/** True when the hook had to fall back to npx — worth telling the user. */
|
|
135
175
|
function hookUsesNpx() {
|
|
@@ -224,7 +264,7 @@ function installSkill() {
|
|
|
224
264
|
* failed target is a lie that surfaces later as "the tools never showed up".
|
|
225
265
|
* So: keep going, summarize, and return the failures for a non-zero exit.
|
|
226
266
|
*/
|
|
227
|
-
export function runInstall() {
|
|
267
|
+
export function runInstall(options = {}) {
|
|
228
268
|
const entry = serverEntry();
|
|
229
269
|
const found = targets().filter((t) => t.detect());
|
|
230
270
|
if (found.length === 0) {
|
|
@@ -274,6 +314,28 @@ export function runInstall() {
|
|
|
274
314
|
console.error(`✗ Claude Code every-prompt hook: ${e.message}`);
|
|
275
315
|
failures.push("Claude Code hook");
|
|
276
316
|
}
|
|
317
|
+
if (options.statusLine) {
|
|
318
|
+
try {
|
|
319
|
+
switch (installStatusLine()) {
|
|
320
|
+
case "installed":
|
|
321
|
+
console.log("✓ Claude Code status line installed — live context size, cache share and cost while you type");
|
|
322
|
+
break;
|
|
323
|
+
case "already":
|
|
324
|
+
console.log("✓ Claude Code status line already installed");
|
|
325
|
+
break;
|
|
326
|
+
case "kept-foreign":
|
|
327
|
+
console.log("– Claude Code status line left alone: you already have your own statusLine command. Remove it first if you want ours.");
|
|
328
|
+
break;
|
|
329
|
+
case "no-claude-code":
|
|
330
|
+
console.log("– status line skipped: Claude Code not detected");
|
|
331
|
+
break;
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
catch (e) {
|
|
335
|
+
console.error(`✗ Claude Code status line: ${e.message}`);
|
|
336
|
+
failures.push("Claude Code status line");
|
|
337
|
+
}
|
|
338
|
+
}
|
|
277
339
|
if (failures.length > 0) {
|
|
278
340
|
console.log(`\nDone with ${failures.length} problem(s): ${failures.join(", ")}. See the ✗ lines above.`);
|
|
279
341
|
console.log("Everything else was installed. Exit code is 1 so scripts can tell; fix the file(s) and re-run install.");
|
|
@@ -306,6 +368,7 @@ export function runUninstall() {
|
|
|
306
368
|
console.log("✓ Agent Skill removed");
|
|
307
369
|
}
|
|
308
370
|
uninstallHook();
|
|
371
|
+
uninstallStatusLine();
|
|
309
372
|
// Remove our bookkeeping files too — uninstall means gone.
|
|
310
373
|
for (const file of [".context-doctor-hook-state.json", ".context-doctor-ledger.jsonl"]) {
|
|
311
374
|
const p = join(homedir(), ".claude", file);
|
package/dist/mcp.js
CHANGED
|
@@ -37,7 +37,7 @@ const STRATEGY_IDS = ["dedupe", "trim-tool-results", "trim-tool-calls", "strip-b
|
|
|
37
37
|
* recommended pattern.
|
|
38
38
|
*/
|
|
39
39
|
function createServer() {
|
|
40
|
-
const server = new McpServer({ name: "context-doctor", version: "0.
|
|
40
|
+
const server = new McpServer({ name: "context-doctor", version: "0.14.2" }, { instructions: SERVER_INSTRUCTIONS });
|
|
41
41
|
server.tool("profile_context", "Profile an LLM conversation or prompt: token breakdown by category, largest messages, and actionable findings about wasted context (duplicates, oversized tool results, base64 blobs, cache-unfriendly ordering). Accepts OpenAI/Anthropic conversation JSON or raw text. Call this immediately whenever the user asks about token usage, context size, LLM cost, or latency — and proactively offer it once a conversation grows long or accumulates large pasted content.", {
|
|
42
42
|
conversation: z.string().describe("Conversation JSON (OpenAI or Anthropic format, or bare message array) or raw prompt text"),
|
|
43
43
|
model: z.string().optional().describe("Target model name for context-window math, e.g. claude-sonnet-5 or gpt-4o"),
|
package/dist/optimize.d.ts
CHANGED
|
@@ -13,6 +13,15 @@ export interface OptimizeOptions {
|
|
|
13
13
|
keepRecent?: number;
|
|
14
14
|
/** Max tokens a trimmed tool result — or tool-call argument set — keeps. */
|
|
15
15
|
maxToolResultTokens?: number;
|
|
16
|
+
/**
|
|
17
|
+
* How many messages the trim boundary moves at a time. Bigger steps keep the
|
|
18
|
+
* prompt cache alive longer but leave stale results in place longer.
|
|
19
|
+
* Measured on a growing agent session: at 400 turns a step of 10 invalidated
|
|
20
|
+
* the cache on 21% of turns with ~2 stale results waiting on average; 20
|
|
21
|
+
* gave 12% and ~4; 40 gave 10% and ~9. Adaptive steps were worse everywhere,
|
|
22
|
+
* because a step that changes size moves the boundary by itself. Default 10.
|
|
23
|
+
*/
|
|
24
|
+
trimBoundaryStep?: number;
|
|
16
25
|
}
|
|
17
26
|
export interface AppliedChange {
|
|
18
27
|
strategy: StrategyId;
|
package/dist/optimize.js
CHANGED
|
@@ -9,10 +9,12 @@
|
|
|
9
9
|
import { createHash } from "node:crypto";
|
|
10
10
|
import { estimateTokens } from "./tokens.js";
|
|
11
11
|
import { hasBase64Blob, stripBase64Blobs } from "./blob.js";
|
|
12
|
+
const TRIM_BOUNDARY_STEP = 10;
|
|
12
13
|
const DEFAULTS = {
|
|
13
14
|
strategies: ["dedupe", "trim-tool-results", "strip-base64"],
|
|
14
15
|
keepRecent: 6,
|
|
15
16
|
maxToolResultTokens: 300,
|
|
17
|
+
trimBoundaryStep: TRIM_BOUNDARY_STEP,
|
|
16
18
|
};
|
|
17
19
|
/**
|
|
18
20
|
* Shrink the arguments of a tool call that has already run.
|
|
@@ -25,6 +27,42 @@ const DEFAULTS = {
|
|
|
25
27
|
* Keys are preserved (so the call still reads as itself) and only long string
|
|
26
28
|
* values are cut, with an explicit marker so nothing looks silently complete.
|
|
27
29
|
*/
|
|
30
|
+
/** The file a Write-like call creates, or null for anything else. */
|
|
31
|
+
function writtenPath(block) {
|
|
32
|
+
if (block?.name === "Write" && typeof block.input?.file_path === "string")
|
|
33
|
+
return block.input.file_path;
|
|
34
|
+
const cmd = block?.name === "Bash" ? String(block.input?.command ?? "") : "";
|
|
35
|
+
const heredoc = /(?:cat|tee)\s*>+\s*([^\s<]+)\s*<</.exec(cmd);
|
|
36
|
+
return heredoc ? heredoc[1] : null;
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Does the model later Edit this file without Reading it first? If so it is
|
|
40
|
+
* relying on the Write input still being in context, and trimming that input
|
|
41
|
+
* would break the Edit.
|
|
42
|
+
*/
|
|
43
|
+
function editedLaterWithoutRead(messages, writeIndex, path) {
|
|
44
|
+
if (!path)
|
|
45
|
+
return false;
|
|
46
|
+
for (let j = writeIndex + 1; j < messages.length; j++) {
|
|
47
|
+
const content = messages[j]?.content;
|
|
48
|
+
if (!Array.isArray(content))
|
|
49
|
+
continue;
|
|
50
|
+
for (const b of content) {
|
|
51
|
+
if (b?.type !== "tool_use")
|
|
52
|
+
continue;
|
|
53
|
+
const target = b.input?.file_path;
|
|
54
|
+
const cmd = b.name === "Bash" ? String(b.input?.command ?? "") : "";
|
|
55
|
+
const touchesPath = target === path || cmd.includes(path);
|
|
56
|
+
if (!touchesPath)
|
|
57
|
+
continue;
|
|
58
|
+
if (b.name === "Read" || /\b(cat|head|tail|sed|less)\b/.test(cmd))
|
|
59
|
+
return false; // it re-read: safe
|
|
60
|
+
if (b.name === "Edit" || b.name === "MultiEdit")
|
|
61
|
+
return true; // edited blind: keep the Write
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
return false;
|
|
65
|
+
}
|
|
28
66
|
function trimCallArguments(input, maxTokens) {
|
|
29
67
|
const budgetChars = maxTokens * 4;
|
|
30
68
|
const out = {};
|
|
@@ -54,17 +92,16 @@ function trimCallArguments(input, maxTokens) {
|
|
|
54
92
|
* STEP turns instead of once per turn. Older content is trimmed slightly later
|
|
55
93
|
* than it otherwise would be; that is much cheaper than losing the cache.
|
|
56
94
|
*/
|
|
57
|
-
|
|
58
|
-
function stableCutoff(messageCount, keepRecent) {
|
|
95
|
+
function stableCutoff(messageCount, keepRecent, step = TRIM_BOUNDARY_STEP) {
|
|
59
96
|
const raw = messageCount - keepRecent;
|
|
60
97
|
if (raw <= 0)
|
|
61
98
|
return 0;
|
|
62
99
|
// Below one step there is nothing to quantize to except zero, which would
|
|
63
100
|
// silently disable trimming on every short conversation. Such a conversation
|
|
64
101
|
// has no long stable prefix worth protecting anyway.
|
|
65
|
-
if (raw <
|
|
102
|
+
if (raw < step)
|
|
66
103
|
return raw;
|
|
67
|
-
return Math.floor(raw /
|
|
104
|
+
return Math.floor(raw / step) * step;
|
|
68
105
|
}
|
|
69
106
|
function hash(text) {
|
|
70
107
|
return createHash("sha1").update(text.replace(/\s+/g, " ").trim()).digest("hex");
|
|
@@ -171,6 +208,7 @@ export function optimizeConversation(input, options = {}) {
|
|
|
171
208
|
strategies: options.strategies ?? DEFAULTS.strategies,
|
|
172
209
|
keepRecent: options.keepRecent ?? DEFAULTS.keepRecent,
|
|
173
210
|
maxToolResultTokens: options.maxToolResultTokens ?? DEFAULTS.maxToolResultTokens,
|
|
211
|
+
trimBoundaryStep: options.trimBoundaryStep && options.trimBoundaryStep > 0 ? Math.floor(options.trimBoundaryStep) : DEFAULTS.trimBoundaryStep,
|
|
174
212
|
};
|
|
175
213
|
let data;
|
|
176
214
|
try {
|
|
@@ -216,7 +254,7 @@ export function optimizeConversation(input, options = {}) {
|
|
|
216
254
|
// their CURRENT question swapped for a pointer to a message ten turns back,
|
|
217
255
|
// and just sees a worse answer with no explanation. Older copies are fair
|
|
218
256
|
// game; the live turn is not.
|
|
219
|
-
const cutoff = stableCutoff(messages.length, opts.keepRecent);
|
|
257
|
+
const cutoff = stableCutoff(messages.length, opts.keepRecent, opts.trimBoundaryStep);
|
|
220
258
|
const seen = new Map();
|
|
221
259
|
messages.forEach((m, i) => {
|
|
222
260
|
const text = textOf(m.content);
|
|
@@ -241,7 +279,7 @@ export function optimizeConversation(input, options = {}) {
|
|
|
241
279
|
}
|
|
242
280
|
// -- trim-tool-results: shrink stale tool output ------------------------------
|
|
243
281
|
if (opts.strategies.includes("trim-tool-results")) {
|
|
244
|
-
const cutoff = stableCutoff(messages.length, opts.keepRecent);
|
|
282
|
+
const cutoff = stableCutoff(messages.length, opts.keepRecent, opts.trimBoundaryStep);
|
|
245
283
|
messages.forEach((m, i) => {
|
|
246
284
|
if (i >= cutoff || !isToolResultMessage(m))
|
|
247
285
|
return;
|
|
@@ -266,7 +304,7 @@ export function optimizeConversation(input, options = {}) {
|
|
|
266
304
|
}
|
|
267
305
|
// -- trim-tool-calls: shrink the arguments of calls that already ran ----------
|
|
268
306
|
if (opts.strategies.includes("trim-tool-calls")) {
|
|
269
|
-
const cutoff = stableCutoff(messages.length, opts.keepRecent);
|
|
307
|
+
const cutoff = stableCutoff(messages.length, opts.keepRecent, opts.trimBoundaryStep);
|
|
270
308
|
messages.forEach((m, i) => {
|
|
271
309
|
if (i >= cutoff)
|
|
272
310
|
return;
|
|
@@ -276,6 +314,14 @@ export function optimizeConversation(input, options = {}) {
|
|
|
276
314
|
for (const b of m.content) {
|
|
277
315
|
if (b?.type !== "tool_use" || b.input == null || typeof b.input !== "object")
|
|
278
316
|
continue;
|
|
317
|
+
// Measured across 42 real sessions: of 73 large Writes, 18 were later
|
|
318
|
+
// Edited and 16 of those Edits had no Read in between — the model
|
|
319
|
+
// built old_string from its own Write input, at a median distance of
|
|
320
|
+
// 43 messages. Trimming such a Write turns that Edit into a failure
|
|
321
|
+
// plus a recovery Read. Offline we can see the future, so leave
|
|
322
|
+
// those alone; the 63% never touched again are still pure gain.
|
|
323
|
+
if (editedLaterWithoutRead(messages, i, writtenPath(b)))
|
|
324
|
+
continue;
|
|
279
325
|
const before = estimateTokens(JSON.stringify(b.input));
|
|
280
326
|
if (before <= opts.maxToolResultTokens)
|
|
281
327
|
continue;
|
package/dist/profile.d.ts
CHANGED
|
@@ -51,6 +51,14 @@ export interface ContextProfile {
|
|
|
51
51
|
/** Present when the model has a known price. All figures are estimates. */
|
|
52
52
|
cost?: CostEstimate;
|
|
53
53
|
sourceFormat: string;
|
|
54
|
+
/**
|
|
55
|
+
* Present when estimates were scaled by a factor learned from this machine's
|
|
56
|
+
* own `--exact` counts for this model family. Absent means raw heuristic.
|
|
57
|
+
*/
|
|
58
|
+
calibration?: {
|
|
59
|
+
factor: number;
|
|
60
|
+
samples: number;
|
|
61
|
+
};
|
|
54
62
|
/** Propagated from parsing: input could not be read as a conversation. */
|
|
55
63
|
parseWarning?: string;
|
|
56
64
|
}
|
package/dist/profile.js
CHANGED
|
@@ -6,6 +6,7 @@ import { createHash } from "node:crypto";
|
|
|
6
6
|
import { contextWindowFor, estimateTokens, MESSAGE_OVERHEAD_TOKENS, providerFor } from "./tokens.js";
|
|
7
7
|
import { estimatedTtftSeconds, inputCostUsd, pricingFor } from "./pricing.js";
|
|
8
8
|
import { hasBase64Blob } from "./blob.js";
|
|
9
|
+
import { calibrationFor } from "./calibration.js";
|
|
9
10
|
function categoryOf(m) {
|
|
10
11
|
switch (m.kind) {
|
|
11
12
|
case "system": return "system";
|
|
@@ -145,9 +146,11 @@ function filesReadBy(toolName, toolCallText) {
|
|
|
145
146
|
return [...paths];
|
|
146
147
|
}
|
|
147
148
|
export function profileConversation(conv, model) {
|
|
149
|
+
// Learned from the user's own exact counts, if they ever fetched any.
|
|
150
|
+
const calibration = calibrationFor(model);
|
|
148
151
|
const perMessage = conv.messages.map((m) => ({
|
|
149
152
|
msg: m,
|
|
150
|
-
tokens: estimateTokens(m.text) + MESSAGE_OVERHEAD_TOKENS,
|
|
153
|
+
tokens: Math.round(estimateTokens(m.text) * calibration.factor) + MESSAGE_OVERHEAD_TOKENS,
|
|
151
154
|
}));
|
|
152
155
|
const totalTokens = perMessage.reduce((sum, p) => sum + p.tokens, 0);
|
|
153
156
|
const categories = {
|
|
@@ -445,6 +448,7 @@ export function profileConversation(conv, model) {
|
|
|
445
448
|
totalEstSavings,
|
|
446
449
|
cost,
|
|
447
450
|
sourceFormat: conv.sourceFormat,
|
|
451
|
+
calibration: calibration.samples > 0 ? calibration : undefined,
|
|
448
452
|
parseWarning: conv.parseWarning,
|
|
449
453
|
};
|
|
450
454
|
}
|
package/dist/proxy.js
CHANGED
|
@@ -66,6 +66,8 @@ export function startProxy(opts = {}) {
|
|
|
66
66
|
};
|
|
67
67
|
/** Last stable-prefix fingerprint per model, for cache-invalidation advice. */
|
|
68
68
|
const prefixFingerprints = new Map();
|
|
69
|
+
/** Per-message fingerprints of the previous request per model, for breakpoint placement. */
|
|
70
|
+
const messageFingerprints = new Map();
|
|
69
71
|
const advise = (msg) => {
|
|
70
72
|
if (stats.advice.includes(msg) || stats.advice.length >= 10)
|
|
71
73
|
return;
|
|
@@ -117,14 +119,24 @@ export function startProxy(opts = {}) {
|
|
|
117
119
|
strategies: route.strategies ?? opts.strategies,
|
|
118
120
|
keepRecent: route.keepRecent ?? opts.keepRecent,
|
|
119
121
|
maxToolResultTokens: route.maxToolResultTokens ?? opts.maxToolResultTokens,
|
|
122
|
+
trimBoundaryStep: opts.trimBoundaryStep,
|
|
120
123
|
};
|
|
121
124
|
}
|
|
122
125
|
// Prompt-cache advisor (Anthropic requests): the proxy sees real
|
|
123
126
|
// sequences, so cache-hostile patterns are observable facts here.
|
|
124
127
|
if (url.startsWith("/v1/messages") && requestModel) {
|
|
125
128
|
const stablePrefix = JSON.stringify(parsedBody.tools ?? null) + JSON.stringify(parsedBody.system ?? null);
|
|
126
|
-
|
|
127
|
-
|
|
129
|
+
const hasBreakpoint = body.includes("cache_control");
|
|
130
|
+
if (stablePrefix.length > 4000 && !hasBreakpoint) {
|
|
131
|
+
// Say WHERE, not just that. A breakpoint caches everything up
|
|
132
|
+
// to and including the block it sits on, so it belongs on the
|
|
133
|
+
// LAST stable block: the final tool definition if there are
|
|
134
|
+
// tools, otherwise the final system block.
|
|
135
|
+
const where = Array.isArray(parsedBody.tools) && parsedBody.tools.length > 0
|
|
136
|
+
? `the last entry in "tools" (tools come before system in the cached prefix)`
|
|
137
|
+
: `the last block of "system"`;
|
|
138
|
+
advise(`~${Math.round(stablePrefix.length / 4)}+ tokens of stable system/tools on ${requestModel} without cache_control. ` +
|
|
139
|
+
`Add {"cache_control":{"type":"ephemeral"}} to ${where}; everything before it then bills at ~10% on every call`);
|
|
128
140
|
}
|
|
129
141
|
const fp = fnv1a(stablePrefix);
|
|
130
142
|
const prev = prefixFingerprints.get(requestModel);
|
|
@@ -132,6 +144,31 @@ export function startProxy(opts = {}) {
|
|
|
132
144
|
advise(`system/tools prefix changed between ${requestModel} requests — every change re-bills the whole cached prefix; keep it byte-stable`);
|
|
133
145
|
}
|
|
134
146
|
prefixFingerprints.set(requestModel, fp);
|
|
147
|
+
// Second breakpoint: the conversation itself. Between two
|
|
148
|
+
// consecutive requests the older messages are usually identical;
|
|
149
|
+
// that run is cacheable too, and it is what re-bills every turn
|
|
150
|
+
// when nothing marks it. Find the longest message prefix that
|
|
151
|
+
// survived from the previous request and point at its last message.
|
|
152
|
+
const msgs = Array.isArray(parsedBody.messages)
|
|
153
|
+
? parsedBody.messages
|
|
154
|
+
: [];
|
|
155
|
+
const hashes = msgs.map((m) => fnv1a(JSON.stringify(m)));
|
|
156
|
+
const prevHashes = messageFingerprints.get(requestModel);
|
|
157
|
+
if (prevHashes && !hasBreakpoint) {
|
|
158
|
+
let stable = 0;
|
|
159
|
+
while (stable < hashes.length && stable < prevHashes.length && hashes[stable] === prevHashes[stable])
|
|
160
|
+
stable++;
|
|
161
|
+
if (stable >= 2) {
|
|
162
|
+
const stableChars = msgs.slice(0, stable).reduce((n, m) => n + JSON.stringify(m).length, 0);
|
|
163
|
+
const stableTokens = Math.round(stableChars / 4);
|
|
164
|
+
// Anthropic will not cache a prefix under ~1024 tokens (2048 on Haiku).
|
|
165
|
+
if (stableTokens >= 1024) {
|
|
166
|
+
advise(`messages #0-#${stable - 1} (~${stableTokens} tokens) were identical to the previous ${requestModel} request and carry no cache_control. ` +
|
|
167
|
+
`Put {"cache_control":{"type":"ephemeral"}} on the last content block of message #${stable - 1}; that run then reads from cache instead of re-billing each turn`);
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
messageFingerprints.set(requestModel, hashes);
|
|
135
172
|
}
|
|
136
173
|
}
|
|
137
174
|
catch {
|
package/dist/report.js
CHANGED
|
@@ -37,6 +37,11 @@ export function renderProfile(profile, options = {}) {
|
|
|
37
37
|
lines.push("");
|
|
38
38
|
}
|
|
39
39
|
lines.push(`Total: ~${formatTokens(p.totalTokens)} tokens across ${p.messageCount} messages (${p.sourceFormat} format)`);
|
|
40
|
+
if (p.calibration) {
|
|
41
|
+
// Scaled numbers must say so, or they read as the raw heuristic.
|
|
42
|
+
const pct = Math.round((p.calibration.factor - 1) * 100);
|
|
43
|
+
lines.push(` estimates calibrated ${pct >= 0 ? "+" : ""}${pct}% from ${p.calibration.samples} exact count(s) you ran on this machine (analyze --exact)`);
|
|
44
|
+
}
|
|
40
45
|
if (p.model) {
|
|
41
46
|
const windowNote = p.contextWindow
|
|
42
47
|
? ` of ${formatTokens(p.contextWindow)} window (${p.usagePct.toFixed(1)}%)`
|
package/dist/session.d.ts
CHANGED
|
@@ -72,4 +72,16 @@ export interface ParsedSession {
|
|
|
72
72
|
}
|
|
73
73
|
/** All session transcripts on this machine, newest first. */
|
|
74
74
|
export declare function listSessions(limit?: number): SessionInfo[];
|
|
75
|
+
/**
|
|
76
|
+
* Read a JSONL transcript line by line without ever materializing the whole
|
|
77
|
+
* file as one string.
|
|
78
|
+
*
|
|
79
|
+
* Agent sessions with large tool results reach hundreds of MB, and those are
|
|
80
|
+
* exactly the sessions that most need analysis — but V8 refuses to build a
|
|
81
|
+
* string past ~512MB, so readFileSync would throw on them (and in the hook,
|
|
82
|
+
* throw *silently*). Streaming has no such ceiling and keeps peak memory at
|
|
83
|
+
* one chunk. StringDecoder carries partial UTF-8 sequences across chunk
|
|
84
|
+
* boundaries so multi-byte characters are never corrupted.
|
|
85
|
+
*/
|
|
86
|
+
export declare function forEachLine(path: string, onLine: (line: string) => void): void;
|
|
75
87
|
export declare function parseSessionFile(path: string): ParsedSession;
|
package/dist/session.js
CHANGED
|
@@ -119,7 +119,7 @@ function parseChatGPTExport(data, path) {
|
|
|
119
119
|
* one chunk. StringDecoder carries partial UTF-8 sequences across chunk
|
|
120
120
|
* boundaries so multi-byte characters are never corrupted.
|
|
121
121
|
*/
|
|
122
|
-
function forEachLine(path, onLine) {
|
|
122
|
+
export function forEachLine(path, onLine) {
|
|
123
123
|
const fd = openSync(path, "r");
|
|
124
124
|
const decoder = new StringDecoder("utf8");
|
|
125
125
|
const buf = Buffer.allocUnsafe(4 * 1024 * 1024);
|