context-doctor 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -1
- package/dist/cli.js +16 -0
- package/dist/mcp.js +1 -1
- package/dist/proxy.d.ts +15 -0
- package/dist/proxy.js +81 -2
- package/dist/test/proxy.test.js +44 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -163,7 +163,11 @@ The proxy dedupes repeated content, trims stale tool results, and strips base64
|
|
|
163
163
|
[context-doctor] POST /v1/messages → 200 in 842ms | optimized 7.3k → 518 tokens (2 changes) | session total: 6.9k tokens ≈ $0.021 saved
|
|
164
164
|
```
|
|
165
165
|
|
|
166
|
-
`GET http://localhost:8787/stats` returns cumulative savings (requests, tokens, estimated USD)
|
|
166
|
+
`GET http://localhost:8787/stats` returns cumulative savings (requests, tokens, estimated USD), **exact upstream usage** read from every response (JSON and SSE), and **prompt-cache advisories** — the proxy watches your real traffic and flags big stable prefixes missing `cache_control` or prefix churn that silently re-bills the cache. Per-model behavior via `--config`:
|
|
167
|
+
|
|
168
|
+
```json
|
|
169
|
+
{ "routes": [{ "modelPrefix": "gpt", "strategies": ["strip-base64"], "keepRecent": 4 }] }
|
|
170
|
+
```
|
|
167
171
|
|
|
168
172
|
Because prompt caching matches byte-identical prefixes, deterministic strategies are chosen so repeated requests stay stable — but if you rely on aggressive cache prefixes, start with `--strategy strip-base64 --strategy dedupe` and add more as you verify.
|
|
169
173
|
|
package/dist/cli.js
CHANGED
|
@@ -61,6 +61,8 @@ Options:
|
|
|
61
61
|
--max-tool-tokens <n> (optimize) Token budget for trimmed tool results (default 300)
|
|
62
62
|
--port <n> (proxy) Port to listen on (default 8787)
|
|
63
63
|
--host <addr> (proxy) Bind address (default 127.0.0.1; use 0.0.0.0 to expose)
|
|
64
|
+
--config <file> (proxy) Per-route overrides: {"routes":[{"modelPrefix":"gpt","strategies":[...],
|
|
65
|
+
"keepRecent":n,"maxToolResultTokens":n}]} — first prefix match wins
|
|
64
66
|
--upstream-anthropic <url> (proxy) Override Anthropic upstream (testing)
|
|
65
67
|
--upstream-openai <url> (proxy) Override OpenAI upstream (testing)
|
|
66
68
|
-h, --help Show this help
|
|
@@ -115,6 +117,9 @@ function parseArgs(argv) {
|
|
|
115
117
|
case "--host":
|
|
116
118
|
args.host = argv[++i];
|
|
117
119
|
break;
|
|
120
|
+
case "--config":
|
|
121
|
+
args.config = argv[++i];
|
|
122
|
+
break;
|
|
118
123
|
case "--upstream-anthropic":
|
|
119
124
|
args.upstreamAnthropic = argv[++i];
|
|
120
125
|
break;
|
|
@@ -195,7 +200,18 @@ function main() {
|
|
|
195
200
|
return;
|
|
196
201
|
}
|
|
197
202
|
if (args.command === "proxy") {
|
|
203
|
+
let routes;
|
|
204
|
+
if (args.config) {
|
|
205
|
+
try {
|
|
206
|
+
routes = JSON.parse(readFileSync(args.config, "utf8")).routes;
|
|
207
|
+
}
|
|
208
|
+
catch (e) {
|
|
209
|
+
console.error(`Could not read --config ${args.config}: ${e.message}`);
|
|
210
|
+
process.exit(1);
|
|
211
|
+
}
|
|
212
|
+
}
|
|
198
213
|
startProxy({
|
|
214
|
+
routes: routes,
|
|
199
215
|
port: args.port,
|
|
200
216
|
host: args.host,
|
|
201
217
|
anthropicUpstream: args.upstreamAnthropic,
|
package/dist/mcp.js
CHANGED
|
@@ -37,7 +37,7 @@ const STRATEGY_IDS = ["dedupe", "trim-tool-results", "strip-base64", "prune-hist
|
|
|
37
37
|
* recommended pattern.
|
|
38
38
|
*/
|
|
39
39
|
function createServer() {
|
|
40
|
-
const server = new McpServer({ name: "context-doctor", version: "0.
|
|
40
|
+
const server = new McpServer({ name: "context-doctor", version: "0.7.0" }, { instructions: SERVER_INSTRUCTIONS });
|
|
41
41
|
server.tool("profile_context", "Profile an LLM conversation or prompt: token breakdown by category, largest messages, and actionable findings about wasted context (duplicates, oversized tool results, base64 blobs, cache-unfriendly ordering). Accepts OpenAI/Anthropic conversation JSON or raw text. Call this immediately whenever the user asks about token usage, context size, LLM cost, or latency — and proactively offer it once a conversation grows long or accumulates large pasted content.", {
|
|
42
42
|
conversation: z.string().describe("Conversation JSON (OpenAI or Anthropic format, or bare message array) or raw prompt text"),
|
|
43
43
|
model: z.string().optional().describe("Target model name for context-window math, e.g. claude-sonnet-5 or gpt-4o"),
|
package/dist/proxy.d.ts
CHANGED
|
@@ -14,6 +14,8 @@
|
|
|
14
14
|
import http from "node:http";
|
|
15
15
|
import { OptimizeOptions } from "./optimize.js";
|
|
16
16
|
export interface ProxyOptions extends OptimizeOptions {
|
|
17
|
+
/** Per-model-prefix overrides; first match wins, global options otherwise. */
|
|
18
|
+
routes?: RouteConfig[];
|
|
17
19
|
port?: number;
|
|
18
20
|
/**
|
|
19
21
|
* Bind address. Defaults to 127.0.0.1 — the proxy relays authenticated
|
|
@@ -33,5 +35,18 @@ export interface ProxyStats {
|
|
|
33
35
|
tokensSaved: number;
|
|
34
36
|
/** USD saved on input tokens, when the request's model has a known price. */
|
|
35
37
|
estUsdSaved: number;
|
|
38
|
+
/** Exact usage reported by upstream responses (both providers, incl. SSE). */
|
|
39
|
+
upstreamInputTokens: number;
|
|
40
|
+
upstreamOutputTokens: number;
|
|
41
|
+
/** Prompt-cache advisories observed on live traffic (unique, capped). */
|
|
42
|
+
advice: string[];
|
|
43
|
+
}
|
|
44
|
+
/** Per-model-prefix strategy overrides for the proxy (`--config`). */
|
|
45
|
+
export interface RouteConfig {
|
|
46
|
+
/** Applies when the request body's model starts with this prefix. */
|
|
47
|
+
modelPrefix: string;
|
|
48
|
+
strategies?: OptimizeOptions["strategies"];
|
|
49
|
+
keepRecent?: number;
|
|
50
|
+
maxToolResultTokens?: number;
|
|
36
51
|
}
|
|
37
52
|
export declare function startProxy(opts?: ProxyOptions): http.Server;
|
package/dist/proxy.js
CHANGED
|
@@ -26,6 +26,29 @@ function upstreamFor(url, opts) {
|
|
|
26
26
|
}
|
|
27
27
|
return undefined;
|
|
28
28
|
}
|
|
29
|
+
/** Pull exact usage out of a response body — JSON or SSE, either provider. */
|
|
30
|
+
function extractUsage(text) {
|
|
31
|
+
const last = (re) => {
|
|
32
|
+
let m;
|
|
33
|
+
let v = -1;
|
|
34
|
+
while ((m = re.exec(text)) !== null)
|
|
35
|
+
v = Number(m[1]);
|
|
36
|
+
return v;
|
|
37
|
+
};
|
|
38
|
+
const input = Math.max(last(/"input_tokens"\s*:\s*(\d+)/g), last(/"prompt_tokens"\s*:\s*(\d+)/g));
|
|
39
|
+
const output = Math.max(last(/"output_tokens"\s*:\s*(\d+)/g), last(/"completion_tokens"\s*:\s*(\d+)/g));
|
|
40
|
+
if (input < 0 && output < 0)
|
|
41
|
+
return null;
|
|
42
|
+
return { input: Math.max(input, 0), output: Math.max(output, 0) };
|
|
43
|
+
}
|
|
44
|
+
function fnv1a(s) {
|
|
45
|
+
let h = 0x811c9dc5;
|
|
46
|
+
for (let i = 0; i < s.length; i++) {
|
|
47
|
+
h ^= s.charCodeAt(i);
|
|
48
|
+
h = Math.imul(h, 0x01000193);
|
|
49
|
+
}
|
|
50
|
+
return h >>> 0;
|
|
51
|
+
}
|
|
29
52
|
export function startProxy(opts = {}) {
|
|
30
53
|
const port = opts.port ?? 8787;
|
|
31
54
|
const stats = {
|
|
@@ -36,6 +59,17 @@ export function startProxy(opts = {}) {
|
|
|
36
59
|
tokensAfter: 0,
|
|
37
60
|
tokensSaved: 0,
|
|
38
61
|
estUsdSaved: 0,
|
|
62
|
+
upstreamInputTokens: 0,
|
|
63
|
+
upstreamOutputTokens: 0,
|
|
64
|
+
advice: [],
|
|
65
|
+
};
|
|
66
|
+
/** Last stable-prefix fingerprint per model, for cache-invalidation advice. */
|
|
67
|
+
const prefixFingerprints = new Map();
|
|
68
|
+
const advise = (msg) => {
|
|
69
|
+
if (stats.advice.includes(msg) || stats.advice.length >= 10)
|
|
70
|
+
return;
|
|
71
|
+
stats.advice.push(msg);
|
|
72
|
+
console.error(`[context-doctor] cache advisor: ${msg}`);
|
|
39
73
|
};
|
|
40
74
|
const server = http.createServer(async (req, res) => {
|
|
41
75
|
const url = req.url ?? "/";
|
|
@@ -70,7 +104,39 @@ export function startProxy(opts = {}) {
|
|
|
70
104
|
let note = "passthrough";
|
|
71
105
|
if (req.method === "POST" && body && !isMeasurement) {
|
|
72
106
|
try {
|
|
73
|
-
|
|
107
|
+
// Per-route overrides: first modelPrefix match wins.
|
|
108
|
+
let effective = opts;
|
|
109
|
+
let requestModel;
|
|
110
|
+
try {
|
|
111
|
+
const parsedBody = JSON.parse(body);
|
|
112
|
+
requestModel = parsedBody.model;
|
|
113
|
+
const route = requestModel ? opts.routes?.find((r) => requestModel.startsWith(r.modelPrefix)) : undefined;
|
|
114
|
+
if (route) {
|
|
115
|
+
effective = {
|
|
116
|
+
strategies: route.strategies ?? opts.strategies,
|
|
117
|
+
keepRecent: route.keepRecent ?? opts.keepRecent,
|
|
118
|
+
maxToolResultTokens: route.maxToolResultTokens ?? opts.maxToolResultTokens,
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
// Prompt-cache advisor (Anthropic requests): the proxy sees real
|
|
122
|
+
// sequences, so cache-hostile patterns are observable facts here.
|
|
123
|
+
if (url.startsWith("/v1/messages") && requestModel) {
|
|
124
|
+
const stablePrefix = JSON.stringify(parsedBody.tools ?? null) + JSON.stringify(parsedBody.system ?? null);
|
|
125
|
+
if (stablePrefix.length > 4000 && !body.includes("cache_control")) {
|
|
126
|
+
advise(`~${Math.round(stablePrefix.length / 4)}+ tokens of stable system/tools on ${requestModel} without cache_control — adding a breakpoint would cut those to ~10% cost per call`);
|
|
127
|
+
}
|
|
128
|
+
const fp = fnv1a(stablePrefix);
|
|
129
|
+
const prev = prefixFingerprints.get(requestModel);
|
|
130
|
+
if (prev !== undefined && prev !== fp) {
|
|
131
|
+
advise(`system/tools prefix changed between ${requestModel} requests — every change re-bills the whole cached prefix; keep it byte-stable`);
|
|
132
|
+
}
|
|
133
|
+
prefixFingerprints.set(requestModel, fp);
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
catch {
|
|
137
|
+
/* body isn't JSON — global opts apply */
|
|
138
|
+
}
|
|
139
|
+
const result = optimizeConversation(body, effective);
|
|
74
140
|
const saved = result.tokensBefore - result.tokensAfter;
|
|
75
141
|
stats.tokensBefore += result.tokensBefore;
|
|
76
142
|
stats.tokensAfter += result.tokensAfter;
|
|
@@ -109,13 +175,26 @@ export function startProxy(opts = {}) {
|
|
|
109
175
|
res.setHeader(key, value);
|
|
110
176
|
});
|
|
111
177
|
if (upstream.body) {
|
|
112
|
-
// Pipe through chunk-by-chunk so SSE streaming works unchanged
|
|
178
|
+
// Pipe through chunk-by-chunk so SSE streaming works unchanged, while
|
|
179
|
+
// accumulating a bounded copy to read exact usage after the fact.
|
|
180
|
+
const USAGE_SCAN_CAP = 2 * 1024 * 1024;
|
|
181
|
+
let scanBuf = "";
|
|
182
|
+
const decoder = new TextDecoder();
|
|
113
183
|
const reader = upstream.body.getReader();
|
|
114
184
|
for (;;) {
|
|
115
185
|
const { done, value } = await reader.read();
|
|
116
186
|
if (done)
|
|
117
187
|
break;
|
|
118
188
|
res.write(value);
|
|
189
|
+
if (scanBuf.length < USAGE_SCAN_CAP)
|
|
190
|
+
scanBuf += decoder.decode(value, { stream: true });
|
|
191
|
+
}
|
|
192
|
+
if (upstream.ok && !isMeasurement) {
|
|
193
|
+
const usage = extractUsage(scanBuf);
|
|
194
|
+
if (usage) {
|
|
195
|
+
stats.upstreamInputTokens += usage.input;
|
|
196
|
+
stats.upstreamOutputTokens += usage.output;
|
|
197
|
+
}
|
|
119
198
|
}
|
|
120
199
|
}
|
|
121
200
|
res.end();
|
package/dist/test/proxy.test.js
CHANGED
|
@@ -32,6 +32,7 @@ const upstream = http.createServer((req, res) => {
|
|
|
32
32
|
receivedApiKey = req.headers["x-api-key"];
|
|
33
33
|
res.writeHead(200, { "content-type": "text/event-stream" });
|
|
34
34
|
res.write("event: message_start\ndata: {}\n\n");
|
|
35
|
+
res.write('event: message_delta\ndata: {"usage":{"input_tokens":120,"output_tokens":45}}\n\n');
|
|
35
36
|
res.write("event: message_stop\ndata: {}\n\n");
|
|
36
37
|
res.end();
|
|
37
38
|
});
|
|
@@ -69,6 +70,49 @@ test("/stats reports cumulative savings with dollar estimate", async () => {
|
|
|
69
70
|
assert.ok(stats.tokensSaved > 1000, `saved tokens tracked (${stats.tokensSaved})`);
|
|
70
71
|
assert.ok(stats.estUsdSaved > 0, "dollar savings estimated from the request's model");
|
|
71
72
|
});
|
|
73
|
+
test("response usage is captured from the SSE stream; cache advisor fires on prefix churn", async () => {
|
|
74
|
+
// Second request with a DIFFERENT system prompt on the same model → advisory.
|
|
75
|
+
const churned = JSON.parse(payload);
|
|
76
|
+
churned.system = "You are helpful. TODAY IS A NEW DAY."; // classic cache-buster
|
|
77
|
+
churned.tools = [{ name: "t", description: "x".repeat(5000), input_schema: { type: "object" } }];
|
|
78
|
+
const first = JSON.parse(payload);
|
|
79
|
+
first.tools = churned.tools;
|
|
80
|
+
for (const body of [first, churned]) {
|
|
81
|
+
await fetch(`http://localhost:${proxyPort}/v1/messages`, {
|
|
82
|
+
method: "POST",
|
|
83
|
+
headers: { "content-type": "application/json", "x-api-key": "sk-test-not-real" },
|
|
84
|
+
body: JSON.stringify(body),
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
const stats = (await (await fetch(`http://localhost:${proxyPort}/stats`)).json());
|
|
88
|
+
// Mock upstream reports usage in its SSE close event (added below).
|
|
89
|
+
assert.ok(stats.upstreamInputTokens >= 100, `usage input captured: ${stats.upstreamInputTokens}`);
|
|
90
|
+
assert.ok(stats.upstreamOutputTokens >= 40, `usage output captured: ${stats.upstreamOutputTokens}`);
|
|
91
|
+
assert.ok(stats.advice.some((a) => a.includes("prefix changed")), `prefix-churn advisory expected, got: ${JSON.stringify(stats.advice)}`);
|
|
92
|
+
assert.ok(stats.advice.some((a) => a.includes("cache_control")), "missing-cache_control advisory expected");
|
|
93
|
+
});
|
|
94
|
+
test("per-route config: empty strategy list disables optimization for matching models", async () => {
|
|
95
|
+
const { startProxy } = await import("../proxy.js");
|
|
96
|
+
const routed = startProxy({
|
|
97
|
+
port: 0,
|
|
98
|
+
anthropicUpstream: `http://localhost:${upstreamPort}`,
|
|
99
|
+
routes: [{ modelPrefix: "claude-sonnet", strategies: [] }],
|
|
100
|
+
});
|
|
101
|
+
await new Promise((r) => routed.once("listening", () => r()));
|
|
102
|
+
const routedPort = routed.address().port;
|
|
103
|
+
try {
|
|
104
|
+
const sent = payload;
|
|
105
|
+
await fetch(`http://localhost:${routedPort}/v1/messages`, {
|
|
106
|
+
method: "POST",
|
|
107
|
+
headers: { "content-type": "application/json" },
|
|
108
|
+
body: sent,
|
|
109
|
+
});
|
|
110
|
+
assert.equal(received.length, sent.length, "route with no strategies must pass body through unmodified");
|
|
111
|
+
}
|
|
112
|
+
finally {
|
|
113
|
+
routed.close();
|
|
114
|
+
}
|
|
115
|
+
});
|
|
72
116
|
test("unsupported paths get a clear 404, health stays up", async () => {
|
|
73
117
|
const notFound = await fetch(`http://localhost:${proxyPort}/v1/nope`, { method: "POST", body: "{}" });
|
|
74
118
|
assert.equal(notFound.status, 404);
|
package/package.json
CHANGED