@aexol/spectral 0.9.181 → 0.9.182
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -5
- package/dist/extensions/spectral-vision-fallback.js +1 -1
- package/dist/generated/zeus/const.d.ts.map +1 -1
- package/dist/generated/zeus/const.js +2 -0
- package/dist/generated/zeus/index.d.ts +8 -0
- package/dist/generated/zeus/index.d.ts.map +1 -1
- package/dist/memory/observer.d.ts.map +1 -1
- package/dist/memory/observer.js +3 -1
- package/dist/relay/models-fetch.d.ts +3 -3
- package/dist/relay/models-fetch.js +3 -3
- package/dist/sdk/ai/cache-benchmark/benchmark.d.ts +6 -0
- package/dist/sdk/ai/cache-benchmark/benchmark.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/benchmark.js +99 -0
- package/dist/sdk/ai/cache-benchmark/index.d.ts +7 -0
- package/dist/sdk/ai/cache-benchmark/index.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/index.js +6 -0
- package/dist/sdk/ai/cache-benchmark/prefix-cache.d.ts +18 -0
- package/dist/sdk/ai/cache-benchmark/prefix-cache.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/prefix-cache.js +60 -0
- package/dist/sdk/ai/cache-benchmark/strategies.d.ts +18 -0
- package/dist/sdk/ai/cache-benchmark/strategies.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/strategies.js +106 -0
- package/dist/sdk/ai/cache-benchmark/synthetic.d.ts +6 -0
- package/dist/sdk/ai/cache-benchmark/synthetic.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/synthetic.js +17 -0
- package/dist/sdk/ai/cache-benchmark/tokens.d.ts +8 -0
- package/dist/sdk/ai/cache-benchmark/tokens.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/tokens.js +22 -0
- package/dist/sdk/ai/cache-benchmark/types.d.ts +106 -0
- package/dist/sdk/ai/cache-benchmark/types.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/types.js +14 -0
- package/dist/sdk/ai/env-api-keys.js +0 -17
- package/dist/sdk/ai/models.d.ts.map +1 -1
- package/dist/sdk/ai/models.js +10 -0
- package/dist/sdk/ai/providers/openai-completions.d.ts +1 -3
- package/dist/sdk/ai/providers/openai-completions.d.ts.map +1 -1
- package/dist/sdk/ai/providers/openai-completions.js +4 -100
- package/dist/sdk/ai/providers/transform-messages.d.ts.map +1 -1
- package/dist/sdk/ai/providers/transform-messages.js +35 -0
- package/dist/sdk/ai/types.d.ts +2 -4
- package/dist/sdk/ai/types.d.ts.map +1 -1
- package/dist/sdk/ai/utils/oauth/index.d.ts +1 -2
- package/dist/sdk/ai/utils/oauth/index.d.ts.map +1 -1
- package/dist/sdk/ai/utils/oauth/index.js +1 -3
- package/dist/sdk/coding-agent/core/agent-session.d.ts +2 -0
- package/dist/sdk/coding-agent/core/agent-session.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/agent-session.js +3 -4
- package/dist/sdk/coding-agent/core/compaction/llm-compaction.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/compaction/llm-compaction.js +2 -1
- package/dist/sdk/coding-agent/core/model-registry.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/model-registry.js +0 -1
- package/dist/sdk/coding-agent/core/model-resolver.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/model-resolver.js +0 -1
- package/dist/sdk/coding-agent/core/provider-display-names.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/provider-display-names.js +0 -2
- package/dist/sdk/coding-agent/core/system-prompt.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/system-prompt.js +52 -48
- package/dist/sdk/coding-agent/utils/shell.d.ts.map +1 -1
- package/dist/sdk/coding-agent/utils/shell.js +10 -0
- package/dist/server/agent-bridge.d.ts +6 -6
- package/dist/server/agent-bridge.d.ts.map +1 -1
- package/dist/server/agent-bridge.js +27 -106
- package/dist/server/handlers/sessions.d.ts +1 -0
- package/dist/server/handlers/sessions.d.ts.map +1 -1
- package/dist/server/handlers/sessions.js +6 -1
- package/dist/server/session-stream.d.ts.map +1 -1
- package/dist/server/session-stream.js +4 -0
- package/dist/server/storage.d.ts +14 -5
- package/dist/server/storage.d.ts.map +1 -1
- package/dist/server/storage.js +37 -10
- package/dist/server/wire.d.ts +1 -0
- package/dist/server/wire.d.ts.map +1 -1
- package/package.json +2 -3
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"observer.d.ts","sourceRoot":"","sources":["../../src/memory/observer.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,
|
|
1
|
+
{"version":3,"file":"observer.d.ts","sourceRoot":"","sources":["../../src/memory/observer.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,SAAS,EAA6B,KAAK,UAAU,EAAwC,MAAM,4BAA4B,CAAC;AACzI,OAAO,KAAK,EAAW,KAAK,EAAE,MAAM,oBAAoB,CAAC;AAOzD,OAAO,KAAK,EAAE,iBAAiB,EAAa,MAAM,YAAY,CAAC;AAE/D,UAAU,eAAe;IACxB,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;IAClB,MAAM,EAAE,MAAM,CAAC;IACf,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACjC,gBAAgB,EAAE,MAAM,EAAE,CAAC;IAC3B,iBAAiB,EAAE,MAAM,EAAE,CAAC;IAC5B,KAAK,EAAE,MAAM,CAAC;IACd,qBAAqB,EAAE,MAAM,EAAE,CAAC;IAChC,MAAM,CAAC,EAAE,WAAW,CAAC;IACrB,SAAS,CAAC,EAAE,OAAO,SAAS,CAAC;IAC7B,OAAO,CAAC,EAAE,CAAC,KAAK,EAAE,UAAU,KAAK,IAAI,CAAC;IACtC,QAAQ,CAAC,EAAE,MAAM,CAAC;CAClB;AASD,eAAO,MAAM,6BAA6B,mDAAmD,CAAC;AAkC9F,wBAAgB,uBAAuB,CACtC,cAAc,EAAE,SAAS,MAAM,EAAE,GAAG,SAAS,EAC7C,qBAAqB,EAAE,SAAS,MAAM,EAAE,GACtC,MAAM,EAAE,GAAG,SAAS,CAYtB;AAED,wBAAsB,WAAW,CAAC,IAAI,EAAE,eAAe,GAAG,OAAO,CAAC,iBAAiB,EAAE,GAAG,SAAS,CAAC,CAkHjG;AAED,wBAAgB,yBAAyB,CAAC,OAAO,EAAE,iBAAiB,EAAE,GAAG,MAAM,EAAE,CAEhF"}
|
package/dist/memory/observer.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { agentLoop } from "../sdk/agent-core/index.js";
|
|
1
|
+
import { agentLoop, uuidv7 } from "../sdk/agent-core/index.js";
|
|
2
2
|
import { Type } from "../sdk/ai/index.js";
|
|
3
3
|
import { hashId } from "./ids.js";
|
|
4
4
|
import { AGENT_LOOP_MAX_TOKENS, boundedMaxTokens } from "./model-budget.js";
|
|
@@ -132,6 +132,8 @@ ${conversation}`;
|
|
|
132
132
|
maxTokens: boundedMaxTokens(model, AGENT_LOOP_MAX_TOKENS),
|
|
133
133
|
convertToLlm: (msgs) => msgs,
|
|
134
134
|
toolExecution: "sequential",
|
|
135
|
+
cacheRetention: "none",
|
|
136
|
+
sessionId: uuidv7(),
|
|
135
137
|
...(reasoning ? { reasoning: "high" } : {}),
|
|
136
138
|
...(effectiveMaxTurns !== undefined
|
|
137
139
|
? {
|
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
* Fetch the admin-managed list of allowed base models from the backend.
|
|
3
3
|
*
|
|
4
4
|
* Used by `AgentBridge` at startup to register synthetic providers
|
|
5
|
-
* (`spectral-proxy-
|
|
6
|
-
* inference call through the backend's `/v1/
|
|
7
|
-
*
|
|
5
|
+
* (`spectral-proxy-openai` / `spectral-proxy-user-model`) that route every
|
|
6
|
+
* inference call through the backend's `/v1/chat/completions` endpoint.
|
|
7
|
+
* The backend authenticates the call
|
|
8
8
|
* with the machine JWT (Bearer) and forwards to the upstream provider
|
|
9
9
|
* using its own (centralized) API keys.
|
|
10
10
|
*
|
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
* Fetch the admin-managed list of allowed base models from the backend.
|
|
3
3
|
*
|
|
4
4
|
* Used by `AgentBridge` at startup to register synthetic providers
|
|
5
|
-
* (`spectral-proxy-
|
|
6
|
-
* inference call through the backend's `/v1/
|
|
7
|
-
*
|
|
5
|
+
* (`spectral-proxy-openai` / `spectral-proxy-user-model`) that route every
|
|
6
|
+
* inference call through the backend's `/v1/chat/completions` endpoint.
|
|
7
|
+
* The backend authenticates the call
|
|
8
8
|
* with the machine JWT (Bearer) and forwards to the upstream provider
|
|
9
9
|
* using its own (centralized) API keys.
|
|
10
10
|
*
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import type { BenchmarkConfig, BenchmarkResult, CacheStrategy } from "./types.js";
|
|
2
|
+
export declare function runBenchmark(strategy: CacheStrategy, config: BenchmarkConfig): BenchmarkResult;
|
|
3
|
+
/** Run several strategies over one scenario and fill in the keep-all cost delta. */
|
|
4
|
+
export declare function runBenchmarkSuite(strategies: CacheStrategy[], config: BenchmarkConfig): BenchmarkResult[];
|
|
5
|
+
export declare function formatBenchmarkReport(results: BenchmarkResult[]): string;
|
|
6
|
+
//# sourceMappingURL=benchmark.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"benchmark.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/benchmark.ts"],"names":[],"mappings":"AAIA,OAAO,KAAK,EAAE,eAAe,EAAE,eAAe,EAAgB,aAAa,EAAE,MAAM,YAAY,CAAC;AAMhG,wBAAgB,YAAY,CAAC,QAAQ,EAAE,aAAa,EAAE,MAAM,EAAE,eAAe,GAAG,eAAe,CAiE9F;AAED,oFAAoF;AACpF,wBAAgB,iBAAiB,CAAC,UAAU,EAAE,aAAa,EAAE,EAAE,MAAM,EAAE,eAAe,GAAG,eAAe,EAAE,CASzG;AASD,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,eAAe,EAAE,GAAG,MAAM,CAmBxE"}
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
import { PrefixCacheSimulator } from "./prefix-cache.js";
|
|
2
|
+
import { createTurnEntry, resolvePerTurn } from "./synthetic.js";
|
|
3
|
+
import { fillerText, serializePrompt } from "./tokens.js";
|
|
4
|
+
import { DEFAULT_CACHE_PRICING } from "./types.js";
|
|
5
|
+
function resolvePricing(config) {
|
|
6
|
+
return { ...DEFAULT_CACHE_PRICING, ...(config.pricing ?? {}) };
|
|
7
|
+
}
|
|
8
|
+
export function runBenchmark(strategy, config) {
|
|
9
|
+
const pricing = resolvePricing(config);
|
|
10
|
+
const minCacheTokens = config.minCacheTokens ?? 1024;
|
|
11
|
+
const systemText = fillerText(Math.max(0, config.systemPromptTokens * 4), 0);
|
|
12
|
+
const toolsText = fillerText(Math.max(0, config.toolsTokens * 4), 1);
|
|
13
|
+
const simulator = new PrefixCacheSimulator({
|
|
14
|
+
sessionId: config.sessionId ?? "benchmark",
|
|
15
|
+
minCacheTokens,
|
|
16
|
+
});
|
|
17
|
+
let state = strategy.init();
|
|
18
|
+
let cutCount = 0;
|
|
19
|
+
let totalPromptTokens = 0;
|
|
20
|
+
let totalInputTokens = 0;
|
|
21
|
+
let totalOutputTokens = 0;
|
|
22
|
+
let totalCacheReadTokens = 0;
|
|
23
|
+
let totalCacheWriteTokens = 0;
|
|
24
|
+
let maxPromptTokens = 0;
|
|
25
|
+
let finalPromptTokens = 0;
|
|
26
|
+
for (let turn = 0; turn < config.turns; turn++) {
|
|
27
|
+
const promptTokens = resolvePerTurn(config.turnPromptTokens, turn);
|
|
28
|
+
const outputTokens = resolvePerTurn(config.turnOutputTokens, turn);
|
|
29
|
+
const entry = createTurnEntry(turn + 1, promptTokens);
|
|
30
|
+
const step = strategy.step(state, turn, entry);
|
|
31
|
+
state = { entries: step.entries };
|
|
32
|
+
if (step.cut)
|
|
33
|
+
cutCount++;
|
|
34
|
+
const promptText = serializePrompt(systemText, toolsText, step.entries);
|
|
35
|
+
const result = simulator.account(promptText, outputTokens);
|
|
36
|
+
totalPromptTokens += result.promptTokens;
|
|
37
|
+
totalInputTokens += result.inputTokens;
|
|
38
|
+
totalOutputTokens += result.outputTokens;
|
|
39
|
+
totalCacheReadTokens += result.cacheReadTokens;
|
|
40
|
+
totalCacheWriteTokens += result.cacheWriteTokens;
|
|
41
|
+
maxPromptTokens = Math.max(maxPromptTokens, result.promptTokens);
|
|
42
|
+
finalPromptTokens = result.promptTokens;
|
|
43
|
+
}
|
|
44
|
+
const totalCost = (totalInputTokens * pricing.input +
|
|
45
|
+
totalOutputTokens * pricing.output +
|
|
46
|
+
totalCacheReadTokens * pricing.cacheRead) /
|
|
47
|
+
1_000_000;
|
|
48
|
+
const cacheHitRate = totalPromptTokens > 0 ? totalCacheReadTokens / totalPromptTokens : 0;
|
|
49
|
+
return {
|
|
50
|
+
name: strategy.name,
|
|
51
|
+
turns: config.turns,
|
|
52
|
+
cutCount,
|
|
53
|
+
totalPromptTokens,
|
|
54
|
+
totalInputTokens,
|
|
55
|
+
totalOutputTokens,
|
|
56
|
+
totalCacheReadTokens,
|
|
57
|
+
totalCacheWriteTokens,
|
|
58
|
+
cacheHitRate,
|
|
59
|
+
avgPromptTokens: Math.round(totalPromptTokens / config.turns),
|
|
60
|
+
maxPromptTokens,
|
|
61
|
+
finalPromptTokens,
|
|
62
|
+
totalCost,
|
|
63
|
+
costPerTurn: totalCost / config.turns,
|
|
64
|
+
costVsKeepAllPct: 0,
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
/** Run several strategies over one scenario and fill in the keep-all cost delta. */
|
|
68
|
+
export function runBenchmarkSuite(strategies, config) {
|
|
69
|
+
const results = strategies.map((strategy) => runBenchmark(strategy, config));
|
|
70
|
+
const baseline = results.find((result) => result.name === "keep-all");
|
|
71
|
+
if (baseline) {
|
|
72
|
+
for (const result of results) {
|
|
73
|
+
result.costVsKeepAllPct = (result.totalCost / baseline.totalCost - 1) * 100;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
return results;
|
|
77
|
+
}
|
|
78
|
+
function fmt(value, digits = 2) {
|
|
79
|
+
return value.toLocaleString("en-US", {
|
|
80
|
+
minimumFractionDigits: digits,
|
|
81
|
+
maximumFractionDigits: digits,
|
|
82
|
+
});
|
|
83
|
+
}
|
|
84
|
+
export function formatBenchmarkReport(results) {
|
|
85
|
+
const lines = [];
|
|
86
|
+
lines.push("## Cache benchmark (OpenAI-style prefix cache)");
|
|
87
|
+
lines.push("");
|
|
88
|
+
lines.push("cacheHitRate = cacheRead / prompt tokens; input == cacheWrite == uncached prompt tokens.");
|
|
89
|
+
lines.push("");
|
|
90
|
+
lines.push("| strategy | cuts | avg prompt | avg cacheRead | avg cacheWrite | cache hit % | cost $ | cost/turn $ | vs keep-all % |");
|
|
91
|
+
lines.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |");
|
|
92
|
+
for (const result of results) {
|
|
93
|
+
const avgCacheRead = result.totalCacheReadTokens / result.turns;
|
|
94
|
+
const avgCacheWrite = result.totalCacheWriteTokens / result.turns;
|
|
95
|
+
lines.push(`| ${result.name} | ${result.cutCount} | ${fmt(result.avgPromptTokens, 0)} | ${fmt(avgCacheRead, 0)} | ${fmt(avgCacheWrite, 0)} | ${fmt(result.cacheHitRate * 100, 1)} | ${fmt(result.totalCost, 4)} | ${fmt(result.costPerTurn, 5)} | ${fmt(result.costVsKeepAllPct, 1)} |`);
|
|
96
|
+
}
|
|
97
|
+
lines.push("");
|
|
98
|
+
return lines.join("\n");
|
|
99
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/index.ts"],"names":[],"mappings":"AAAA,cAAc,gBAAgB,CAAC;AAC/B,cAAc,mBAAmB,CAAC;AAClC,cAAc,iBAAiB,CAAC;AAChC,cAAc,gBAAgB,CAAC;AAC/B,cAAc,aAAa,CAAC;AAC5B,cAAc,YAAY,CAAC"}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import type { PrefixCacheConfig, PrefixCacheTurnResult } from "./types.js";
|
|
2
|
+
/** Length of the longest common character prefix between two strings. */
|
|
3
|
+
export declare function commonPrefixLength(a: string, b: string): number;
|
|
4
|
+
/**
|
|
5
|
+
* Deterministic simulator of OpenAI-style automatic prompt caching.
|
|
6
|
+
*
|
|
7
|
+
* Caching is scoped to `sessionId` and matches the longest common prefix with
|
|
8
|
+
* the previous request in that session. A prefix must reach `minCacheTokens`
|
|
9
|
+
* to be served from cache at all (OpenAI's real threshold is ~1024 tokens).
|
|
10
|
+
*/
|
|
11
|
+
export declare class PrefixCacheSimulator {
|
|
12
|
+
private readonly config;
|
|
13
|
+
private lastPrompt;
|
|
14
|
+
constructor(config: PrefixCacheConfig);
|
|
15
|
+
/** Account one provider request. `outputTokens` does not affect cache, only reporting. */
|
|
16
|
+
account(promptText: string, outputTokens: number): PrefixCacheTurnResult;
|
|
17
|
+
}
|
|
18
|
+
//# sourceMappingURL=prefix-cache.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"prefix-cache.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/prefix-cache.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,iBAAiB,EAAE,qBAAqB,EAAE,MAAM,YAAY,CAAC;AAE3E,yEAAyE;AACzE,wBAAgB,kBAAkB,CAAC,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,GAAG,MAAM,CAO/D;AAED;;;;;;GAMG;AACH,qBAAa,oBAAoB;IAGpB,OAAO,CAAC,QAAQ,CAAC,MAAM;IAFnC,OAAO,CAAC,UAAU,CAAqB;gBAEV,MAAM,EAAE,iBAAiB;IAEtD,0FAA0F;IAC1F,OAAO,CAAC,UAAU,EAAE,MAAM,EAAE,YAAY,EAAE,MAAM,GAAG,qBAAqB;CAwCxE"}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { estimateTextTokens } from "./tokens.js";
|
|
2
|
+
/** Length of the longest common character prefix between two strings. */
|
|
3
|
+
export function commonPrefixLength(a, b) {
|
|
4
|
+
const length = Math.min(a.length, b.length);
|
|
5
|
+
let index = 0;
|
|
6
|
+
while (index < length && a[index] === b[index]) {
|
|
7
|
+
index++;
|
|
8
|
+
}
|
|
9
|
+
return index;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* Deterministic simulator of OpenAI-style automatic prompt caching.
|
|
13
|
+
*
|
|
14
|
+
* Caching is scoped to `sessionId` and matches the longest common prefix with
|
|
15
|
+
* the previous request in that session. A prefix must reach `minCacheTokens`
|
|
16
|
+
* to be served from cache at all (OpenAI's real threshold is ~1024 tokens).
|
|
17
|
+
*/
|
|
18
|
+
export class PrefixCacheSimulator {
|
|
19
|
+
config;
|
|
20
|
+
lastPrompt;
|
|
21
|
+
constructor(config) {
|
|
22
|
+
this.config = config;
|
|
23
|
+
}
|
|
24
|
+
/** Account one provider request. `outputTokens` does not affect cache, only reporting. */
|
|
25
|
+
account(promptText, outputTokens) {
|
|
26
|
+
const promptTokens = estimateTextTokens(promptText);
|
|
27
|
+
if (this.config.cacheRetention === "none" || !this.config.sessionId) {
|
|
28
|
+
return {
|
|
29
|
+
promptTokens,
|
|
30
|
+
cacheReadTokens: 0,
|
|
31
|
+
inputTokens: promptTokens,
|
|
32
|
+
cacheWriteTokens: promptTokens,
|
|
33
|
+
outputTokens,
|
|
34
|
+
commonPrefixTokens: 0,
|
|
35
|
+
cacheHit: false,
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
let commonPrefixTokens = 0;
|
|
39
|
+
let cacheReadTokens = 0;
|
|
40
|
+
let inputTokens = promptTokens;
|
|
41
|
+
if (this.lastPrompt !== undefined) {
|
|
42
|
+
const prefixChars = commonPrefixLength(this.lastPrompt, promptText);
|
|
43
|
+
commonPrefixTokens = estimateTextTokens(this.lastPrompt.slice(0, prefixChars));
|
|
44
|
+
if (commonPrefixTokens >= this.config.minCacheTokens) {
|
|
45
|
+
cacheReadTokens = commonPrefixTokens;
|
|
46
|
+
inputTokens = Math.max(0, promptTokens - cacheReadTokens);
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
this.lastPrompt = promptText;
|
|
50
|
+
return {
|
|
51
|
+
promptTokens,
|
|
52
|
+
cacheReadTokens,
|
|
53
|
+
inputTokens,
|
|
54
|
+
cacheWriteTokens: inputTokens,
|
|
55
|
+
outputTokens,
|
|
56
|
+
commonPrefixTokens,
|
|
57
|
+
cacheHit: cacheReadTokens > 0,
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import type { CacheStrategy, PromptEntry } from "./types.js";
|
|
2
|
+
/** Create a deterministic compaction summary that is byte-stable after creation. */
|
|
3
|
+
export declare function createSummaryEntry(cutTurn: number, summaryTokens: number): PromptEntry;
|
|
4
|
+
/** Baseline: never compact. Prompt grows monotonically (best cache, until overflow). */
|
|
5
|
+
export declare function keepAll(): CacheStrategy;
|
|
6
|
+
/** Continuous sliding window: always keep only the last `windowTurns` entries. */
|
|
7
|
+
export declare function truncateSliding(windowTurns: number): CacheStrategy;
|
|
8
|
+
/** Batched truncation: cut to the last `keepLastTurns` entries once per `cutEveryTurns`. */
|
|
9
|
+
export declare function truncateBatched(cutEveryTurns: number, keepLastTurns: number): CacheStrategy;
|
|
10
|
+
/** Batched summarization: replace the dropped span with a fixed-size summary. */
|
|
11
|
+
export declare function summarizeBatched(cutEveryTurns: number, keepRecentTurns: number, summaryTokens: number): CacheStrategy;
|
|
12
|
+
/**
|
|
13
|
+
* Mirror the fork's autoOlderHistory policy: when estimated context tokens exceed
|
|
14
|
+
* `contextWindow * ratio`, replace the older span with a summary and keep a raw
|
|
15
|
+
* recent tail of ~`keepRecentTokens` tokens.
|
|
16
|
+
*/
|
|
17
|
+
export declare function summarizeOnTokenRatio(contextWindow: number, ratio: number, keepRecentTokens: number, summaryTokens: number): CacheStrategy;
|
|
18
|
+
//# sourceMappingURL=strategies.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"strategies.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/strategies.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,aAAa,EAAE,WAAW,EAAiB,MAAM,YAAY,CAAC;AAM5E,oFAAoF;AACpF,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,MAAM,EAAE,aAAa,EAAE,MAAM,GAAG,WAAW,CAUtF;AAED,wFAAwF;AACxF,wBAAgB,OAAO,IAAI,aAAa,CAMvC;AAED,kFAAkF;AAClF,wBAAgB,eAAe,CAAC,WAAW,EAAE,MAAM,GAAG,aAAa,CAUlE;AAED,4FAA4F;AAC5F,wBAAgB,eAAe,CAAC,aAAa,EAAE,MAAM,EAAE,aAAa,EAAE,MAAM,GAAG,aAAa,CAc3F;AAED,iFAAiF;AACjF,wBAAgB,gBAAgB,CAC/B,aAAa,EAAE,MAAM,EACrB,eAAe,EAAE,MAAM,EACvB,aAAa,EAAE,MAAM,GACnB,aAAa,CAiBf;AAED;;;;GAIG;AACH,wBAAgB,qBAAqB,CACpC,aAAa,EAAE,MAAM,EACrB,KAAK,EAAE,MAAM,EACb,gBAAgB,EAAE,MAAM,EACxB,aAAa,EAAE,MAAM,GACnB,aAAa,CA8Bf"}
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
import { fillerText } from "./tokens.js";
|
|
2
|
+
function append(state, entry) {
|
|
3
|
+
return [...state.entries, entry];
|
|
4
|
+
}
|
|
5
|
+
/** Create a deterministic compaction summary that is byte-stable after creation. */
|
|
6
|
+
export function createSummaryEntry(cutTurn, summaryTokens) {
|
|
7
|
+
const header = `summary(cut=${cutTurn}):`;
|
|
8
|
+
const bodyLength = Math.max(0, summaryTokens * 4 - header.length);
|
|
9
|
+
const text = header + fillerText(bodyLength, cutTurn);
|
|
10
|
+
return {
|
|
11
|
+
id: -cutTurn,
|
|
12
|
+
kind: "summary",
|
|
13
|
+
text,
|
|
14
|
+
tokens: Math.ceil(text.length / 4),
|
|
15
|
+
};
|
|
16
|
+
}
|
|
17
|
+
/** Baseline: never compact. Prompt grows monotonically (best cache, until overflow). */
|
|
18
|
+
export function keepAll() {
|
|
19
|
+
return {
|
|
20
|
+
name: "keep-all",
|
|
21
|
+
init: () => ({ entries: [] }),
|
|
22
|
+
step: (state, _turn, entry) => ({ entries: append(state, entry), cut: false }),
|
|
23
|
+
};
|
|
24
|
+
}
|
|
25
|
+
/** Continuous sliding window: always keep only the last `windowTurns` entries. */
|
|
26
|
+
export function truncateSliding(windowTurns) {
|
|
27
|
+
return {
|
|
28
|
+
name: `truncate-sliding(window=${windowTurns})`,
|
|
29
|
+
init: () => ({ entries: [] }),
|
|
30
|
+
step: (state, _turn, entry) => {
|
|
31
|
+
const appended = append(state, entry);
|
|
32
|
+
const entries = appended.slice(-windowTurns);
|
|
33
|
+
return { entries, cut: appended.length > entries.length };
|
|
34
|
+
},
|
|
35
|
+
};
|
|
36
|
+
}
|
|
37
|
+
/** Batched truncation: cut to the last `keepLastTurns` entries once per `cutEveryTurns`. */
|
|
38
|
+
export function truncateBatched(cutEveryTurns, keepLastTurns) {
|
|
39
|
+
return {
|
|
40
|
+
name: `truncate-batched(every=${cutEveryTurns},keep=${keepLastTurns})`,
|
|
41
|
+
init: () => ({ entries: [] }),
|
|
42
|
+
step: (state, turn, entry) => {
|
|
43
|
+
const appended = append(state, entry);
|
|
44
|
+
// turn is 0-based; cut once a full batch has accumulated.
|
|
45
|
+
if ((turn + 1) % cutEveryTurns === 0) {
|
|
46
|
+
const entries = appended.slice(-keepLastTurns);
|
|
47
|
+
return { entries, cut: appended.length > entries.length };
|
|
48
|
+
}
|
|
49
|
+
return { entries: appended, cut: false };
|
|
50
|
+
},
|
|
51
|
+
};
|
|
52
|
+
}
|
|
53
|
+
/** Batched summarization: replace the dropped span with a fixed-size summary. */
|
|
54
|
+
export function summarizeBatched(cutEveryTurns, keepRecentTurns, summaryTokens) {
|
|
55
|
+
return {
|
|
56
|
+
name: `summarize-batched(every=${cutEveryTurns},keep=${keepRecentTurns},summary=${summaryTokens})`,
|
|
57
|
+
init: () => ({ entries: [] }),
|
|
58
|
+
step: (state, turn, entry) => {
|
|
59
|
+
const appended = append(state, entry);
|
|
60
|
+
if ((turn + 1) % cutEveryTurns !== 0) {
|
|
61
|
+
return { entries: appended, cut: false };
|
|
62
|
+
}
|
|
63
|
+
const recent = appended.slice(-keepRecentTurns);
|
|
64
|
+
if (recent.length === appended.length) {
|
|
65
|
+
return { entries: appended, cut: false };
|
|
66
|
+
}
|
|
67
|
+
const summary = createSummaryEntry(turn + 1, summaryTokens);
|
|
68
|
+
return { entries: [summary, ...recent], cut: true };
|
|
69
|
+
},
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* Mirror the fork's autoOlderHistory policy: when estimated context tokens exceed
|
|
74
|
+
* `contextWindow * ratio`, replace the older span with a summary and keep a raw
|
|
75
|
+
* recent tail of ~`keepRecentTokens` tokens.
|
|
76
|
+
*/
|
|
77
|
+
export function summarizeOnTokenRatio(contextWindow, ratio, keepRecentTokens, summaryTokens) {
|
|
78
|
+
return {
|
|
79
|
+
name: `summarize-on-token-ratio(window=${contextWindow},ratio=${ratio},keep=${keepRecentTokens},summary=${summaryTokens})`,
|
|
80
|
+
init: () => ({ entries: [] }),
|
|
81
|
+
step: (state, turn, entry) => {
|
|
82
|
+
const appended = append(state, entry);
|
|
83
|
+
const total = appended.reduce((sum, e) => sum + e.tokens, 0);
|
|
84
|
+
if (total <= Math.floor(contextWindow * ratio)) {
|
|
85
|
+
return { entries: appended, cut: false };
|
|
86
|
+
}
|
|
87
|
+
// Walk from the newest entry backward, keeping the newest suffix whose
|
|
88
|
+
// total is <= keepRecentTokens (always keeping at least the newest entry).
|
|
89
|
+
let keptTokens = 0;
|
|
90
|
+
let keepFrom = appended.length;
|
|
91
|
+
for (let i = appended.length - 1; i >= 0; i--) {
|
|
92
|
+
const candidate = appended[i];
|
|
93
|
+
if (i < appended.length - 1 && keptTokens + candidate.tokens > keepRecentTokens)
|
|
94
|
+
break;
|
|
95
|
+
keptTokens += candidate.tokens;
|
|
96
|
+
keepFrom = i;
|
|
97
|
+
}
|
|
98
|
+
const recent = appended.slice(keepFrom);
|
|
99
|
+
if (keepFrom === 0) {
|
|
100
|
+
return { entries: appended, cut: false };
|
|
101
|
+
}
|
|
102
|
+
const summary = createSummaryEntry(turn + 1, summaryTokens);
|
|
103
|
+
return { entries: [summary, ...recent], cut: true };
|
|
104
|
+
},
|
|
105
|
+
};
|
|
106
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import type { PromptEntry } from "./types.js";
|
|
2
|
+
/** Build a deterministic, byte-stable prompt entry for a given turn. */
|
|
3
|
+
export declare function createTurnEntry(turn: number, tokens: number): PromptEntry;
|
|
4
|
+
/** Resolve a number-or-factory config value for a given turn. */
|
|
5
|
+
export declare function resolvePerTurn(value: number | ((turn: number) => number), turn: number): number;
|
|
6
|
+
//# sourceMappingURL=synthetic.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"synthetic.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/synthetic.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,YAAY,CAAC;AAE9C,wEAAwE;AACxE,wBAAgB,eAAe,CAAC,IAAI,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM,GAAG,WAAW,CAUzE;AAED,iEAAiE;AACjE,wBAAgB,cAAc,CAAC,KAAK,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,CAAC,EAAE,IAAI,EAAE,MAAM,GAAG,MAAM,CAE/F"}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import { fillerText } from "./tokens.js";
|
|
2
|
+
/** Build a deterministic, byte-stable prompt entry for a given turn. */
|
|
3
|
+
export function createTurnEntry(turn, tokens) {
|
|
4
|
+
const header = `turn:${turn}:`;
|
|
5
|
+
const bodyLength = Math.max(0, tokens * 4 - header.length);
|
|
6
|
+
const text = header + fillerText(bodyLength, turn);
|
|
7
|
+
return {
|
|
8
|
+
id: turn,
|
|
9
|
+
kind: "turn",
|
|
10
|
+
text,
|
|
11
|
+
tokens: Math.ceil(text.length / 4),
|
|
12
|
+
};
|
|
13
|
+
}
|
|
14
|
+
/** Resolve a number-or-factory config value for a given turn. */
|
|
15
|
+
export function resolvePerTurn(value, turn) {
|
|
16
|
+
return typeof value === "function" ? value(turn) : value;
|
|
17
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
import type { PromptEntry } from "./types.js";
|
|
2
|
+
/** Conservative chars/4 token heuristic, matching cli-node's estimateTokens. */
|
|
3
|
+
export declare function estimateTextTokens(text: string): number;
|
|
4
|
+
/** Deterministic filler of exactly `length` chars (a-z cycling, offset by `seed`). */
|
|
5
|
+
export declare function fillerText(length: number, seed?: number): string;
|
|
6
|
+
/** Join a stable head and ordered entries into one deterministic prompt string. */
|
|
7
|
+
export declare function serializePrompt(systemText: string, toolsText: string, entries: PromptEntry[]): string;
|
|
8
|
+
//# sourceMappingURL=tokens.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tokens.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/tokens.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,YAAY,CAAC;AAE9C,gFAAgF;AAChF,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEvD;AAED,sFAAsF;AACtF,wBAAgB,UAAU,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,SAAI,GAAG,MAAM,CAQ3D;AAED,mFAAmF;AACnF,wBAAgB,eAAe,CAAC,UAAU,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,WAAW,EAAE,GAAG,MAAM,CAMrG"}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/** Conservative chars/4 token heuristic, matching cli-node's estimateTokens. */
|
|
2
|
+
export function estimateTextTokens(text) {
|
|
3
|
+
return Math.ceil(text.length / 4);
|
|
4
|
+
}
|
|
5
|
+
/** Deterministic filler of exactly `length` chars (a-z cycling, offset by `seed`). */
|
|
6
|
+
export function fillerText(length, seed = 0) {
|
|
7
|
+
const base = "abcdefghijklmnopqrstuvwxyz";
|
|
8
|
+
const offset = ((seed % base.length) + base.length) % base.length;
|
|
9
|
+
let out = "";
|
|
10
|
+
for (let i = 0; i < length; i++) {
|
|
11
|
+
out += base[(offset + i) % base.length];
|
|
12
|
+
}
|
|
13
|
+
return out;
|
|
14
|
+
}
|
|
15
|
+
/** Join a stable head and ordered entries into one deterministic prompt string. */
|
|
16
|
+
export function serializePrompt(systemText, toolsText, entries) {
|
|
17
|
+
const parts = [`[system]${systemText}`, `[tools]${toolsText}`];
|
|
18
|
+
for (const entry of entries) {
|
|
19
|
+
parts.push(`[${entry.kind}:${entry.id}]${entry.text}`);
|
|
20
|
+
}
|
|
21
|
+
return parts.join("\n");
|
|
22
|
+
}
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cache-benchmark model types.
|
|
3
|
+
*
|
|
4
|
+
* This is an offline, deterministic simulator of OpenAI-style automatic prompt
|
|
5
|
+
* caching. It answers one question cheaply and reproducibly: what happens to
|
|
6
|
+
* cacheRead / cacheWrite / input tokens (and therefore cost) when a compaction
|
|
7
|
+
* or truncation strategy rewrites or drops part of the conversation?
|
|
8
|
+
*/
|
|
9
|
+
/** Per-1M-token pricing used to turn token counters into an estimated cost. */
|
|
10
|
+
export interface CachePricing {
|
|
11
|
+
/** $/1M uncached prompt (input) tokens. */
|
|
12
|
+
input: number;
|
|
13
|
+
/** $/1M completion (output) tokens. */
|
|
14
|
+
output: number;
|
|
15
|
+
/** $/1M cached prompt tokens. */
|
|
16
|
+
cacheRead: number;
|
|
17
|
+
}
|
|
18
|
+
export declare const DEFAULT_CACHE_PRICING: CachePricing;
|
|
19
|
+
/** A single synthetic entry appended to (or injected into) the prompt. */
|
|
20
|
+
export interface PromptEntry {
|
|
21
|
+
/** Stable id for this entry; summaries use negative ids. */
|
|
22
|
+
id: number;
|
|
23
|
+
/** Deterministic text serialized into the request. */
|
|
24
|
+
text: string;
|
|
25
|
+
/** Estimated token count for this entry. */
|
|
26
|
+
tokens: number;
|
|
27
|
+
/** Distinguishes real turns from injected compaction summaries. */
|
|
28
|
+
kind: "turn" | "summary";
|
|
29
|
+
}
|
|
30
|
+
/** Mutable per-strategy state carried across turns. */
|
|
31
|
+
export interface StrategyState {
|
|
32
|
+
/** Ordered prompt entries currently in the active prefix (oldest first). */
|
|
33
|
+
entries: PromptEntry[];
|
|
34
|
+
}
|
|
35
|
+
/** Result of one strategy step. */
|
|
36
|
+
export interface StrategyStep {
|
|
37
|
+
/** Ordered entries to send this turn (oldest first). */
|
|
38
|
+
entries: PromptEntry[];
|
|
39
|
+
/** True when this turn dropped or summarized part of the history (a cache breakpoint). */
|
|
40
|
+
cut: boolean;
|
|
41
|
+
}
|
|
42
|
+
export interface CacheStrategy {
|
|
43
|
+
name: string;
|
|
44
|
+
init(): StrategyState;
|
|
45
|
+
step(state: StrategyState, turn: number, entry: PromptEntry): StrategyStep;
|
|
46
|
+
}
|
|
47
|
+
/** Configuration for the prefix-cache simulator. */
|
|
48
|
+
export interface PrefixCacheConfig {
|
|
49
|
+
sessionId?: string;
|
|
50
|
+
/** OpenAI only caches prefixes of at least this many tokens. */
|
|
51
|
+
minCacheTokens: number;
|
|
52
|
+
/** "none" disables cache accounting entirely (no reads/writes). */
|
|
53
|
+
cacheRetention?: "none" | "short" | "long";
|
|
54
|
+
}
|
|
55
|
+
/** One turn of prefix-cache accounting. */
|
|
56
|
+
export interface PrefixCacheTurnResult {
|
|
57
|
+
/** Total prompt tokens sent (cached + uncached). */
|
|
58
|
+
promptTokens: number;
|
|
59
|
+
/** Prompt tokens served from cache. */
|
|
60
|
+
cacheReadTokens: number;
|
|
61
|
+
/** Prompt tokens not served from cache (== cacheWriteTokens in spectral usage semantics). */
|
|
62
|
+
inputTokens: number;
|
|
63
|
+
/** Prompt tokens written to cache this turn (the uncached prompt). */
|
|
64
|
+
cacheWriteTokens: number;
|
|
65
|
+
/** Completion tokens billed this turn. */
|
|
66
|
+
outputTokens: number;
|
|
67
|
+
/** Raw longest-common-prefix token count before the minimum-cache threshold. */
|
|
68
|
+
commonPrefixTokens: number;
|
|
69
|
+
/** Whether any prompt tokens were served from cache this turn. */
|
|
70
|
+
cacheHit: boolean;
|
|
71
|
+
}
|
|
72
|
+
/** Aggregate result for one strategy over a full scenario. */
|
|
73
|
+
export interface BenchmarkResult {
|
|
74
|
+
name: string;
|
|
75
|
+
turns: number;
|
|
76
|
+
cutCount: number;
|
|
77
|
+
totalPromptTokens: number;
|
|
78
|
+
totalInputTokens: number;
|
|
79
|
+
totalOutputTokens: number;
|
|
80
|
+
totalCacheReadTokens: number;
|
|
81
|
+
totalCacheWriteTokens: number;
|
|
82
|
+
/** cacheRead / prompt tokens, in [0, 1]. */
|
|
83
|
+
cacheHitRate: number;
|
|
84
|
+
avgPromptTokens: number;
|
|
85
|
+
maxPromptTokens: number;
|
|
86
|
+
finalPromptTokens: number;
|
|
87
|
+
totalCost: number;
|
|
88
|
+
costPerTurn: number;
|
|
89
|
+
/** Percent cost delta vs the keep-all baseline (0 == same cost). */
|
|
90
|
+
costVsKeepAllPct: number;
|
|
91
|
+
}
|
|
92
|
+
/** Scenario configuration for a benchmark run. */
|
|
93
|
+
export interface BenchmarkConfig {
|
|
94
|
+
turns: number;
|
|
95
|
+
systemPromptTokens: number;
|
|
96
|
+
toolsTokens: number;
|
|
97
|
+
/** Prompt tokens appended per turn. Number or per-turn factory. */
|
|
98
|
+
turnPromptTokens: number | ((turn: number) => number);
|
|
99
|
+
/** Completion tokens billed per turn. Number or per-turn factory. */
|
|
100
|
+
turnOutputTokens: number | ((turn: number) => number);
|
|
101
|
+
pricing?: Partial<CachePricing>;
|
|
102
|
+
/** OpenAI only caches prefixes of at least this many tokens. */
|
|
103
|
+
minCacheTokens?: number;
|
|
104
|
+
sessionId?: string;
|
|
105
|
+
}
|
|
106
|
+
//# sourceMappingURL=types.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/types.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,+EAA+E;AAC/E,MAAM,WAAW,YAAY;IAC5B,2CAA2C;IAC3C,KAAK,EAAE,MAAM,CAAC;IACd,uCAAuC;IACvC,MAAM,EAAE,MAAM,CAAC;IACf,iCAAiC;IACjC,SAAS,EAAE,MAAM,CAAC;CAClB;AAED,eAAO,MAAM,qBAAqB,EAAE,YAKnC,CAAC;AAEF,0EAA0E;AAC1E,MAAM,WAAW,WAAW;IAC3B,4DAA4D;IAC5D,EAAE,EAAE,MAAM,CAAC;IACX,sDAAsD;IACtD,IAAI,EAAE,MAAM,CAAC;IACb,4CAA4C;IAC5C,MAAM,EAAE,MAAM,CAAC;IACf,mEAAmE;IACnE,IAAI,EAAE,MAAM,GAAG,SAAS,CAAC;CACzB;AAED,uDAAuD;AACvD,MAAM,WAAW,aAAa;IAC7B,4EAA4E;IAC5E,OAAO,EAAE,WAAW,EAAE,CAAC;CACvB;AAED,mCAAmC;AACnC,MAAM,WAAW,YAAY;IAC5B,wDAAwD;IACxD,OAAO,EAAE,WAAW,EAAE,CAAC;IACvB,0FAA0F;IAC1F,GAAG,EAAE,OAAO,CAAC;CACb;AAED,MAAM,WAAW,aAAa;IAC7B,IAAI,EAAE,MAAM,CAAC;IACb,IAAI,IAAI,aAAa,CAAC;IACtB,IAAI,CAAC,KAAK,EAAE,aAAa,EAAE,IAAI,EAAE,MAAM,EAAE,KAAK,EAAE,WAAW,GAAG,YAAY,CAAC;CAC3E;AAED,oDAAoD;AACpD,MAAM,WAAW,iBAAiB;IACjC,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gEAAgE;IAChE,cAAc,EAAE,MAAM,CAAC;IACvB,mEAAmE;IACnE,cAAc,CAAC,EAAE,MAAM,GAAG,OAAO,GAAG,MAAM,CAAC;CAC3C;AAED,2CAA2C;AAC3C,MAAM,WAAW,qBAAqB;IACrC,oDAAoD;IACpD,YAAY,EAAE,MAAM,CAAC;IACrB,uCAAuC;IACvC,eAAe,EAAE,MAAM,CAAC;IACxB,6FAA6F;IAC7F,WAAW,EAAE,MAAM,CAAC;IACpB,sEAAsE;IACtE,gBAAgB,EAAE,MAAM,CAAC;IACzB,0CAA0C;IAC1C,YAAY,EAAE,MAAM,CAAC;IACrB,gFAAgF;IAChF,kBAAkB,EAAE,MAAM,CAAC;IAC3B,kEAAkE;IAClE,QAAQ,EAAE,OAAO,CAAC;CAClB;AAED,8DAA8D;AAC9D,MAAM,WAAW,eAAe;IAC/B,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,CAAC;IACjB,iBAAiB,EAAE,MAAM,CAAC;IAC1B,gBAAgB,EAAE,MAAM,CAAC;IACzB,iBAAiB,EAAE,MAAM,CAAC;IAC1B,oBAAoB,EAAE,MAAM,CAAC;IAC7B,qBAAqB,EAAE,MAAM,CAAC;IAC9B,4CAA4C;IAC5C,YAAY,EAAE,MAAM,CAAC;IACrB,eAAe,EAAE,MAAM,CAAC;IACxB,eAAe,EAAE,MAAM,CAAC;IACxB,iBAAiB,EAAE,MAAM,CAAC;IAC1B,SAAS,EAAE,MAAM,CAAC;IAClB,WAAW,EAAE,MAAM,CAAC;IACpB,oEAAoE;IACpE,gBAAgB,EAAE,MAAM,CAAC;CACzB;AAED,kDAAkD;AAClD,MAAM,WAAW,eAAe;IAC/B,KAAK,EAAE,MAAM,CAAC;IACd,kBAAkB,EAAE,MAAM,CAAC;IAC3B,WAAW,EAAE,MAAM,CAAC;IACpB,mEAAmE;IACnE,gBAAgB,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,CAAC,CAAC;IACtD,qEAAqE;IACrE,gBAAgB,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,CAAC,CAAC;IACtD,OAAO,CAAC,EAAE,OAAO,CAAC,YAAY,CAAC,CAAC;IAChC,gEAAgE;IAChE,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,SAAS,CAAC,EAAE,MAAM,CAAC;CACnB"}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cache-benchmark model types.
|
|
3
|
+
*
|
|
4
|
+
* This is an offline, deterministic simulator of OpenAI-style automatic prompt
|
|
5
|
+
* caching. It answers one question cheaply and reproducibly: what happens to
|
|
6
|
+
* cacheRead / cacheWrite / input tokens (and therefore cost) when a compaction
|
|
7
|
+
* or truncation strategy rewrites or drops part of the conversation?
|
|
8
|
+
*/
|
|
9
|
+
export const DEFAULT_CACHE_PRICING = {
|
|
10
|
+
// Illustrative GPT-5-class pricing: cached input is 10% of uncached input.
|
|
11
|
+
input: 1.25,
|
|
12
|
+
output: 10,
|
|
13
|
+
cacheRead: 0.125,
|
|
14
|
+
};
|
|
@@ -101,22 +101,5 @@ export function getEnvApiKey(provider) {
|
|
|
101
101
|
return "<authenticated>";
|
|
102
102
|
}
|
|
103
103
|
}
|
|
104
|
-
if (provider === "amazon-bedrock") {
|
|
105
|
-
// Amazon Bedrock supports multiple credential sources:
|
|
106
|
-
// 1. AWS_PROFILE - named profile from ~/.aws/credentials
|
|
107
|
-
// 2. AWS_ACCESS_KEY_ID + AWS_SECRET_ACCESS_KEY - standard IAM keys
|
|
108
|
-
// 3. AWS_BEARER_TOKEN_BEDROCK - Bedrock bearer token
|
|
109
|
-
// 4. AWS_CONTAINER_CREDENTIALS_RELATIVE_URI - ECS task roles
|
|
110
|
-
// 5. AWS_CONTAINER_CREDENTIALS_FULL_URI - ECS task roles (full URI)
|
|
111
|
-
// 6. AWS_WEB_IDENTITY_TOKEN_FILE - IRSA (IAM Roles for Service Accounts)
|
|
112
|
-
if (process.env.AWS_PROFILE ||
|
|
113
|
-
(process.env.AWS_ACCESS_KEY_ID && process.env.AWS_SECRET_ACCESS_KEY) ||
|
|
114
|
-
process.env.AWS_BEARER_TOKEN_BEDROCK ||
|
|
115
|
-
process.env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI ||
|
|
116
|
-
process.env.AWS_CONTAINER_CREDENTIALS_FULL_URI ||
|
|
117
|
-
process.env.AWS_WEB_IDENTITY_TOKEN_FILE) {
|
|
118
|
-
return "<authenticated>";
|
|
119
|
-
}
|
|
120
|
-
}
|
|
121
104
|
return undefined;
|
|
122
105
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../../../src/sdk/ai/models.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,MAAM,EAAE,MAAM,uBAAuB,CAAC;AAC/C,OAAO,EAEN,KAAK,8BAA8B,EACnC,MAAM,0BAA0B,CAAC;AAClC,OAAO,KAAK,EAAE,GAAG,EAAE,aAAa,EAAE,KAAK,EAAE,kBAAkB,EAAE,KAAK,EAAE,MAAM,YAAY,CAAC;
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../../../src/sdk/ai/models.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,MAAM,EAAE,MAAM,uBAAuB,CAAC;AAC/C,OAAO,EAEN,KAAK,8BAA8B,EACnC,MAAM,0BAA0B,CAAC;AAClC,OAAO,KAAK,EAAE,GAAG,EAAE,aAAa,EAAE,KAAK,EAAE,kBAAkB,EAAE,KAAK,EAAE,MAAM,YAAY,CAAC;AAiCvF,KAAK,kBAAkB,GAAG,OAAO,MAAM,GAAG;IACzC,cAAc,EAAE,CAAC,OAAO,MAAM,CAAC,CAAC,cAAc,CAAC,GAC9C,MAAM,CAAC,8BAA8B,EAAE,KAAK,CAAC,wBAAwB,CAAC,CAAC,CAAC;CACzE,CAAC;AAEF,KAAK,QAAQ,CACZ,SAAS,SAAS,aAAa,EAC/B,QAAQ,SAAS,MAAM,kBAAkB,CAAC,SAAS,CAAC,IACjD,kBAAkB,CAAC,SAAS,CAAC,CAAC,QAAQ,CAAC,SAAS;IAAE,GAAG,EAAE,MAAM,IAAI,CAAA;CAAE,GAAG,CAAC,IAAI,SAAS,GAAG,GAAG,IAAI,GAAG,KAAK,CAAC,GAAG,KAAK,CAAC;AAEpH,wBAAgB,QAAQ,CAAC,SAAS,SAAS,aAAa,EAAE,QAAQ,SAAS,MAAM,kBAAkB,CAAC,SAAS,CAAC,EAC7G,QAAQ,EAAE,SAAS,EACnB,OAAO,EAAE,QAAQ,GACf,KAAK,CAAC,QAAQ,CAAC,SAAS,EAAE,QAAQ,CAAC,CAAC,CAGtC;AAED,wBAAgB,YAAY,IAAI,aAAa,EAAE,CAE9C;AAED,wBAAgB,SAAS,CAAC,SAAS,SAAS,aAAa,EACxD,QAAQ,EAAE,SAAS,GACjB,KAAK,CAAC,QAAQ,CAAC,SAAS,EAAE,MAAM,kBAAkB,CAAC,SAAS,CAAC,CAAC,CAAC,EAAE,CAGnE;AAED,wBAAgB,aAAa,CAAC,IAAI,SAAS,GAAG,EAAE,KAAK,EAAE,KAAK,CAAC,IAAI,CAAC,EAAE,KAAK,EAAE,KAAK,GAAG,KAAK,CAAC,MAAM,CAAC,CAO/F;AAID,wBAAgB,0BAA0B,CAAC,IAAI,SAAS,GAAG,EAAE,KAAK,EAAE,KAAK,CAAC,IAAI,CAAC,GAAG,kBAAkB,EAAE,CASrG;AAED,wBAAgB,kBAAkB,CAAC,IAAI,SAAS,GAAG,EAClD,KAAK,EAAE,KAAK,CAAC,IAAI,CAAC,EAClB,KAAK,EAAE,kBAAkB,GACvB,kBAAkB,CAgBpB;AAED;;;GAGG;AACH,wBAAgB,cAAc,CAAC,IAAI,SAAS,GAAG,EAC9C,CAAC,EAAE,KAAK,CAAC,IAAI,CAAC,GAAG,IAAI,GAAG,SAAS,EACjC,CAAC,EAAE,KAAK,CAAC,IAAI,CAAC,GAAG,IAAI,GAAG,SAAS,GAC/B,OAAO,CAGT"}
|