@aexol/spectral 0.9.181 → 0.9.183
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -5
- package/dist/extensions/spectral-vision-fallback.js +1 -1
- package/dist/generated/zeus/const.d.ts.map +1 -1
- package/dist/generated/zeus/const.js +2 -0
- package/dist/generated/zeus/index.d.ts +8 -0
- package/dist/generated/zeus/index.d.ts.map +1 -1
- package/dist/memory/hooks/observer-trigger.d.ts.map +1 -1
- package/dist/memory/hooks/observer-trigger.js +11 -6
- package/dist/memory/observer.d.ts.map +1 -1
- package/dist/memory/observer.js +3 -1
- package/dist/relay/client.d.ts +40 -4
- package/dist/relay/client.d.ts.map +1 -1
- package/dist/relay/client.js +351 -180
- package/dist/relay/models-fetch.d.ts +3 -3
- package/dist/relay/models-fetch.js +3 -3
- package/dist/sdk/ai/cache-benchmark/benchmark.d.ts +6 -0
- package/dist/sdk/ai/cache-benchmark/benchmark.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/benchmark.js +123 -0
- package/dist/sdk/ai/cache-benchmark/index.d.ts +7 -0
- package/dist/sdk/ai/cache-benchmark/index.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/index.js +6 -0
- package/dist/sdk/ai/cache-benchmark/prefix-cache.d.ts +18 -0
- package/dist/sdk/ai/cache-benchmark/prefix-cache.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/prefix-cache.js +60 -0
- package/dist/sdk/ai/cache-benchmark/strategies.d.ts +18 -0
- package/dist/sdk/ai/cache-benchmark/strategies.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/strategies.js +131 -0
- package/dist/sdk/ai/cache-benchmark/synthetic.d.ts +6 -0
- package/dist/sdk/ai/cache-benchmark/synthetic.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/synthetic.js +17 -0
- package/dist/sdk/ai/cache-benchmark/tokens.d.ts +8 -0
- package/dist/sdk/ai/cache-benchmark/tokens.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/tokens.js +22 -0
- package/dist/sdk/ai/cache-benchmark/types.d.ts +137 -0
- package/dist/sdk/ai/cache-benchmark/types.d.ts.map +1 -0
- package/dist/sdk/ai/cache-benchmark/types.js +14 -0
- package/dist/sdk/ai/env-api-keys.js +0 -17
- package/dist/sdk/ai/models.d.ts.map +1 -1
- package/dist/sdk/ai/models.js +10 -0
- package/dist/sdk/ai/providers/openai-completions.d.ts +1 -3
- package/dist/sdk/ai/providers/openai-completions.d.ts.map +1 -1
- package/dist/sdk/ai/providers/openai-completions.js +4 -100
- package/dist/sdk/ai/providers/transform-messages.d.ts.map +1 -1
- package/dist/sdk/ai/providers/transform-messages.js +35 -0
- package/dist/sdk/ai/types.d.ts +2 -4
- package/dist/sdk/ai/types.d.ts.map +1 -1
- package/dist/sdk/ai/utils/oauth/index.d.ts +1 -2
- package/dist/sdk/ai/utils/oauth/index.d.ts.map +1 -1
- package/dist/sdk/ai/utils/oauth/index.js +1 -3
- package/dist/sdk/coding-agent/core/agent-session.d.ts +2 -0
- package/dist/sdk/coding-agent/core/agent-session.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/agent-session.js +3 -4
- package/dist/sdk/coding-agent/core/compaction/llm-compaction.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/compaction/llm-compaction.js +2 -1
- package/dist/sdk/coding-agent/core/model-registry.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/model-registry.js +0 -1
- package/dist/sdk/coding-agent/core/model-resolver.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/model-resolver.js +0 -1
- package/dist/sdk/coding-agent/core/provider-display-names.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/provider-display-names.js +0 -2
- package/dist/sdk/coding-agent/core/system-prompt.d.ts.map +1 -1
- package/dist/sdk/coding-agent/core/system-prompt.js +52 -48
- package/dist/sdk/coding-agent/utils/shell.d.ts.map +1 -1
- package/dist/sdk/coding-agent/utils/shell.js +10 -0
- package/dist/server/agent-bridge.d.ts +6 -6
- package/dist/server/agent-bridge.d.ts.map +1 -1
- package/dist/server/agent-bridge.js +27 -106
- package/dist/server/handlers/sessions.d.ts +1 -0
- package/dist/server/handlers/sessions.d.ts.map +1 -1
- package/dist/server/handlers/sessions.js +6 -1
- package/dist/server/session-stream.d.ts.map +1 -1
- package/dist/server/session-stream.js +4 -0
- package/dist/server/storage.d.ts +14 -5
- package/dist/server/storage.d.ts.map +1 -1
- package/dist/server/storage.js +37 -10
- package/dist/server/wire.d.ts +1 -0
- package/dist/server/wire.d.ts.map +1 -1
- package/package.json +2 -3
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"benchmark.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/benchmark.ts"],"names":[],"mappings":"AAIA,OAAO,KAAK,EAAE,eAAe,EAAE,eAAe,EAAgB,aAAa,EAAE,MAAM,YAAY,CAAC;AAMhG,wBAAgB,YAAY,CAAC,QAAQ,EAAE,aAAa,EAAE,MAAM,EAAE,eAAe,GAAG,eAAe,CA0F9F;AAED,oFAAoF;AACpF,wBAAgB,iBAAiB,CAAC,UAAU,EAAE,aAAa,EAAE,EAAE,MAAM,EAAE,eAAe,GAAG,eAAe,EAAE,CASzG;AASD,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,eAAe,EAAE,GAAG,MAAM,CAuBxE"}
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
import { PrefixCacheSimulator } from "./prefix-cache.js";
|
|
2
|
+
import { createTurnEntry, resolvePerTurn } from "./synthetic.js";
|
|
3
|
+
import { fillerText, serializePrompt } from "./tokens.js";
|
|
4
|
+
import { DEFAULT_CACHE_PRICING } from "./types.js";
|
|
5
|
+
function resolvePricing(config) {
|
|
6
|
+
return { ...DEFAULT_CACHE_PRICING, ...(config.pricing ?? {}) };
|
|
7
|
+
}
|
|
8
|
+
export function runBenchmark(strategy, config) {
|
|
9
|
+
const pricing = resolvePricing(config);
|
|
10
|
+
const minCacheTokens = config.minCacheTokens ?? 1024;
|
|
11
|
+
const reReadFraction = config.reReadFraction ?? 0;
|
|
12
|
+
const systemText = fillerText(Math.max(0, config.systemPromptTokens * 4), 0);
|
|
13
|
+
const toolsText = fillerText(Math.max(0, config.toolsTokens * 4), 1);
|
|
14
|
+
const simulator = new PrefixCacheSimulator({
|
|
15
|
+
sessionId: config.sessionId ?? "benchmark",
|
|
16
|
+
minCacheTokens,
|
|
17
|
+
});
|
|
18
|
+
let state = strategy.init();
|
|
19
|
+
let cutCount = 0;
|
|
20
|
+
let totalPromptTokens = 0;
|
|
21
|
+
let totalInputTokens = 0;
|
|
22
|
+
let totalOutputTokens = 0;
|
|
23
|
+
let totalCacheReadTokens = 0;
|
|
24
|
+
let totalCacheWriteTokens = 0;
|
|
25
|
+
let maxPromptTokens = 0;
|
|
26
|
+
let finalPromptTokens = 0;
|
|
27
|
+
let totalTurnPromptTokens = 0;
|
|
28
|
+
let totalDroppedTokens = 0;
|
|
29
|
+
let totalSummaryTokens = 0;
|
|
30
|
+
let totalReReadTokens = 0;
|
|
31
|
+
for (let turn = 0; turn < config.turns; turn++) {
|
|
32
|
+
const promptTokens = resolvePerTurn(config.turnPromptTokens, turn);
|
|
33
|
+
const outputTokens = resolvePerTurn(config.turnOutputTokens, turn);
|
|
34
|
+
const entry = createTurnEntry(turn + 1, promptTokens);
|
|
35
|
+
const step = strategy.step(state, turn, entry);
|
|
36
|
+
state = { entries: step.entries };
|
|
37
|
+
if (step.cut)
|
|
38
|
+
cutCount++;
|
|
39
|
+
if (step.lost) {
|
|
40
|
+
totalDroppedTokens += step.lost.droppedTokens;
|
|
41
|
+
totalSummaryTokens += step.lost.summaryTokens;
|
|
42
|
+
const hardLost = Math.max(0, step.lost.droppedTokens - step.lost.summaryTokens);
|
|
43
|
+
totalReReadTokens += hardLost * reReadFraction;
|
|
44
|
+
}
|
|
45
|
+
totalTurnPromptTokens += promptTokens;
|
|
46
|
+
const promptText = serializePrompt(systemText, toolsText, step.entries);
|
|
47
|
+
const result = simulator.account(promptText, outputTokens);
|
|
48
|
+
totalPromptTokens += result.promptTokens;
|
|
49
|
+
totalInputTokens += result.inputTokens;
|
|
50
|
+
totalOutputTokens += result.outputTokens;
|
|
51
|
+
totalCacheReadTokens += result.cacheReadTokens;
|
|
52
|
+
totalCacheWriteTokens += result.cacheWriteTokens;
|
|
53
|
+
maxPromptTokens = Math.max(maxPromptTokens, result.promptTokens);
|
|
54
|
+
finalPromptTokens = result.promptTokens;
|
|
55
|
+
}
|
|
56
|
+
const contextLossRate = totalDroppedTokens > 0
|
|
57
|
+
? Math.min(1, Math.max(0, 1 - totalSummaryTokens / totalDroppedTokens))
|
|
58
|
+
: 0;
|
|
59
|
+
const avgTurnPromptTokens = totalTurnPromptTokens / config.turns;
|
|
60
|
+
const reReadTurns = avgTurnPromptTokens > 0 ? totalReReadTokens / avgTurnPromptTokens : 0;
|
|
61
|
+
const reReadCost = (totalReReadTokens * pricing.input) / 1_000_000;
|
|
62
|
+
const totalCost = ((totalInputTokens + totalReReadTokens) * pricing.input +
|
|
63
|
+
totalOutputTokens * pricing.output +
|
|
64
|
+
totalCacheReadTokens * pricing.cacheRead) /
|
|
65
|
+
1_000_000;
|
|
66
|
+
const cacheHitRate = totalPromptTokens > 0 ? totalCacheReadTokens / totalPromptTokens : 0;
|
|
67
|
+
return {
|
|
68
|
+
name: strategy.name,
|
|
69
|
+
turns: config.turns,
|
|
70
|
+
cutCount,
|
|
71
|
+
totalPromptTokens,
|
|
72
|
+
totalInputTokens,
|
|
73
|
+
totalOutputTokens,
|
|
74
|
+
totalCacheReadTokens,
|
|
75
|
+
totalCacheWriteTokens,
|
|
76
|
+
cacheHitRate,
|
|
77
|
+
avgPromptTokens: Math.round(totalPromptTokens / config.turns),
|
|
78
|
+
maxPromptTokens,
|
|
79
|
+
finalPromptTokens,
|
|
80
|
+
totalDroppedTokens,
|
|
81
|
+
totalSummaryTokens,
|
|
82
|
+
contextLossRate,
|
|
83
|
+
totalReReadTokens,
|
|
84
|
+
reReadTurns,
|
|
85
|
+
reReadCost,
|
|
86
|
+
totalCost,
|
|
87
|
+
costPerTurn: totalCost / config.turns,
|
|
88
|
+
costVsKeepAllPct: 0,
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
/** Run several strategies over one scenario and fill in the keep-all cost delta. */
|
|
92
|
+
export function runBenchmarkSuite(strategies, config) {
|
|
93
|
+
const results = strategies.map((strategy) => runBenchmark(strategy, config));
|
|
94
|
+
const baseline = results.find((result) => result.name === "keep-all");
|
|
95
|
+
if (baseline) {
|
|
96
|
+
for (const result of results) {
|
|
97
|
+
result.costVsKeepAllPct = (result.totalCost / baseline.totalCost - 1) * 100;
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
return results;
|
|
101
|
+
}
|
|
102
|
+
function fmt(value, digits = 2) {
|
|
103
|
+
return value.toLocaleString("en-US", {
|
|
104
|
+
minimumFractionDigits: digits,
|
|
105
|
+
maximumFractionDigits: digits,
|
|
106
|
+
});
|
|
107
|
+
}
|
|
108
|
+
export function formatBenchmarkReport(results) {
|
|
109
|
+
const lines = [];
|
|
110
|
+
lines.push("## Cache benchmark (OpenAI-style prefix cache)");
|
|
111
|
+
lines.push("");
|
|
112
|
+
lines.push("cacheHitRate = cacheRead / prompt tokens.");
|
|
113
|
+
lines.push("ctx loss % = 1 - summaryTokens/droppedTokens (100% == hard truncation destroyed everything).");
|
|
114
|
+
lines.push("re-read tokens = hard-lost context re-injected as new uncached input; re-read turns = re-read tokens / avg turn prompt. cost $ includes the re-read penalty.");
|
|
115
|
+
lines.push("");
|
|
116
|
+
lines.push("| strategy | cuts | avg prompt | cache hit % | ctx loss % | dropped ctx | re-read tok | re-read turns | cost $ | vs keep-all % |");
|
|
117
|
+
lines.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |");
|
|
118
|
+
for (const result of results) {
|
|
119
|
+
lines.push(`| ${result.name} | ${result.cutCount} | ${fmt(result.avgPromptTokens, 0)} | ${fmt(result.cacheHitRate * 100, 1)} | ${fmt(result.contextLossRate * 100, 1)} | ${fmt(result.totalDroppedTokens, 0)} | ${fmt(result.totalReReadTokens, 0)} | ${fmt(result.reReadTurns, 1)} | ${fmt(result.totalCost, 4)} | ${fmt(result.costVsKeepAllPct, 1)} |`);
|
|
120
|
+
}
|
|
121
|
+
lines.push("");
|
|
122
|
+
return lines.join("\n");
|
|
123
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/index.ts"],"names":[],"mappings":"AAAA,cAAc,gBAAgB,CAAC;AAC/B,cAAc,mBAAmB,CAAC;AAClC,cAAc,iBAAiB,CAAC;AAChC,cAAc,gBAAgB,CAAC;AAC/B,cAAc,aAAa,CAAC;AAC5B,cAAc,YAAY,CAAC"}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import type { PrefixCacheConfig, PrefixCacheTurnResult } from "./types.js";
|
|
2
|
+
/** Length of the longest common character prefix between two strings. */
|
|
3
|
+
export declare function commonPrefixLength(a: string, b: string): number;
|
|
4
|
+
/**
|
|
5
|
+
* Deterministic simulator of OpenAI-style automatic prompt caching.
|
|
6
|
+
*
|
|
7
|
+
* Caching is scoped to `sessionId` and matches the longest common prefix with
|
|
8
|
+
* the previous request in that session. A prefix must reach `minCacheTokens`
|
|
9
|
+
* to be served from cache at all (OpenAI's real threshold is ~1024 tokens).
|
|
10
|
+
*/
|
|
11
|
+
export declare class PrefixCacheSimulator {
|
|
12
|
+
private readonly config;
|
|
13
|
+
private lastPrompt;
|
|
14
|
+
constructor(config: PrefixCacheConfig);
|
|
15
|
+
/** Account one provider request. `outputTokens` does not affect cache, only reporting. */
|
|
16
|
+
account(promptText: string, outputTokens: number): PrefixCacheTurnResult;
|
|
17
|
+
}
|
|
18
|
+
//# sourceMappingURL=prefix-cache.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"prefix-cache.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/prefix-cache.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,iBAAiB,EAAE,qBAAqB,EAAE,MAAM,YAAY,CAAC;AAE3E,yEAAyE;AACzE,wBAAgB,kBAAkB,CAAC,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,MAAM,GAAG,MAAM,CAO/D;AAED;;;;;;GAMG;AACH,qBAAa,oBAAoB;IAGpB,OAAO,CAAC,QAAQ,CAAC,MAAM;IAFnC,OAAO,CAAC,UAAU,CAAqB;gBAEV,MAAM,EAAE,iBAAiB;IAEtD,0FAA0F;IAC1F,OAAO,CAAC,UAAU,EAAE,MAAM,EAAE,YAAY,EAAE,MAAM,GAAG,qBAAqB;CAwCxE"}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { estimateTextTokens } from "./tokens.js";
|
|
2
|
+
/** Length of the longest common character prefix between two strings. */
|
|
3
|
+
export function commonPrefixLength(a, b) {
|
|
4
|
+
const length = Math.min(a.length, b.length);
|
|
5
|
+
let index = 0;
|
|
6
|
+
while (index < length && a[index] === b[index]) {
|
|
7
|
+
index++;
|
|
8
|
+
}
|
|
9
|
+
return index;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* Deterministic simulator of OpenAI-style automatic prompt caching.
|
|
13
|
+
*
|
|
14
|
+
* Caching is scoped to `sessionId` and matches the longest common prefix with
|
|
15
|
+
* the previous request in that session. A prefix must reach `minCacheTokens`
|
|
16
|
+
* to be served from cache at all (OpenAI's real threshold is ~1024 tokens).
|
|
17
|
+
*/
|
|
18
|
+
export class PrefixCacheSimulator {
|
|
19
|
+
config;
|
|
20
|
+
lastPrompt;
|
|
21
|
+
constructor(config) {
|
|
22
|
+
this.config = config;
|
|
23
|
+
}
|
|
24
|
+
/** Account one provider request. `outputTokens` does not affect cache, only reporting. */
|
|
25
|
+
account(promptText, outputTokens) {
|
|
26
|
+
const promptTokens = estimateTextTokens(promptText);
|
|
27
|
+
if (this.config.cacheRetention === "none" || !this.config.sessionId) {
|
|
28
|
+
return {
|
|
29
|
+
promptTokens,
|
|
30
|
+
cacheReadTokens: 0,
|
|
31
|
+
inputTokens: promptTokens,
|
|
32
|
+
cacheWriteTokens: promptTokens,
|
|
33
|
+
outputTokens,
|
|
34
|
+
commonPrefixTokens: 0,
|
|
35
|
+
cacheHit: false,
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
let commonPrefixTokens = 0;
|
|
39
|
+
let cacheReadTokens = 0;
|
|
40
|
+
let inputTokens = promptTokens;
|
|
41
|
+
if (this.lastPrompt !== undefined) {
|
|
42
|
+
const prefixChars = commonPrefixLength(this.lastPrompt, promptText);
|
|
43
|
+
commonPrefixTokens = estimateTextTokens(this.lastPrompt.slice(0, prefixChars));
|
|
44
|
+
if (commonPrefixTokens >= this.config.minCacheTokens) {
|
|
45
|
+
cacheReadTokens = commonPrefixTokens;
|
|
46
|
+
inputTokens = Math.max(0, promptTokens - cacheReadTokens);
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
this.lastPrompt = promptText;
|
|
50
|
+
return {
|
|
51
|
+
promptTokens,
|
|
52
|
+
cacheReadTokens,
|
|
53
|
+
inputTokens,
|
|
54
|
+
cacheWriteTokens: inputTokens,
|
|
55
|
+
outputTokens,
|
|
56
|
+
commonPrefixTokens,
|
|
57
|
+
cacheHit: cacheReadTokens > 0,
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import type { CacheStrategy, PromptEntry } from "./types.js";
|
|
2
|
+
/** Create a deterministic compaction summary that is byte-stable after creation. */
|
|
3
|
+
export declare function createSummaryEntry(cutTurn: number, summaryTokens: number): PromptEntry;
|
|
4
|
+
/** Baseline: never compact. Prompt grows monotonically (best cache, until overflow). */
|
|
5
|
+
export declare function keepAll(): CacheStrategy;
|
|
6
|
+
/** Continuous sliding window: always keep only the last `windowTurns` entries. */
|
|
7
|
+
export declare function truncateSliding(windowTurns: number): CacheStrategy;
|
|
8
|
+
/** Batched truncation: cut to the last `keepLastTurns` entries once per `cutEveryTurns`. */
|
|
9
|
+
export declare function truncateBatched(cutEveryTurns: number, keepLastTurns: number): CacheStrategy;
|
|
10
|
+
/** Batched summarization: replace the dropped span with a fixed-size summary. */
|
|
11
|
+
export declare function summarizeBatched(cutEveryTurns: number, keepRecentTurns: number, summaryTokens: number): CacheStrategy;
|
|
12
|
+
/**
|
|
13
|
+
* Mirror the fork's autoOlderHistory policy: when estimated context tokens exceed
|
|
14
|
+
* `contextWindow * ratio`, replace the older span with a summary and keep a raw
|
|
15
|
+
* recent tail of ~`keepRecentTokens` tokens.
|
|
16
|
+
*/
|
|
17
|
+
export declare function summarizeOnTokenRatio(contextWindow: number, ratio: number, keepRecentTokens: number, summaryTokens: number): CacheStrategy;
|
|
18
|
+
//# sourceMappingURL=strategies.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"strategies.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/strategies.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,aAAa,EAAE,WAAW,EAAiB,MAAM,YAAY,CAAC;AAU5E,oFAAoF;AACpF,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,MAAM,EAAE,aAAa,EAAE,MAAM,GAAG,WAAW,CAUtF;AAED,wFAAwF;AACxF,wBAAgB,OAAO,IAAI,aAAa,CAMvC;AAED,kFAAkF;AAClF,wBAAgB,eAAe,CAAC,WAAW,EAAE,MAAM,GAAG,aAAa,CAgBlE;AAED,4FAA4F;AAC5F,wBAAgB,eAAe,CAAC,aAAa,EAAE,MAAM,EAAE,aAAa,EAAE,MAAM,GAAG,aAAa,CAoB3F;AAED,iFAAiF;AACjF,wBAAgB,gBAAgB,CAC/B,aAAa,EAAE,MAAM,EACrB,eAAe,EAAE,MAAM,EACvB,aAAa,EAAE,MAAM,GACnB,aAAa,CAsBf;AAED;;;;GAIG;AACH,wBAAgB,qBAAqB,CACpC,aAAa,EAAE,MAAM,EACrB,KAAK,EAAE,MAAM,EACb,gBAAgB,EAAE,MAAM,EACxB,aAAa,EAAE,MAAM,GACnB,aAAa,CAmCf"}
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import { fillerText } from "./tokens.js";
|
|
2
|
+
function append(state, entry) {
|
|
3
|
+
return [...state.entries, entry];
|
|
4
|
+
}
|
|
5
|
+
function droppedTokens(entries) {
|
|
6
|
+
return entries.reduce((sum, entry) => sum + entry.tokens, 0);
|
|
7
|
+
}
|
|
8
|
+
/** Create a deterministic compaction summary that is byte-stable after creation. */
|
|
9
|
+
export function createSummaryEntry(cutTurn, summaryTokens) {
|
|
10
|
+
const header = `summary(cut=${cutTurn}):`;
|
|
11
|
+
const bodyLength = Math.max(0, summaryTokens * 4 - header.length);
|
|
12
|
+
const text = header + fillerText(bodyLength, cutTurn);
|
|
13
|
+
return {
|
|
14
|
+
id: -cutTurn,
|
|
15
|
+
kind: "summary",
|
|
16
|
+
text,
|
|
17
|
+
tokens: Math.ceil(text.length / 4),
|
|
18
|
+
};
|
|
19
|
+
}
|
|
20
|
+
/** Baseline: never compact. Prompt grows monotonically (best cache, until overflow). */
|
|
21
|
+
export function keepAll() {
|
|
22
|
+
return {
|
|
23
|
+
name: "keep-all",
|
|
24
|
+
init: () => ({ entries: [] }),
|
|
25
|
+
step: (state, _turn, entry) => ({ entries: append(state, entry), cut: false }),
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
/** Continuous sliding window: always keep only the last `windowTurns` entries. */
|
|
29
|
+
export function truncateSliding(windowTurns) {
|
|
30
|
+
return {
|
|
31
|
+
name: `truncate-sliding(window=${windowTurns})`,
|
|
32
|
+
init: () => ({ entries: [] }),
|
|
33
|
+
step: (state, _turn, entry) => {
|
|
34
|
+
const appended = append(state, entry);
|
|
35
|
+
const entries = appended.slice(-windowTurns);
|
|
36
|
+
const dropped = appended.slice(0, Math.max(0, appended.length - entries.length));
|
|
37
|
+
const cut = dropped.length > 0;
|
|
38
|
+
return {
|
|
39
|
+
entries,
|
|
40
|
+
cut,
|
|
41
|
+
lost: cut ? { droppedTokens: droppedTokens(dropped), summaryTokens: 0 } : undefined,
|
|
42
|
+
};
|
|
43
|
+
},
|
|
44
|
+
};
|
|
45
|
+
}
|
|
46
|
+
/** Batched truncation: cut to the last `keepLastTurns` entries once per `cutEveryTurns`. */
|
|
47
|
+
export function truncateBatched(cutEveryTurns, keepLastTurns) {
|
|
48
|
+
return {
|
|
49
|
+
name: `truncate-batched(every=${cutEveryTurns},keep=${keepLastTurns})`,
|
|
50
|
+
init: () => ({ entries: [] }),
|
|
51
|
+
step: (state, turn, entry) => {
|
|
52
|
+
const appended = append(state, entry);
|
|
53
|
+
// turn is 0-based; cut once a full batch has accumulated.
|
|
54
|
+
if ((turn + 1) % cutEveryTurns === 0) {
|
|
55
|
+
const entries = appended.slice(-keepLastTurns);
|
|
56
|
+
const dropped = appended.slice(0, Math.max(0, appended.length - entries.length));
|
|
57
|
+
const cut = dropped.length > 0;
|
|
58
|
+
return {
|
|
59
|
+
entries,
|
|
60
|
+
cut,
|
|
61
|
+
lost: cut ? { droppedTokens: droppedTokens(dropped), summaryTokens: 0 } : undefined,
|
|
62
|
+
};
|
|
63
|
+
}
|
|
64
|
+
return { entries: appended, cut: false };
|
|
65
|
+
},
|
|
66
|
+
};
|
|
67
|
+
}
|
|
68
|
+
/** Batched summarization: replace the dropped span with a fixed-size summary. */
|
|
69
|
+
export function summarizeBatched(cutEveryTurns, keepRecentTurns, summaryTokens) {
|
|
70
|
+
return {
|
|
71
|
+
name: `summarize-batched(every=${cutEveryTurns},keep=${keepRecentTurns},summary=${summaryTokens})`,
|
|
72
|
+
init: () => ({ entries: [] }),
|
|
73
|
+
step: (state, turn, entry) => {
|
|
74
|
+
const appended = append(state, entry);
|
|
75
|
+
if ((turn + 1) % cutEveryTurns !== 0) {
|
|
76
|
+
return { entries: appended, cut: false };
|
|
77
|
+
}
|
|
78
|
+
const recent = appended.slice(-keepRecentTurns);
|
|
79
|
+
if (recent.length === appended.length) {
|
|
80
|
+
return { entries: appended, cut: false };
|
|
81
|
+
}
|
|
82
|
+
const summary = createSummaryEntry(turn + 1, summaryTokens);
|
|
83
|
+
const dropped = appended.slice(0, appended.length - recent.length);
|
|
84
|
+
return {
|
|
85
|
+
entries: [summary, ...recent],
|
|
86
|
+
cut: true,
|
|
87
|
+
lost: { droppedTokens: droppedTokens(dropped), summaryTokens: summary.tokens },
|
|
88
|
+
};
|
|
89
|
+
},
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Mirror the fork's autoOlderHistory policy: when estimated context tokens exceed
|
|
94
|
+
* `contextWindow * ratio`, replace the older span with a summary and keep a raw
|
|
95
|
+
* recent tail of ~`keepRecentTokens` tokens.
|
|
96
|
+
*/
|
|
97
|
+
export function summarizeOnTokenRatio(contextWindow, ratio, keepRecentTokens, summaryTokens) {
|
|
98
|
+
return {
|
|
99
|
+
name: `summarize-on-token-ratio(window=${contextWindow},ratio=${ratio},keep=${keepRecentTokens},summary=${summaryTokens})`,
|
|
100
|
+
init: () => ({ entries: [] }),
|
|
101
|
+
step: (state, turn, entry) => {
|
|
102
|
+
const appended = append(state, entry);
|
|
103
|
+
const total = appended.reduce((sum, e) => sum + e.tokens, 0);
|
|
104
|
+
if (total <= Math.floor(contextWindow * ratio)) {
|
|
105
|
+
return { entries: appended, cut: false };
|
|
106
|
+
}
|
|
107
|
+
// Walk from the newest entry backward, keeping the newest suffix whose
|
|
108
|
+
// total is <= keepRecentTokens (always keeping at least the newest entry).
|
|
109
|
+
let keptTokens = 0;
|
|
110
|
+
let keepFrom = appended.length;
|
|
111
|
+
for (let i = appended.length - 1; i >= 0; i--) {
|
|
112
|
+
const candidate = appended[i];
|
|
113
|
+
if (i < appended.length - 1 && keptTokens + candidate.tokens > keepRecentTokens)
|
|
114
|
+
break;
|
|
115
|
+
keptTokens += candidate.tokens;
|
|
116
|
+
keepFrom = i;
|
|
117
|
+
}
|
|
118
|
+
const recent = appended.slice(keepFrom);
|
|
119
|
+
if (keepFrom === 0) {
|
|
120
|
+
return { entries: appended, cut: false };
|
|
121
|
+
}
|
|
122
|
+
const summary = createSummaryEntry(turn + 1, summaryTokens);
|
|
123
|
+
const dropped = appended.slice(0, keepFrom);
|
|
124
|
+
return {
|
|
125
|
+
entries: [summary, ...recent],
|
|
126
|
+
cut: true,
|
|
127
|
+
lost: { droppedTokens: droppedTokens(dropped), summaryTokens: summary.tokens },
|
|
128
|
+
};
|
|
129
|
+
},
|
|
130
|
+
};
|
|
131
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import type { PromptEntry } from "./types.js";
|
|
2
|
+
/** Build a deterministic, byte-stable prompt entry for a given turn. */
|
|
3
|
+
export declare function createTurnEntry(turn: number, tokens: number): PromptEntry;
|
|
4
|
+
/** Resolve a number-or-factory config value for a given turn. */
|
|
5
|
+
export declare function resolvePerTurn(value: number | ((turn: number) => number), turn: number): number;
|
|
6
|
+
//# sourceMappingURL=synthetic.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"synthetic.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/synthetic.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,YAAY,CAAC;AAE9C,wEAAwE;AACxE,wBAAgB,eAAe,CAAC,IAAI,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM,GAAG,WAAW,CAUzE;AAED,iEAAiE;AACjE,wBAAgB,cAAc,CAAC,KAAK,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,CAAC,EAAE,IAAI,EAAE,MAAM,GAAG,MAAM,CAE/F"}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import { fillerText } from "./tokens.js";
|
|
2
|
+
/** Build a deterministic, byte-stable prompt entry for a given turn. */
|
|
3
|
+
export function createTurnEntry(turn, tokens) {
|
|
4
|
+
const header = `turn:${turn}:`;
|
|
5
|
+
const bodyLength = Math.max(0, tokens * 4 - header.length);
|
|
6
|
+
const text = header + fillerText(bodyLength, turn);
|
|
7
|
+
return {
|
|
8
|
+
id: turn,
|
|
9
|
+
kind: "turn",
|
|
10
|
+
text,
|
|
11
|
+
tokens: Math.ceil(text.length / 4),
|
|
12
|
+
};
|
|
13
|
+
}
|
|
14
|
+
/** Resolve a number-or-factory config value for a given turn. */
|
|
15
|
+
export function resolvePerTurn(value, turn) {
|
|
16
|
+
return typeof value === "function" ? value(turn) : value;
|
|
17
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
import type { PromptEntry } from "./types.js";
|
|
2
|
+
/** Conservative chars/4 token heuristic, matching cli-node's estimateTokens. */
|
|
3
|
+
export declare function estimateTextTokens(text: string): number;
|
|
4
|
+
/** Deterministic filler of exactly `length` chars (a-z cycling, offset by `seed`). */
|
|
5
|
+
export declare function fillerText(length: number, seed?: number): string;
|
|
6
|
+
/** Join a stable head and ordered entries into one deterministic prompt string. */
|
|
7
|
+
export declare function serializePrompt(systemText: string, toolsText: string, entries: PromptEntry[]): string;
|
|
8
|
+
//# sourceMappingURL=tokens.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tokens.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/tokens.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,YAAY,CAAC;AAE9C,gFAAgF;AAChF,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEvD;AAED,sFAAsF;AACtF,wBAAgB,UAAU,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,SAAI,GAAG,MAAM,CAQ3D;AAED,mFAAmF;AACnF,wBAAgB,eAAe,CAAC,UAAU,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,OAAO,EAAE,WAAW,EAAE,GAAG,MAAM,CAMrG"}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/** Conservative chars/4 token heuristic, matching cli-node's estimateTokens. */
|
|
2
|
+
export function estimateTextTokens(text) {
|
|
3
|
+
return Math.ceil(text.length / 4);
|
|
4
|
+
}
|
|
5
|
+
/** Deterministic filler of exactly `length` chars (a-z cycling, offset by `seed`). */
|
|
6
|
+
export function fillerText(length, seed = 0) {
|
|
7
|
+
const base = "abcdefghijklmnopqrstuvwxyz";
|
|
8
|
+
const offset = ((seed % base.length) + base.length) % base.length;
|
|
9
|
+
let out = "";
|
|
10
|
+
for (let i = 0; i < length; i++) {
|
|
11
|
+
out += base[(offset + i) % base.length];
|
|
12
|
+
}
|
|
13
|
+
return out;
|
|
14
|
+
}
|
|
15
|
+
/** Join a stable head and ordered entries into one deterministic prompt string. */
|
|
16
|
+
export function serializePrompt(systemText, toolsText, entries) {
|
|
17
|
+
const parts = [`[system]${systemText}`, `[tools]${toolsText}`];
|
|
18
|
+
for (const entry of entries) {
|
|
19
|
+
parts.push(`[${entry.kind}:${entry.id}]${entry.text}`);
|
|
20
|
+
}
|
|
21
|
+
return parts.join("\n");
|
|
22
|
+
}
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cache-benchmark model types.
|
|
3
|
+
*
|
|
4
|
+
* This is an offline, deterministic simulator of OpenAI-style automatic prompt
|
|
5
|
+
* caching. It answers one question cheaply and reproducibly: what happens to
|
|
6
|
+
* cacheRead / cacheWrite / input tokens (and therefore cost) when a compaction
|
|
7
|
+
* or truncation strategy rewrites or drops part of the conversation?
|
|
8
|
+
*/
|
|
9
|
+
/** Per-1M-token pricing used to turn token counters into an estimated cost. */
|
|
10
|
+
export interface CachePricing {
|
|
11
|
+
/** $/1M uncached prompt (input) tokens. */
|
|
12
|
+
input: number;
|
|
13
|
+
/** $/1M completion (output) tokens. */
|
|
14
|
+
output: number;
|
|
15
|
+
/** $/1M cached prompt tokens. */
|
|
16
|
+
cacheRead: number;
|
|
17
|
+
}
|
|
18
|
+
export declare const DEFAULT_CACHE_PRICING: CachePricing;
|
|
19
|
+
/** A single synthetic entry appended to (or injected into) the prompt. */
|
|
20
|
+
export interface PromptEntry {
|
|
21
|
+
/** Stable id for this entry; summaries use negative ids. */
|
|
22
|
+
id: number;
|
|
23
|
+
/** Deterministic text serialized into the request. */
|
|
24
|
+
text: string;
|
|
25
|
+
/** Estimated token count for this entry. */
|
|
26
|
+
tokens: number;
|
|
27
|
+
/** Distinguishes real turns from injected compaction summaries. */
|
|
28
|
+
kind: "turn" | "summary";
|
|
29
|
+
}
|
|
30
|
+
/** Mutable per-strategy state carried across turns. */
|
|
31
|
+
export interface StrategyState {
|
|
32
|
+
/** Ordered prompt entries currently in the active prefix (oldest first). */
|
|
33
|
+
entries: PromptEntry[];
|
|
34
|
+
}
|
|
35
|
+
/** Accounting for the context a cut destroyed (feeds quality/duration metrics). */
|
|
36
|
+
export interface LostContext {
|
|
37
|
+
/** Raw prompt tokens removed from the live prefix by this cut. */
|
|
38
|
+
droppedTokens: number;
|
|
39
|
+
/** Tokens of the replacement summary (0 for hard truncation). */
|
|
40
|
+
summaryTokens: number;
|
|
41
|
+
}
|
|
42
|
+
/** Result of one strategy step. */
|
|
43
|
+
export interface StrategyStep {
|
|
44
|
+
/** Ordered entries to send this turn (oldest first). */
|
|
45
|
+
entries: PromptEntry[];
|
|
46
|
+
/** True when this turn dropped or summarized part of the history (a cache breakpoint). */
|
|
47
|
+
cut: boolean;
|
|
48
|
+
/** Present when this turn dropped part of the history. */
|
|
49
|
+
lost?: LostContext;
|
|
50
|
+
}
|
|
51
|
+
export interface CacheStrategy {
|
|
52
|
+
name: string;
|
|
53
|
+
init(): StrategyState;
|
|
54
|
+
step(state: StrategyState, turn: number, entry: PromptEntry): StrategyStep;
|
|
55
|
+
}
|
|
56
|
+
/** Configuration for the prefix-cache simulator. */
|
|
57
|
+
export interface PrefixCacheConfig {
|
|
58
|
+
sessionId?: string;
|
|
59
|
+
/** OpenAI only caches prefixes of at least this many tokens. */
|
|
60
|
+
minCacheTokens: number;
|
|
61
|
+
/** "none" disables cache accounting entirely (no reads/writes). */
|
|
62
|
+
cacheRetention?: "none" | "short" | "long";
|
|
63
|
+
}
|
|
64
|
+
/** One turn of prefix-cache accounting. */
|
|
65
|
+
export interface PrefixCacheTurnResult {
|
|
66
|
+
/** Total prompt tokens sent (cached + uncached). */
|
|
67
|
+
promptTokens: number;
|
|
68
|
+
/** Prompt tokens served from cache. */
|
|
69
|
+
cacheReadTokens: number;
|
|
70
|
+
/** Prompt tokens not served from cache (== cacheWriteTokens in spectral usage semantics). */
|
|
71
|
+
inputTokens: number;
|
|
72
|
+
/** Prompt tokens written to cache this turn (the uncached prompt). */
|
|
73
|
+
cacheWriteTokens: number;
|
|
74
|
+
/** Completion tokens billed this turn. */
|
|
75
|
+
outputTokens: number;
|
|
76
|
+
/** Raw longest-common-prefix token count before the minimum-cache threshold. */
|
|
77
|
+
commonPrefixTokens: number;
|
|
78
|
+
/** Whether any prompt tokens were served from cache this turn. */
|
|
79
|
+
cacheHit: boolean;
|
|
80
|
+
}
|
|
81
|
+
/** Aggregate result for one strategy over a full scenario. */
|
|
82
|
+
export interface BenchmarkResult {
|
|
83
|
+
name: string;
|
|
84
|
+
turns: number;
|
|
85
|
+
cutCount: number;
|
|
86
|
+
totalPromptTokens: number;
|
|
87
|
+
totalInputTokens: number;
|
|
88
|
+
totalOutputTokens: number;
|
|
89
|
+
totalCacheReadTokens: number;
|
|
90
|
+
totalCacheWriteTokens: number;
|
|
91
|
+
/** cacheRead / prompt tokens, in [0, 1]. */
|
|
92
|
+
cacheHitRate: number;
|
|
93
|
+
avgPromptTokens: number;
|
|
94
|
+
maxPromptTokens: number;
|
|
95
|
+
finalPromptTokens: number;
|
|
96
|
+
/** Raw context tokens dropped by cuts (oldest-first history removed). */
|
|
97
|
+
totalDroppedTokens: number;
|
|
98
|
+
/** Replacement summary tokens injected by cuts. */
|
|
99
|
+
totalSummaryTokens: number;
|
|
100
|
+
/**
|
|
101
|
+
* Fraction of dropped context NOT carried forward by a summary, in [0, 1].
|
|
102
|
+
* 0 == nothing lost, 1 == hard truncation destroyed everything.
|
|
103
|
+
*/
|
|
104
|
+
contextLossRate: number;
|
|
105
|
+
/** Extra uncached input tokens from forced re-reads of hard-lost context. */
|
|
106
|
+
totalReReadTokens: number;
|
|
107
|
+
/** First-order duration proxy: how many extra turns those re-reads add. */
|
|
108
|
+
reReadTurns: number;
|
|
109
|
+
/** Dollar cost of the re-read recovery tokens. */
|
|
110
|
+
reReadCost: number;
|
|
111
|
+
totalCost: number;
|
|
112
|
+
costPerTurn: number;
|
|
113
|
+
/** Percent cost delta vs the keep-all baseline (0 == same cost). */
|
|
114
|
+
costVsKeepAllPct: number;
|
|
115
|
+
}
|
|
116
|
+
/** Scenario configuration for a benchmark run. */
|
|
117
|
+
export interface BenchmarkConfig {
|
|
118
|
+
turns: number;
|
|
119
|
+
systemPromptTokens: number;
|
|
120
|
+
toolsTokens: number;
|
|
121
|
+
/** Prompt tokens appended per turn. Number or per-turn factory. */
|
|
122
|
+
turnPromptTokens: number | ((turn: number) => number);
|
|
123
|
+
/** Completion tokens billed per turn. Number or per-turn factory. */
|
|
124
|
+
turnOutputTokens: number | ((turn: number) => number);
|
|
125
|
+
pricing?: Partial<CachePricing>;
|
|
126
|
+
/** OpenAI only caches prefixes of at least this many tokens. */
|
|
127
|
+
minCacheTokens?: number;
|
|
128
|
+
sessionId?: string;
|
|
129
|
+
/**
|
|
130
|
+
* When a cut hard-loses context (dropped without a summary), the agent must
|
|
131
|
+
* re-read it. This is the fraction of hard-lost tokens re-injected as new
|
|
132
|
+
* uncached input. 0 disables recovery modeling (default); 1 means the agent
|
|
133
|
+
* fully re-reads everything it destroyed.
|
|
134
|
+
*/
|
|
135
|
+
reReadFraction?: number;
|
|
136
|
+
}
|
|
137
|
+
//# sourceMappingURL=types.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../../src/sdk/ai/cache-benchmark/types.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,+EAA+E;AAC/E,MAAM,WAAW,YAAY;IAC5B,2CAA2C;IAC3C,KAAK,EAAE,MAAM,CAAC;IACd,uCAAuC;IACvC,MAAM,EAAE,MAAM,CAAC;IACf,iCAAiC;IACjC,SAAS,EAAE,MAAM,CAAC;CAClB;AAED,eAAO,MAAM,qBAAqB,EAAE,YAKnC,CAAC;AAEF,0EAA0E;AAC1E,MAAM,WAAW,WAAW;IAC3B,4DAA4D;IAC5D,EAAE,EAAE,MAAM,CAAC;IACX,sDAAsD;IACtD,IAAI,EAAE,MAAM,CAAC;IACb,4CAA4C;IAC5C,MAAM,EAAE,MAAM,CAAC;IACf,mEAAmE;IACnE,IAAI,EAAE,MAAM,GAAG,SAAS,CAAC;CACzB;AAED,uDAAuD;AACvD,MAAM,WAAW,aAAa;IAC7B,4EAA4E;IAC5E,OAAO,EAAE,WAAW,EAAE,CAAC;CACvB;AAED,mFAAmF;AACnF,MAAM,WAAW,WAAW;IAC3B,kEAAkE;IAClE,aAAa,EAAE,MAAM,CAAC;IACtB,iEAAiE;IACjE,aAAa,EAAE,MAAM,CAAC;CACtB;AAED,mCAAmC;AACnC,MAAM,WAAW,YAAY;IAC5B,wDAAwD;IACxD,OAAO,EAAE,WAAW,EAAE,CAAC;IACvB,0FAA0F;IAC1F,GAAG,EAAE,OAAO,CAAC;IACb,0DAA0D;IAC1D,IAAI,CAAC,EAAE,WAAW,CAAC;CACnB;AAED,MAAM,WAAW,aAAa;IAC7B,IAAI,EAAE,MAAM,CAAC;IACb,IAAI,IAAI,aAAa,CAAC;IACtB,IAAI,CAAC,KAAK,EAAE,aAAa,EAAE,IAAI,EAAE,MAAM,EAAE,KAAK,EAAE,WAAW,GAAG,YAAY,CAAC;CAC3E;AAED,oDAAoD;AACpD,MAAM,WAAW,iBAAiB;IACjC,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,gEAAgE;IAChE,cAAc,EAAE,MAAM,CAAC;IACvB,mEAAmE;IACnE,cAAc,CAAC,EAAE,MAAM,GAAG,OAAO,GAAG,MAAM,CAAC;CAC3C;AAED,2CAA2C;AAC3C,MAAM,WAAW,qBAAqB;IACrC,oDAAoD;IACpD,YAAY,EAAE,MAAM,CAAC;IACrB,uCAAuC;IACvC,eAAe,EAAE,MAAM,CAAC;IACxB,6FAA6F;IAC7F,WAAW,EAAE,MAAM,CAAC;IACpB,sEAAsE;IACtE,gBAAgB,EAAE,MAAM,CAAC;IACzB,0CAA0C;IAC1C,YAAY,EAAE,MAAM,CAAC;IACrB,gFAAgF;IAChF,kBAAkB,EAAE,MAAM,CAAC;IAC3B,kEAAkE;IAClE,QAAQ,EAAE,OAAO,CAAC;CAClB;AAED,8DAA8D;AAC9D,MAAM,WAAW,eAAe;IAC/B,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,CAAC;IACjB,iBAAiB,EAAE,MAAM,CAAC;IAC1B,gBAAgB,EAAE,MAAM,CAAC;IACzB,iBAAiB,EAAE,MAAM,CAAC;IAC1B,oBAAoB,EAAE,MAAM,CAAC;IAC7B,qBAAqB,EAAE,MAAM,CAAC;IAC9B,4CAA4C;IAC5C,YAAY,EAAE,MAAM,CAAC;IACrB,eAAe,EAAE,MAAM,CAAC;IACxB,eAAe,EAAE,MAAM,CAAC;IACxB,iBAAiB,EAAE,MAAM,CAAC;IAC1B,yEAAyE;IACzE,kBAAkB,EAAE,MAAM,CAAC;IAC3B,mDAAmD;IACnD,kBAAkB,EAAE,MAAM,CAAC;IAC3B;;;OAGG;IACH,eAAe,EAAE,MAAM,CAAC;IACxB,6EAA6E;IAC7E,iBAAiB,EAAE,MAAM,CAAC;IAC1B,2EAA2E;IAC3E,WAAW,EAAE,MAAM,CAAC;IACpB,kDAAkD;IAClD,UAAU,EAAE,MAAM,CAAC;IACnB,SAAS,EAAE,MAAM,CAAC;IAClB,WAAW,EAAE,MAAM,CAAC;IACpB,oEAAoE;IACpE,gBAAgB,EAAE,MAAM,CAAC;CACzB;AAED,kDAAkD;AAClD,MAAM,WAAW,eAAe;IAC/B,KAAK,EAAE,MAAM,CAAC;IACd,kBAAkB,EAAE,MAAM,CAAC;IAC3B,WAAW,EAAE,MAAM,CAAC;IACpB,mEAAmE;IACnE,gBAAgB,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,CAAC,CAAC;IACtD,qEAAqE;IACrE,gBAAgB,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,CAAC,CAAC;IACtD,OAAO,CAAC,EAAE,OAAO,CAAC,YAAY,CAAC,CAAC;IAChC,gEAAgE;IAChE,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB;;;;;OAKG;IACH,cAAc,CAAC,EAAE,MAAM,CAAC;CACxB"}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cache-benchmark model types.
|
|
3
|
+
*
|
|
4
|
+
* This is an offline, deterministic simulator of OpenAI-style automatic prompt
|
|
5
|
+
* caching. It answers one question cheaply and reproducibly: what happens to
|
|
6
|
+
* cacheRead / cacheWrite / input tokens (and therefore cost) when a compaction
|
|
7
|
+
* or truncation strategy rewrites or drops part of the conversation?
|
|
8
|
+
*/
|
|
9
|
+
export const DEFAULT_CACHE_PRICING = {
|
|
10
|
+
// Illustrative GPT-5-class pricing: cached input is 10% of uncached input.
|
|
11
|
+
input: 1.25,
|
|
12
|
+
output: 10,
|
|
13
|
+
cacheRead: 0.125,
|
|
14
|
+
};
|
|
@@ -101,22 +101,5 @@ export function getEnvApiKey(provider) {
|
|
|
101
101
|
return "<authenticated>";
|
|
102
102
|
}
|
|
103
103
|
}
|
|
104
|
-
if (provider === "amazon-bedrock") {
|
|
105
|
-
// Amazon Bedrock supports multiple credential sources:
|
|
106
|
-
// 1. AWS_PROFILE - named profile from ~/.aws/credentials
|
|
107
|
-
// 2. AWS_ACCESS_KEY_ID + AWS_SECRET_ACCESS_KEY - standard IAM keys
|
|
108
|
-
// 3. AWS_BEARER_TOKEN_BEDROCK - Bedrock bearer token
|
|
109
|
-
// 4. AWS_CONTAINER_CREDENTIALS_RELATIVE_URI - ECS task roles
|
|
110
|
-
// 5. AWS_CONTAINER_CREDENTIALS_FULL_URI - ECS task roles (full URI)
|
|
111
|
-
// 6. AWS_WEB_IDENTITY_TOKEN_FILE - IRSA (IAM Roles for Service Accounts)
|
|
112
|
-
if (process.env.AWS_PROFILE ||
|
|
113
|
-
(process.env.AWS_ACCESS_KEY_ID && process.env.AWS_SECRET_ACCESS_KEY) ||
|
|
114
|
-
process.env.AWS_BEARER_TOKEN_BEDROCK ||
|
|
115
|
-
process.env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI ||
|
|
116
|
-
process.env.AWS_CONTAINER_CREDENTIALS_FULL_URI ||
|
|
117
|
-
process.env.AWS_WEB_IDENTITY_TOKEN_FILE) {
|
|
118
|
-
return "<authenticated>";
|
|
119
|
-
}
|
|
120
|
-
}
|
|
121
104
|
return undefined;
|
|
122
105
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../../../src/sdk/ai/models.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,MAAM,EAAE,MAAM,uBAAuB,CAAC;AAC/C,OAAO,EAEN,KAAK,8BAA8B,EACnC,MAAM,0BAA0B,CAAC;AAClC,OAAO,KAAK,EAAE,GAAG,EAAE,aAAa,EAAE,KAAK,EAAE,kBAAkB,EAAE,KAAK,EAAE,MAAM,YAAY,CAAC;
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../../../src/sdk/ai/models.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,MAAM,EAAE,MAAM,uBAAuB,CAAC;AAC/C,OAAO,EAEN,KAAK,8BAA8B,EACnC,MAAM,0BAA0B,CAAC;AAClC,OAAO,KAAK,EAAE,GAAG,EAAE,aAAa,EAAE,KAAK,EAAE,kBAAkB,EAAE,KAAK,EAAE,MAAM,YAAY,CAAC;AAiCvF,KAAK,kBAAkB,GAAG,OAAO,MAAM,GAAG;IACzC,cAAc,EAAE,CAAC,OAAO,MAAM,CAAC,CAAC,cAAc,CAAC,GAC9C,MAAM,CAAC,8BAA8B,EAAE,KAAK,CAAC,wBAAwB,CAAC,CAAC,CAAC;CACzE,CAAC;AAEF,KAAK,QAAQ,CACZ,SAAS,SAAS,aAAa,EAC/B,QAAQ,SAAS,MAAM,kBAAkB,CAAC,SAAS,CAAC,IACjD,kBAAkB,CAAC,SAAS,CAAC,CAAC,QAAQ,CAAC,SAAS;IAAE,GAAG,EAAE,MAAM,IAAI,CAAA;CAAE,GAAG,CAAC,IAAI,SAAS,GAAG,GAAG,IAAI,GAAG,KAAK,CAAC,GAAG,KAAK,CAAC;AAEpH,wBAAgB,QAAQ,CAAC,SAAS,SAAS,aAAa,EAAE,QAAQ,SAAS,MAAM,kBAAkB,CAAC,SAAS,CAAC,EAC7G,QAAQ,EAAE,SAAS,EACnB,OAAO,EAAE,QAAQ,GACf,KAAK,CAAC,QAAQ,CAAC,SAAS,EAAE,QAAQ,CAAC,CAAC,CAGtC;AAED,wBAAgB,YAAY,IAAI,aAAa,EAAE,CAE9C;AAED,wBAAgB,SAAS,CAAC,SAAS,SAAS,aAAa,EACxD,QAAQ,EAAE,SAAS,GACjB,KAAK,CAAC,QAAQ,CAAC,SAAS,EAAE,MAAM,kBAAkB,CAAC,SAAS,CAAC,CAAC,CAAC,EAAE,CAGnE;AAED,wBAAgB,aAAa,CAAC,IAAI,SAAS,GAAG,EAAE,KAAK,EAAE,KAAK,CAAC,IAAI,CAAC,EAAE,KAAK,EAAE,KAAK,GAAG,KAAK,CAAC,MAAM,CAAC,CAO/F;AAID,wBAAgB,0BAA0B,CAAC,IAAI,SAAS,GAAG,EAAE,KAAK,EAAE,KAAK,CAAC,IAAI,CAAC,GAAG,kBAAkB,EAAE,CASrG;AAED,wBAAgB,kBAAkB,CAAC,IAAI,SAAS,GAAG,EAClD,KAAK,EAAE,KAAK,CAAC,IAAI,CAAC,EAClB,KAAK,EAAE,kBAAkB,GACvB,kBAAkB,CAgBpB;AAED;;;GAGG;AACH,wBAAgB,cAAc,CAAC,IAAI,SAAS,GAAG,EAC9C,CAAC,EAAE,KAAK,CAAC,IAAI,CAAC,GAAG,IAAI,GAAG,SAAS,EACjC,CAAC,EAAE,KAAK,CAAC,IAAI,CAAC,GAAG,IAAI,GAAG,SAAS,GAC/B,OAAO,CAGT"}
|