@oh-my-pi/pi-ai 18.3.3 → 18.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/providers/claude-code-fingerprint.d.ts +0 -2
- package/dist/types/types.d.ts +1 -12
- package/package.json +6 -6
- package/src/index.ts +1 -0
- package/src/providers/anthropic.ts +8 -117
- package/src/providers/claude-code-fingerprint.ts +0 -2
- package/src/stream.ts +2 -284
- package/src/types.ts +1 -12
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,18 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.3.5] - 2026-09-27
|
|
6
|
+
|
|
7
|
+
### Breaking Changes
|
|
8
|
+
|
|
9
|
+
- Removed the stream-level Anthropic prompt-cache keep-alive: `StreamOptions.anthropicCacheRefresh`, `StreamOptions.anthropicCacheRefreshRequest`, and the zero-output refresh request path. Prompt-cache warming now lives in the coding agent's session-level cache warmer ([#12699](https://github.com/can1357/oh-my-pi/pull/12699) by [@KamijoToma](https://github.com/KamijoToma)).
|
|
10
|
+
|
|
11
|
+
## [18.3.4] - 2026-09-27
|
|
12
|
+
|
|
13
|
+
### Fixed
|
|
14
|
+
|
|
15
|
+
- Fixed Anthropic OAuth requests capping output at 64k tokens; they now request the model's full ceiling (128k on Opus 5.5), matching Claude Code and API-key requests
|
|
16
|
+
|
|
5
17
|
## [18.3.2] - 2026-09-25
|
|
6
18
|
|
|
7
19
|
### Fixed
|
package/dist/types/index.d.ts
CHANGED
|
@@ -38,6 +38,7 @@ export type * from "./providers/openai-completions.js";
|
|
|
38
38
|
export type * from "./providers/openai-responses.js";
|
|
39
39
|
export type * from "./providers/synthetic.js";
|
|
40
40
|
export * from "./registry/index.js";
|
|
41
|
+
export { resolveCacheRetention } from "./utils.js";
|
|
41
42
|
export * from "./stream.js";
|
|
42
43
|
export * from "./types.js";
|
|
43
44
|
export * from "./usage.js";
|
|
@@ -36,5 +36,3 @@ export declare function adoptRequiredClaudeCodeVersion(error: unknown): boolean;
|
|
|
36
36
|
export declare const claudeToolPrefix: string;
|
|
37
37
|
/** Identity block prepended by Claude Code's CLI runtime. */
|
|
38
38
|
export declare const claudeCodeSystemInstruction = "You are Claude Code, Anthropic's official CLI for Claude.";
|
|
39
|
-
/** Claude Code's per-request output-token ceiling. */
|
|
40
|
-
export declare const CLAUDE_CODE_MAX_OUTPUT_TOKENS = 64000;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -27,8 +27,7 @@ export type { StopDetails } from "./providers/anthropic-wire.js";
|
|
|
27
27
|
export type { AssistantMessageEventStream } from "./utils/event-stream.js";
|
|
28
28
|
/**
|
|
29
29
|
* Ceiling on the output-token count omp requests from any OpenAI-family endpoint
|
|
30
|
-
* (openai-responses, azure/xai responses, and openai-completions).
|
|
31
|
-
* Anthropic's {@link CLAUDE_CODE_MAX_OUTPUT_TOKENS}.
|
|
30
|
+
* (openai-responses, azure/xai responses, and openai-completions).
|
|
32
31
|
*
|
|
33
32
|
* Catalog `maxTokens` frequently reflects a model's context window rather than a
|
|
34
33
|
* given upstream's real per-request output cap. OpenRouter, for instance,
|
|
@@ -238,21 +237,11 @@ export interface StreamOptions {
|
|
|
238
237
|
/** @internal Stored credential row serving this request, when known. */
|
|
239
238
|
credentialId?: number;
|
|
240
239
|
cacheRetention?: CacheRetention;
|
|
241
|
-
/**
|
|
242
|
-
* Keep Anthropic's 5-minute prompt cache warm across bounded idle gaps.
|
|
243
|
-
*
|
|
244
|
-
* This is an ownership flag, not a general provider default: exactly one
|
|
245
|
-
* primary agent loop sharing `providerSessionState` should enable it.
|
|
246
|
-
* Side-channel and advisor requests must leave it unset.
|
|
247
|
-
*/
|
|
248
|
-
anthropicCacheRefresh?: boolean;
|
|
249
240
|
/**
|
|
250
241
|
* Anthropic preserved-thinking behavior when a signed block no longer matches
|
|
251
242
|
* its conversation prefix. Binding-capable models default to `"drop_block"`.
|
|
252
243
|
*/
|
|
253
244
|
anthropicPrefixMismatchBehavior?: "drop_block" | "error";
|
|
254
|
-
/** @internal Marks a replay-only Anthropic request that must use non-streaming `max_tokens: 0`. */
|
|
255
|
-
anthropicCacheRefreshRequest?: boolean;
|
|
256
245
|
/**
|
|
257
246
|
* Anthropic on-demand compaction (`compact-2026-09-04` beta). Sends a
|
|
258
247
|
* top-level `compaction: { type: "summarize", instructions? }` request; the
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@oh-my-pi/pi-ai",
|
|
3
|
-
"version": "18.3.
|
|
3
|
+
"version": "18.3.5",
|
|
4
4
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"ai",
|
|
@@ -155,11 +155,11 @@
|
|
|
155
155
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
156
156
|
},
|
|
157
157
|
"dependencies": {
|
|
158
|
-
"@oh-my-pi/omptype": "18.3.
|
|
159
|
-
"@oh-my-pi/pi-catalog": "18.3.
|
|
160
|
-
"@oh-my-pi/pi-natives": "18.3.
|
|
161
|
-
"@oh-my-pi/pi-utils": "18.3.
|
|
162
|
-
"@oh-my-pi/pi-wire": "18.3.
|
|
158
|
+
"@oh-my-pi/omptype": "18.3.5",
|
|
159
|
+
"@oh-my-pi/pi-catalog": "18.3.5",
|
|
160
|
+
"@oh-my-pi/pi-natives": "18.3.5",
|
|
161
|
+
"@oh-my-pi/pi-utils": "18.3.5",
|
|
162
|
+
"@oh-my-pi/pi-wire": "18.3.5"
|
|
163
163
|
},
|
|
164
164
|
"devDependencies": {
|
|
165
165
|
"@types/bun": "^1.3.14"
|
package/src/index.ts
CHANGED
|
@@ -39,6 +39,7 @@ export type * from "./providers/openai-completions";
|
|
|
39
39
|
export type * from "./providers/openai-responses";
|
|
40
40
|
export type * from "./providers/synthetic";
|
|
41
41
|
export * from "./registry";
|
|
42
|
+
export { resolveCacheRetention } from "./utils";
|
|
42
43
|
export * from "./stream";
|
|
43
44
|
export * from "./types";
|
|
44
45
|
export * from "./usage";
|
|
@@ -116,7 +116,6 @@ import {
|
|
|
116
116
|
type TextBlockParam,
|
|
117
117
|
} from "./anthropic-wire";
|
|
118
118
|
import {
|
|
119
|
-
CLAUDE_CODE_MAX_OUTPUT_TOKENS,
|
|
120
119
|
claudeCodeSdkVersion,
|
|
121
120
|
claudeCodeSystemInstruction,
|
|
122
121
|
adoptRequiredClaudeCodeVersion,
|
|
@@ -1591,31 +1590,6 @@ export function applyAnthropicUsageExtras(usage: Usage, source: AnthropicUsageLi
|
|
|
1591
1590
|
}
|
|
1592
1591
|
}
|
|
1593
1592
|
|
|
1594
|
-
function parseAnthropicWireUsage(value: unknown): AnthropicWireUsage | undefined {
|
|
1595
|
-
if (!isRecord(value)) return undefined;
|
|
1596
|
-
const cacheCreation = isRecord(value.cache_creation)
|
|
1597
|
-
? {
|
|
1598
|
-
...(typeof value.cache_creation.ephemeral_5m_input_tokens === "number"
|
|
1599
|
-
? { ephemeral_5m_input_tokens: value.cache_creation.ephemeral_5m_input_tokens }
|
|
1600
|
-
: {}),
|
|
1601
|
-
...(typeof value.cache_creation.ephemeral_1h_input_tokens === "number"
|
|
1602
|
-
? { ephemeral_1h_input_tokens: value.cache_creation.ephemeral_1h_input_tokens }
|
|
1603
|
-
: {}),
|
|
1604
|
-
}
|
|
1605
|
-
: undefined;
|
|
1606
|
-
return {
|
|
1607
|
-
...(typeof value.input_tokens === "number" ? { input_tokens: value.input_tokens } : {}),
|
|
1608
|
-
...(typeof value.output_tokens === "number" ? { output_tokens: value.output_tokens } : {}),
|
|
1609
|
-
...(typeof value.cache_read_input_tokens === "number"
|
|
1610
|
-
? { cache_read_input_tokens: value.cache_read_input_tokens }
|
|
1611
|
-
: {}),
|
|
1612
|
-
...(typeof value.cache_creation_input_tokens === "number"
|
|
1613
|
-
? { cache_creation_input_tokens: value.cache_creation_input_tokens }
|
|
1614
|
-
: {}),
|
|
1615
|
-
...(cacheCreation === undefined ? {} : { cache_creation: cacheCreation }),
|
|
1616
|
-
};
|
|
1617
|
-
}
|
|
1618
|
-
|
|
1619
1593
|
function parseAnthropicFallbackWireBlock(value: unknown): AnthropicFallbackContent | undefined {
|
|
1620
1594
|
if (!isRecord(value) || value.type !== "fallback") return undefined;
|
|
1621
1595
|
const from = isRecord(value.from) && typeof value.from.model === "string" ? value.from.model : undefined;
|
|
@@ -2072,7 +2046,6 @@ const streamAnthropicOnce = (
|
|
|
2072
2046
|
});
|
|
2073
2047
|
}
|
|
2074
2048
|
|
|
2075
|
-
const zeroOutputCacheRefresh = options?.anthropicCacheRefreshRequest === true;
|
|
2076
2049
|
let client: AnthropicMessagesClientLike;
|
|
2077
2050
|
let isOAuthToken: boolean;
|
|
2078
2051
|
// Retained so a Claude Code version bump can rebuild the client's fingerprint headers.
|
|
@@ -2208,7 +2181,7 @@ const streamAnthropicOnce = (
|
|
|
2208
2181
|
model,
|
|
2209
2182
|
apiKey,
|
|
2210
2183
|
extraBetas,
|
|
2211
|
-
stream:
|
|
2184
|
+
stream: true,
|
|
2212
2185
|
interleavedThinking: options?.interleavedThinking ?? true,
|
|
2213
2186
|
headers: options?.headers,
|
|
2214
2187
|
dynamicHeaders: copilotDynamicHeaders?.headers,
|
|
@@ -2322,89 +2295,6 @@ const streamAnthropicOnce = (
|
|
|
2322
2295
|
const requestTimeoutMs =
|
|
2323
2296
|
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
|
|
2324
2297
|
|
|
2325
|
-
if (zeroOutputCacheRefresh) {
|
|
2326
|
-
const refreshParams: MessageCreateParams = { ...params, max_tokens: 0, stream: false };
|
|
2327
|
-
// Anthropic rejects `tool_choice: {type:"tool"|"any"}` with `max_tokens: 0`
|
|
2328
|
-
// ("tool_choice ... cannot be used when max_tokens is 0", #12597). A refresh
|
|
2329
|
-
// replays the captured turn's payload, which can carry a forced selector
|
|
2330
|
-
// (e.g. a forced yield). A zero-output keep-alive produces no tokens, so the
|
|
2331
|
-
// forced choice is meaningless here — drop it so the request is accepted.
|
|
2332
|
-
const refreshChoiceType = refreshParams.tool_choice?.type;
|
|
2333
|
-
if (refreshChoiceType === "tool" || refreshChoiceType === "any") {
|
|
2334
|
-
delete refreshParams.tool_choice;
|
|
2335
|
-
}
|
|
2336
|
-
rawRequestDump = {
|
|
2337
|
-
provider: model.provider,
|
|
2338
|
-
api: output.api,
|
|
2339
|
-
model: model.id,
|
|
2340
|
-
method: "POST",
|
|
2341
|
-
url: `${baseUrl}/v1/messages${isOAuthToken ? "?beta=true" : ""}`,
|
|
2342
|
-
body: refreshParams,
|
|
2343
|
-
};
|
|
2344
|
-
const { requestSignal } = activeAbortTracker;
|
|
2345
|
-
// A replayed compaction block needs the beta on injected clients too.
|
|
2346
|
-
// Route by the client's own endpoint when it exposes one.
|
|
2347
|
-
const refreshBetaRouteUrl =
|
|
2348
|
-
options?.client !== undefined ? (injectedClientBaseUrl(options.client) ?? baseUrl) : baseUrl;
|
|
2349
|
-
let refreshHeaders: Record<string, string> | undefined;
|
|
2350
|
-
if (options?.client !== undefined && !isVertexRawPredictUrl(refreshBetaRouteUrl)) {
|
|
2351
|
-
if (carriesSignedCompaction(refreshParams)) {
|
|
2352
|
-
refreshHeaders = mergeAnthropicBetaHeader(refreshHeaders ?? mergedCallerHeaders, COMPACTION_BETA);
|
|
2353
|
-
}
|
|
2354
|
-
if (carriesLegacyCompactionEdit(refreshParams)) {
|
|
2355
|
-
refreshHeaders = mergeAnthropicBetaHeader(
|
|
2356
|
-
refreshHeaders ?? mergedCallerHeaders,
|
|
2357
|
-
LEGACY_COMPACTION_BETA,
|
|
2358
|
-
);
|
|
2359
|
-
}
|
|
2360
|
-
}
|
|
2361
|
-
const requestOptions = {
|
|
2362
|
-
...createSdkStreamRequestOptions(requestSignal, requestTimeoutMs),
|
|
2363
|
-
maxRetries: 0,
|
|
2364
|
-
...(refreshHeaders ? { headers: refreshHeaders } : {}),
|
|
2365
|
-
};
|
|
2366
|
-
const request: unknown =
|
|
2367
|
-
isOAuthToken && client.beta
|
|
2368
|
-
? client.beta.messages.create(refreshParams, requestOptions)
|
|
2369
|
-
: client.messages.create(refreshParams, requestOptions);
|
|
2370
|
-
if (!hasAnthropicRawResponseRequest(request)) {
|
|
2371
|
-
throw new AIError.AnthropicStreamEnvelopeError(
|
|
2372
|
-
"Anthropic cache refresh request did not expose a raw response",
|
|
2373
|
-
);
|
|
2374
|
-
}
|
|
2375
|
-
const response = await request.asResponse();
|
|
2376
|
-
await notifyProviderResponse(options, response, model, response.headers.get("request-id"));
|
|
2377
|
-
const body: unknown = await response.json();
|
|
2378
|
-
if (!isRecord(body)) {
|
|
2379
|
-
throw new AIError.AnthropicStreamEnvelopeError("Anthropic cache refresh returned a malformed response");
|
|
2380
|
-
}
|
|
2381
|
-
const wireUsage = parseAnthropicWireUsage(body.usage);
|
|
2382
|
-
if (!wireUsage) {
|
|
2383
|
-
throw new AIError.AnthropicStreamEnvelopeError("Anthropic cache refresh response omitted usage");
|
|
2384
|
-
}
|
|
2385
|
-
if (typeof body.id === "string") output.responseId = body.id;
|
|
2386
|
-
applyReportedInputTransformations(
|
|
2387
|
-
output,
|
|
2388
|
-
params,
|
|
2389
|
-
providerSessionState,
|
|
2390
|
-
body.input_transformations,
|
|
2391
|
-
seenInputTransformations,
|
|
2392
|
-
);
|
|
2393
|
-
output.usage.input = wireUsage.input_tokens ?? 0;
|
|
2394
|
-
output.usage.output = wireUsage.output_tokens ?? 0;
|
|
2395
|
-
output.usage.cacheRead = wireUsage.cache_read_input_tokens ?? 0;
|
|
2396
|
-
output.usage.cacheWrite = wireUsage.cache_creation_input_tokens ?? 0;
|
|
2397
|
-
applyAnthropicUsageExtras(output.usage, wireUsage);
|
|
2398
|
-
output.usage.totalTokens =
|
|
2399
|
-
output.usage.input + output.usage.output + output.usage.cacheRead + output.usage.cacheWrite;
|
|
2400
|
-
calculateCost(model, output.usage, output.timestamp);
|
|
2401
|
-
output.duration = performance.now() - startTime;
|
|
2402
|
-
stream.push({ type: "start", partial: output });
|
|
2403
|
-
stream.push({ type: "done", reason: "stop", message: output });
|
|
2404
|
-
stream.end();
|
|
2405
|
-
return;
|
|
2406
|
-
}
|
|
2407
|
-
|
|
2408
2298
|
// Opt-in flag: the response parser only honors `fallback` content
|
|
2409
2299
|
// blocks and `usage.iterations` when the current request opted into
|
|
2410
2300
|
// server-side-fallback beta chain. Leaving `fallbacks` unset preserves
|
|
@@ -4255,6 +4145,9 @@ function extractClaudeCodeFirstUserMessageText(messages: readonly Message[]): st
|
|
|
4255
4145
|
*/
|
|
4256
4146
|
const ANTHROPIC_CONTROL = Symbol("anthropicControl");
|
|
4257
4147
|
|
|
4148
|
+
/** `max_tokens` requested when the catalog has no output ceiling for the model. */
|
|
4149
|
+
const UNKNOWN_MODEL_MAX_OUTPUT_TOKENS = 64_000;
|
|
4150
|
+
|
|
4258
4151
|
/** One control message: tool changes (removals first) and an optional per-message effort. */
|
|
4259
4152
|
type AnthropicControlSpec = { toolChanges: AnthropicToolChange[]; effort?: AnthropicOutputEffort };
|
|
4260
4153
|
|
|
@@ -4738,11 +4631,9 @@ function buildParams(
|
|
|
4738
4631
|
}
|
|
4739
4632
|
const outputConfig = Object.keys(outputConfigEntries).length ? outputConfigEntries : undefined;
|
|
4740
4633
|
|
|
4741
|
-
//
|
|
4742
|
-
//
|
|
4743
|
-
|
|
4744
|
-
const modelMaxTokens = model.maxTokens ?? CLAUDE_CODE_MAX_OUTPUT_TOKENS;
|
|
4745
|
-
const maxOutputTokens = isOAuthToken ? Math.min(CLAUDE_CODE_MAX_OUTPUT_TOKENS, modelMaxTokens) : modelMaxTokens;
|
|
4634
|
+
// OAuth and API-key requests alike get the full model ceiling; Claude Code
|
|
4635
|
+
// itself requests 128k on Opus 5.5.
|
|
4636
|
+
const maxOutputTokens = model.maxTokens ?? UNKNOWN_MODEL_MAX_OUTPUT_TOKENS;
|
|
4746
4637
|
|
|
4747
4638
|
// A caller-owned client targets its own endpoint: route body betas by the
|
|
4748
4639
|
// client's URL when it exposes one, not the model's routing. Otherwise the
|
|
@@ -4773,7 +4664,7 @@ function buildParams(
|
|
|
4773
4664
|
...(systemBlocks && { system: systemBlocks }),
|
|
4774
4665
|
...(tools !== undefined && { tools }),
|
|
4775
4666
|
...(metadata && { metadata }),
|
|
4776
|
-
max_tokens: Math.min(maxOutputTokens, options?.maxTokens ??
|
|
4667
|
+
max_tokens: Math.min(maxOutputTokens, options?.maxTokens ?? maxOutputTokens),
|
|
4777
4668
|
...(thinking && { thinking }),
|
|
4778
4669
|
...(contextManagement && { context_management: contextManagement }),
|
|
4779
4670
|
...(compactionRequest && {
|
|
@@ -70,5 +70,3 @@ export function adoptRequiredClaudeCodeVersion(error: unknown): boolean {
|
|
|
70
70
|
export const claudeToolPrefix: string = "_";
|
|
71
71
|
/** Identity block prepended by Claude Code's CLI runtime. */
|
|
72
72
|
export const claudeCodeSystemInstruction = "You are Claude Code, Anthropic's official CLI for Claude.";
|
|
73
|
-
/** Claude Code's per-request output-token ceiling. */
|
|
74
|
-
export const CLAUDE_CODE_MAX_OUTPUT_TOKENS = 64000;
|
package/src/stream.ts
CHANGED
|
@@ -24,7 +24,6 @@ import { isConcurrencyCapExclusion, isUsageLimitOutcome } from "./error/rate-lim
|
|
|
24
24
|
import type { BedrockOptions } from "./providers/amazon-bedrock";
|
|
25
25
|
import type { AnthropicOptions } from "./providers/anthropic";
|
|
26
26
|
import type { AppleFoundationModelsOptions } from "./providers/apple-foundation-models";
|
|
27
|
-
import type { MessageCreateParamsStreaming } from "./providers/anthropic-wire";
|
|
28
27
|
import type { CursorOptions } from "./providers/cursor";
|
|
29
28
|
import type { DevinOptions } from "./providers/devin";
|
|
30
29
|
import { streamGitLabDuo } from "./providers/gitlab-duo";
|
|
@@ -62,7 +61,6 @@ import type {
|
|
|
62
61
|
FetchImpl,
|
|
63
62
|
Model,
|
|
64
63
|
OptionsForApi,
|
|
65
|
-
ProviderSessionState,
|
|
66
64
|
SimpleStreamOptions,
|
|
67
65
|
StreamOptions,
|
|
68
66
|
ThinkingBudgets,
|
|
@@ -1205,285 +1203,6 @@ function emitBufferedEvents(stream: AssistantMessageEventStream, events: Assista
|
|
|
1205
1203
|
stream.push(event);
|
|
1206
1204
|
}
|
|
1207
1205
|
}
|
|
1208
|
-
|
|
1209
|
-
const ANTHROPIC_CACHE_TTL_MS = 5 * 60_000;
|
|
1210
|
-
const ANTHROPIC_CACHE_REFRESH_LEAD_MS = 15_000;
|
|
1211
|
-
const ANTHROPIC_CACHE_REFRESH_LIMIT = 3;
|
|
1212
|
-
const ANTHROPIC_CACHE_REFRESH_STATE_KEY = "anthropic-cache-refresh";
|
|
1213
|
-
|
|
1214
|
-
interface AnthropicCacheRefreshPlan {
|
|
1215
|
-
refresh(controller: AbortController): Promise<number | undefined>;
|
|
1216
|
-
}
|
|
1217
|
-
|
|
1218
|
-
class AnthropicCacheRefreshState implements ProviderSessionState {
|
|
1219
|
-
#controller: AbortController | undefined;
|
|
1220
|
-
#generation = 0;
|
|
1221
|
-
#plan: AnthropicCacheRefreshPlan | undefined;
|
|
1222
|
-
#refreshesRemaining = 0;
|
|
1223
|
-
#timer: NodeJS.Timeout | undefined;
|
|
1224
|
-
|
|
1225
|
-
cancel(): void {
|
|
1226
|
-
this.#generation++;
|
|
1227
|
-
if (this.#timer !== undefined) {
|
|
1228
|
-
clearTimeout(this.#timer);
|
|
1229
|
-
this.#timer = undefined;
|
|
1230
|
-
}
|
|
1231
|
-
this.#controller?.abort();
|
|
1232
|
-
this.#controller = undefined;
|
|
1233
|
-
this.#plan = undefined;
|
|
1234
|
-
this.#refreshesRemaining = 0;
|
|
1235
|
-
}
|
|
1236
|
-
|
|
1237
|
-
arm(plan: AnthropicCacheRefreshPlan, cacheTouchedAtMs: number): void {
|
|
1238
|
-
this.cancel();
|
|
1239
|
-
this.#plan = plan;
|
|
1240
|
-
this.#refreshesRemaining = ANTHROPIC_CACHE_REFRESH_LIMIT;
|
|
1241
|
-
this.#schedule(cacheTouchedAtMs, this.#generation);
|
|
1242
|
-
}
|
|
1243
|
-
|
|
1244
|
-
close(): void {
|
|
1245
|
-
this.cancel();
|
|
1246
|
-
}
|
|
1247
|
-
|
|
1248
|
-
#schedule(cacheTouchedAtMs: number, generation: number): void {
|
|
1249
|
-
const refreshAtMs = cacheTouchedAtMs + ANTHROPIC_CACHE_TTL_MS - ANTHROPIC_CACHE_REFRESH_LEAD_MS;
|
|
1250
|
-
this.#timer = setTimeout(
|
|
1251
|
-
() => {
|
|
1252
|
-
this.#timer = undefined;
|
|
1253
|
-
void this.#refresh(generation);
|
|
1254
|
-
},
|
|
1255
|
-
Math.max(0, refreshAtMs - Date.now()),
|
|
1256
|
-
);
|
|
1257
|
-
this.#timer.unref?.();
|
|
1258
|
-
}
|
|
1259
|
-
|
|
1260
|
-
async #refresh(generation: number): Promise<void> {
|
|
1261
|
-
const plan = this.#plan;
|
|
1262
|
-
if (generation !== this.#generation || !plan || this.#refreshesRemaining <= 0) return;
|
|
1263
|
-
|
|
1264
|
-
const controller = new AbortController();
|
|
1265
|
-
this.#controller = controller;
|
|
1266
|
-
let cacheTouchedAtMs: number | undefined;
|
|
1267
|
-
try {
|
|
1268
|
-
cacheTouchedAtMs = await plan.refresh(controller);
|
|
1269
|
-
} catch (error) {
|
|
1270
|
-
if (generation === this.#generation && !controller.signal.aborted) {
|
|
1271
|
-
logger.debug("Anthropic prompt-cache refresh failed", { error: String(error) });
|
|
1272
|
-
}
|
|
1273
|
-
}
|
|
1274
|
-
if (generation !== this.#generation) return;
|
|
1275
|
-
|
|
1276
|
-
this.#controller = undefined;
|
|
1277
|
-
if (cacheTouchedAtMs === undefined) {
|
|
1278
|
-
this.#plan = undefined;
|
|
1279
|
-
this.#refreshesRemaining = 0;
|
|
1280
|
-
return;
|
|
1281
|
-
}
|
|
1282
|
-
|
|
1283
|
-
this.#refreshesRemaining--;
|
|
1284
|
-
if (this.#refreshesRemaining <= 0) {
|
|
1285
|
-
this.#plan = undefined;
|
|
1286
|
-
return;
|
|
1287
|
-
}
|
|
1288
|
-
this.#schedule(cacheTouchedAtMs, generation);
|
|
1289
|
-
}
|
|
1290
|
-
}
|
|
1291
|
-
|
|
1292
|
-
function supportsAnthropicCacheRefresh<TApi extends Api>(model: Model<TApi>): boolean {
|
|
1293
|
-
return (
|
|
1294
|
-
model.api === "anthropic-messages" &&
|
|
1295
|
-
model.provider === "anthropic" &&
|
|
1296
|
-
model.transport !== "pi-native" &&
|
|
1297
|
-
isLeakedThinkingHealExempt(model)
|
|
1298
|
-
);
|
|
1299
|
-
}
|
|
1300
|
-
|
|
1301
|
-
function isAnthropicRefreshPayload(payload: unknown): payload is MessageCreateParamsStreaming {
|
|
1302
|
-
return (
|
|
1303
|
-
typeof payload === "object" &&
|
|
1304
|
-
payload !== null &&
|
|
1305
|
-
"messages" in payload &&
|
|
1306
|
-
Array.isArray(payload.messages) &&
|
|
1307
|
-
"max_tokens" in payload &&
|
|
1308
|
-
typeof payload.max_tokens === "number"
|
|
1309
|
-
);
|
|
1310
|
-
}
|
|
1311
|
-
|
|
1312
|
-
function isShortAnthropicCacheControl(cacheControl: unknown): boolean {
|
|
1313
|
-
return (
|
|
1314
|
-
typeof cacheControl === "object" &&
|
|
1315
|
-
cacheControl !== null &&
|
|
1316
|
-
"type" in cacheControl &&
|
|
1317
|
-
cacheControl.type === "ephemeral" &&
|
|
1318
|
-
(!("ttl" in cacheControl) || cacheControl.ttl !== "1h")
|
|
1319
|
-
);
|
|
1320
|
-
}
|
|
1321
|
-
|
|
1322
|
-
function hasShortAnthropicMessageBreakpoint(payload: MessageCreateParamsStreaming): boolean {
|
|
1323
|
-
for (const message of payload.messages) {
|
|
1324
|
-
if (!Array.isArray(message.content)) continue;
|
|
1325
|
-
for (const block of message.content) {
|
|
1326
|
-
if ("cache_control" in block && isShortAnthropicCacheControl(block.cache_control)) return true;
|
|
1327
|
-
}
|
|
1328
|
-
}
|
|
1329
|
-
return false;
|
|
1330
|
-
}
|
|
1331
|
-
|
|
1332
|
-
function isAnthropicGenerationEvent(event: AssistantMessageEvent): boolean {
|
|
1333
|
-
switch (event.type) {
|
|
1334
|
-
case "text_start":
|
|
1335
|
-
case "thinking_start":
|
|
1336
|
-
case "toolcall_start":
|
|
1337
|
-
case "image_end":
|
|
1338
|
-
return true;
|
|
1339
|
-
case "text_delta":
|
|
1340
|
-
case "thinking_delta":
|
|
1341
|
-
case "toolcall_delta":
|
|
1342
|
-
return event.delta.length > 0;
|
|
1343
|
-
default:
|
|
1344
|
-
return false;
|
|
1345
|
-
}
|
|
1346
|
-
}
|
|
1347
|
-
|
|
1348
|
-
function isAnthropicThinkingActive(model: Model<Api>, payload: MessageCreateParamsStreaming): boolean {
|
|
1349
|
-
if (payload.thinking) return payload.thinking.type !== "disabled";
|
|
1350
|
-
return model.thinking?.mode === "anthropic-adaptive" && payload.output_config?.effort != null;
|
|
1351
|
-
}
|
|
1352
|
-
|
|
1353
|
-
function createAnthropicCacheRefreshPlan<TApi extends Api>(
|
|
1354
|
-
model: Model<TApi>,
|
|
1355
|
-
context: Context,
|
|
1356
|
-
options: SimpleStreamOptions | undefined,
|
|
1357
|
-
payload: MessageCreateParamsStreaming,
|
|
1358
|
-
): AnthropicCacheRefreshPlan {
|
|
1359
|
-
const thinkingEnabled = isAnthropicThinkingActive(model, payload);
|
|
1360
|
-
return {
|
|
1361
|
-
async refresh(controller) {
|
|
1362
|
-
let cacheRead = 0;
|
|
1363
|
-
let cacheWrite = 0;
|
|
1364
|
-
let cacheTouchedAtMs: number | undefined;
|
|
1365
|
-
let canceledAfterGenerationStarted = false;
|
|
1366
|
-
const response = streamSimpleRequest(model, context, {
|
|
1367
|
-
...options,
|
|
1368
|
-
acceptEmptyResponse: true,
|
|
1369
|
-
anthropicCacheRefreshRequest: !thinkingEnabled,
|
|
1370
|
-
cacheRetention: "short",
|
|
1371
|
-
maxTokens: thinkingEnabled ? options?.maxTokens : 0,
|
|
1372
|
-
onPayload: () => ({
|
|
1373
|
-
...payload,
|
|
1374
|
-
max_tokens: thinkingEnabled ? payload.max_tokens : 0,
|
|
1375
|
-
}),
|
|
1376
|
-
onResponse: () => {
|
|
1377
|
-
cacheTouchedAtMs = Date.now();
|
|
1378
|
-
},
|
|
1379
|
-
onSseEvent: undefined,
|
|
1380
|
-
signal: controller.signal,
|
|
1381
|
-
});
|
|
1382
|
-
|
|
1383
|
-
for await (const event of response) {
|
|
1384
|
-
if ("partial" in event) {
|
|
1385
|
-
cacheRead = event.partial.usage.cacheRead;
|
|
1386
|
-
cacheWrite = event.partial.usage.cacheWrite;
|
|
1387
|
-
}
|
|
1388
|
-
if (event.type === "error") return undefined;
|
|
1389
|
-
if (event.type === "done") {
|
|
1390
|
-
cacheRead = event.message.usage.cacheRead;
|
|
1391
|
-
cacheWrite = event.message.usage.cacheWrite;
|
|
1392
|
-
return cacheTouchedAtMs !== undefined && cacheRead > 0 && cacheWrite === 0
|
|
1393
|
-
? cacheTouchedAtMs
|
|
1394
|
-
: undefined;
|
|
1395
|
-
}
|
|
1396
|
-
if (thinkingEnabled && isAnthropicGenerationEvent(event)) {
|
|
1397
|
-
canceledAfterGenerationStarted = true;
|
|
1398
|
-
controller.abort();
|
|
1399
|
-
break;
|
|
1400
|
-
}
|
|
1401
|
-
}
|
|
1402
|
-
|
|
1403
|
-
if (canceledAfterGenerationStarted) {
|
|
1404
|
-
try {
|
|
1405
|
-
await response.result();
|
|
1406
|
-
} catch (error) {
|
|
1407
|
-
if (!controller.signal.aborted) throw error;
|
|
1408
|
-
}
|
|
1409
|
-
}
|
|
1410
|
-
return cacheTouchedAtMs !== undefined && cacheRead > 0 && cacheWrite === 0 ? cacheTouchedAtMs : undefined;
|
|
1411
|
-
},
|
|
1412
|
-
};
|
|
1413
|
-
}
|
|
1414
|
-
|
|
1415
|
-
function streamSimpleWithAnthropicCacheRefresh<TApi extends Api>(
|
|
1416
|
-
model: Model<TApi>,
|
|
1417
|
-
context: Context,
|
|
1418
|
-
options: SimpleStreamOptions | undefined,
|
|
1419
|
-
): AssistantMessageEventStream {
|
|
1420
|
-
const providerSessionState = options?.providerSessionState;
|
|
1421
|
-
if (!options?.anthropicCacheRefresh || !providerSessionState) {
|
|
1422
|
-
return streamSimpleRequest(model, context, options);
|
|
1423
|
-
}
|
|
1424
|
-
|
|
1425
|
-
const existingState = providerSessionState.get(ANTHROPIC_CACHE_REFRESH_STATE_KEY);
|
|
1426
|
-
if (existingState instanceof AnthropicCacheRefreshState) {
|
|
1427
|
-
existingState.cancel();
|
|
1428
|
-
} else if (existingState) {
|
|
1429
|
-
return streamSimpleRequest(model, context, options);
|
|
1430
|
-
}
|
|
1431
|
-
if (!supportsAnthropicCacheRefresh(model) || resolveCacheRetention(options.cacheRetention) !== "short") {
|
|
1432
|
-
return streamSimpleRequest(model, context, options);
|
|
1433
|
-
}
|
|
1434
|
-
|
|
1435
|
-
const refreshState = existingState ?? new AnthropicCacheRefreshState();
|
|
1436
|
-
if (!existingState) providerSessionState.set(ANTHROPIC_CACHE_REFRESH_STATE_KEY, refreshState);
|
|
1437
|
-
|
|
1438
|
-
let cacheTouchedAtMs: number | undefined;
|
|
1439
|
-
let capturedPayload: MessageCreateParamsStreaming | undefined;
|
|
1440
|
-
const inner = streamSimpleRequest(model, context, {
|
|
1441
|
-
...options,
|
|
1442
|
-
onPayload: async (payload, payloadModel) => {
|
|
1443
|
-
const replacement = await options?.onPayload?.(payload, payloadModel);
|
|
1444
|
-
const finalPayload = replacement ?? payload;
|
|
1445
|
-
if (isAnthropicRefreshPayload(finalPayload)) capturedPayload = finalPayload;
|
|
1446
|
-
return replacement;
|
|
1447
|
-
},
|
|
1448
|
-
onResponse: async (response, responseModel) => {
|
|
1449
|
-
cacheTouchedAtMs = Date.now();
|
|
1450
|
-
await options?.onResponse?.(response, responseModel);
|
|
1451
|
-
},
|
|
1452
|
-
});
|
|
1453
|
-
const outer = new AssistantMessageEventStream();
|
|
1454
|
-
const armRefresh = (message: AssistantMessage): void => {
|
|
1455
|
-
if (
|
|
1456
|
-
message.stopReason === "error" ||
|
|
1457
|
-
message.stopReason === "aborted" ||
|
|
1458
|
-
message.usage.cacheRead + message.usage.cacheWrite <= 0 ||
|
|
1459
|
-
cacheTouchedAtMs === undefined ||
|
|
1460
|
-
capturedPayload === undefined ||
|
|
1461
|
-
!hasShortAnthropicMessageBreakpoint(capturedPayload)
|
|
1462
|
-
) {
|
|
1463
|
-
return;
|
|
1464
|
-
}
|
|
1465
|
-
refreshState.arm(createAnthropicCacheRefreshPlan(model, context, options, capturedPayload), cacheTouchedAtMs);
|
|
1466
|
-
};
|
|
1467
|
-
|
|
1468
|
-
void (async () => {
|
|
1469
|
-
try {
|
|
1470
|
-
for await (const event of inner) {
|
|
1471
|
-
if (event.type === "done") armRefresh(event.message);
|
|
1472
|
-
outer.push(event);
|
|
1473
|
-
if (outer.done) return;
|
|
1474
|
-
}
|
|
1475
|
-
if (!outer.done) {
|
|
1476
|
-
const result = await inner.result();
|
|
1477
|
-
armRefresh(result);
|
|
1478
|
-
outer.end(result);
|
|
1479
|
-
}
|
|
1480
|
-
} catch (error) {
|
|
1481
|
-
outer.fail(error);
|
|
1482
|
-
}
|
|
1483
|
-
})();
|
|
1484
|
-
return outer;
|
|
1485
|
-
}
|
|
1486
|
-
|
|
1487
1206
|
function withInferenceSessionId(options?: SimpleStreamOptions): SimpleStreamOptions {
|
|
1488
1207
|
if (options?.sessionId) return options;
|
|
1489
1208
|
return { ...options, sessionId: crypto.randomUUID() };
|
|
@@ -1496,7 +1215,7 @@ export function streamSimple<TApi extends Api>(
|
|
|
1496
1215
|
): AssistantMessageEventStream {
|
|
1497
1216
|
const sessionOptions = withInferenceSessionId(options);
|
|
1498
1217
|
if (!model.requiresGlyphTokenization) {
|
|
1499
|
-
return
|
|
1218
|
+
return streamSimpleRequest(model, context, sessionOptions);
|
|
1500
1219
|
}
|
|
1501
1220
|
const codec = applyGlyphCodec(context);
|
|
1502
1221
|
const execHandlers = sessionOptions.cursorExecHandlers ?? sessionOptions.execHandlers;
|
|
@@ -1509,7 +1228,7 @@ export function streamSimple<TApi extends Api>(
|
|
|
1509
1228
|
execHandlers: wrappedExecHandlers,
|
|
1510
1229
|
cursorExecHandlers: wrappedExecHandlers,
|
|
1511
1230
|
};
|
|
1512
|
-
return codec.wrap(
|
|
1231
|
+
return codec.wrap(streamSimpleRequest(model, codec.context, wireOptions));
|
|
1513
1232
|
}
|
|
1514
1233
|
|
|
1515
1234
|
/**
|
|
@@ -2093,7 +1812,6 @@ function mapOptionsForApi<TApi extends Api>(
|
|
|
2093
1812
|
fetch: options?.fetch,
|
|
2094
1813
|
fallbacks: options?.fallbacks,
|
|
2095
1814
|
acceptEmptyResponse: options?.acceptEmptyResponse,
|
|
2096
|
-
anthropicCacheRefreshRequest: options?.anthropicCacheRefreshRequest,
|
|
2097
1815
|
anthropicPrefixMismatchBehavior: options?.anthropicPrefixMismatchBehavior,
|
|
2098
1816
|
anthropicCompaction: options?.anthropicCompaction,
|
|
2099
1817
|
anthropicSlowMode: options?.anthropicSlowMode,
|
package/src/types.ts
CHANGED
|
@@ -60,8 +60,7 @@ export type { AssistantMessageEventStream } from "./utils/event-stream";
|
|
|
60
60
|
|
|
61
61
|
/**
|
|
62
62
|
* Ceiling on the output-token count omp requests from any OpenAI-family endpoint
|
|
63
|
-
* (openai-responses, azure/xai responses, and openai-completions).
|
|
64
|
-
* Anthropic's {@link CLAUDE_CODE_MAX_OUTPUT_TOKENS}.
|
|
63
|
+
* (openai-responses, azure/xai responses, and openai-completions).
|
|
65
64
|
*
|
|
66
65
|
* Catalog `maxTokens` frequently reflects a model's context window rather than a
|
|
67
66
|
* given upstream's real per-request output cap. OpenRouter, for instance,
|
|
@@ -429,21 +428,11 @@ export interface StreamOptions {
|
|
|
429
428
|
/** @internal Stored credential row serving this request, when known. */
|
|
430
429
|
credentialId?: number;
|
|
431
430
|
cacheRetention?: CacheRetention;
|
|
432
|
-
/**
|
|
433
|
-
* Keep Anthropic's 5-minute prompt cache warm across bounded idle gaps.
|
|
434
|
-
*
|
|
435
|
-
* This is an ownership flag, not a general provider default: exactly one
|
|
436
|
-
* primary agent loop sharing `providerSessionState` should enable it.
|
|
437
|
-
* Side-channel and advisor requests must leave it unset.
|
|
438
|
-
*/
|
|
439
|
-
anthropicCacheRefresh?: boolean;
|
|
440
431
|
/**
|
|
441
432
|
* Anthropic preserved-thinking behavior when a signed block no longer matches
|
|
442
433
|
* its conversation prefix. Binding-capable models default to `"drop_block"`.
|
|
443
434
|
*/
|
|
444
435
|
anthropicPrefixMismatchBehavior?: "drop_block" | "error";
|
|
445
|
-
/** @internal Marks a replay-only Anthropic request that must use non-streaming `max_tokens: 0`. */
|
|
446
|
-
anthropicCacheRefreshRequest?: boolean;
|
|
447
436
|
/**
|
|
448
437
|
* Anthropic on-demand compaction (`compact-2026-09-04` beta). Sends a
|
|
449
438
|
* top-level `compaction: { type: "summarize", instructions? }` request; the
|