@oh-my-pi/pi-ai 18.3.3 → 18.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,18 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.3.5] - 2026-09-27
6
+
7
+ ### Breaking Changes
8
+
9
+ - Removed the stream-level Anthropic prompt-cache keep-alive: `StreamOptions.anthropicCacheRefresh`, `StreamOptions.anthropicCacheRefreshRequest`, and the zero-output refresh request path. Prompt-cache warming now lives in the coding agent's session-level cache warmer ([#12699](https://github.com/can1357/oh-my-pi/pull/12699) by [@KamijoToma](https://github.com/KamijoToma)).
10
+
11
+ ## [18.3.4] - 2026-09-27
12
+
13
+ ### Fixed
14
+
15
+ - Fixed Anthropic OAuth requests capping output at 64k tokens; they now request the model's full ceiling (128k on Opus 5.5), matching Claude Code and API-key requests
16
+
5
17
  ## [18.3.2] - 2026-09-25
6
18
 
7
19
  ### Fixed
@@ -38,6 +38,7 @@ export type * from "./providers/openai-completions.js";
38
38
  export type * from "./providers/openai-responses.js";
39
39
  export type * from "./providers/synthetic.js";
40
40
  export * from "./registry/index.js";
41
+ export { resolveCacheRetention } from "./utils.js";
41
42
  export * from "./stream.js";
42
43
  export * from "./types.js";
43
44
  export * from "./usage.js";
@@ -36,5 +36,3 @@ export declare function adoptRequiredClaudeCodeVersion(error: unknown): boolean;
36
36
  export declare const claudeToolPrefix: string;
37
37
  /** Identity block prepended by Claude Code's CLI runtime. */
38
38
  export declare const claudeCodeSystemInstruction = "You are Claude Code, Anthropic's official CLI for Claude.";
39
- /** Claude Code's per-request output-token ceiling. */
40
- export declare const CLAUDE_CODE_MAX_OUTPUT_TOKENS = 64000;
@@ -27,8 +27,7 @@ export type { StopDetails } from "./providers/anthropic-wire.js";
27
27
  export type { AssistantMessageEventStream } from "./utils/event-stream.js";
28
28
  /**
29
29
  * Ceiling on the output-token count omp requests from any OpenAI-family endpoint
30
- * (openai-responses, azure/xai responses, and openai-completions). Mirrors
31
- * Anthropic's {@link CLAUDE_CODE_MAX_OUTPUT_TOKENS}.
30
+ * (openai-responses, azure/xai responses, and openai-completions).
32
31
  *
33
32
  * Catalog `maxTokens` frequently reflects a model's context window rather than a
34
33
  * given upstream's real per-request output cap. OpenRouter, for instance,
@@ -238,21 +237,11 @@ export interface StreamOptions {
238
237
  /** @internal Stored credential row serving this request, when known. */
239
238
  credentialId?: number;
240
239
  cacheRetention?: CacheRetention;
241
- /**
242
- * Keep Anthropic's 5-minute prompt cache warm across bounded idle gaps.
243
- *
244
- * This is an ownership flag, not a general provider default: exactly one
245
- * primary agent loop sharing `providerSessionState` should enable it.
246
- * Side-channel and advisor requests must leave it unset.
247
- */
248
- anthropicCacheRefresh?: boolean;
249
240
  /**
250
241
  * Anthropic preserved-thinking behavior when a signed block no longer matches
251
242
  * its conversation prefix. Binding-capable models default to `"drop_block"`.
252
243
  */
253
244
  anthropicPrefixMismatchBehavior?: "drop_block" | "error";
254
- /** @internal Marks a replay-only Anthropic request that must use non-streaming `max_tokens: 0`. */
255
- anthropicCacheRefreshRequest?: boolean;
256
245
  /**
257
246
  * Anthropic on-demand compaction (`compact-2026-09-04` beta). Sends a
258
247
  * top-level `compaction: { type: "summarize", instructions? }` request; the
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@oh-my-pi/pi-ai",
3
- "version": "18.3.3",
3
+ "version": "18.3.5",
4
4
  "description": "Unified LLM API with automatic model discovery and provider configuration",
5
5
  "keywords": [
6
6
  "ai",
@@ -155,11 +155,11 @@
155
155
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
156
156
  },
157
157
  "dependencies": {
158
- "@oh-my-pi/omptype": "18.3.3",
159
- "@oh-my-pi/pi-catalog": "18.3.3",
160
- "@oh-my-pi/pi-natives": "18.3.3",
161
- "@oh-my-pi/pi-utils": "18.3.3",
162
- "@oh-my-pi/pi-wire": "18.3.3"
158
+ "@oh-my-pi/omptype": "18.3.5",
159
+ "@oh-my-pi/pi-catalog": "18.3.5",
160
+ "@oh-my-pi/pi-natives": "18.3.5",
161
+ "@oh-my-pi/pi-utils": "18.3.5",
162
+ "@oh-my-pi/pi-wire": "18.3.5"
163
163
  },
164
164
  "devDependencies": {
165
165
  "@types/bun": "^1.3.14"
package/src/index.ts CHANGED
@@ -39,6 +39,7 @@ export type * from "./providers/openai-completions";
39
39
  export type * from "./providers/openai-responses";
40
40
  export type * from "./providers/synthetic";
41
41
  export * from "./registry";
42
+ export { resolveCacheRetention } from "./utils";
42
43
  export * from "./stream";
43
44
  export * from "./types";
44
45
  export * from "./usage";
@@ -116,7 +116,6 @@ import {
116
116
  type TextBlockParam,
117
117
  } from "./anthropic-wire";
118
118
  import {
119
- CLAUDE_CODE_MAX_OUTPUT_TOKENS,
120
119
  claudeCodeSdkVersion,
121
120
  claudeCodeSystemInstruction,
122
121
  adoptRequiredClaudeCodeVersion,
@@ -1591,31 +1590,6 @@ export function applyAnthropicUsageExtras(usage: Usage, source: AnthropicUsageLi
1591
1590
  }
1592
1591
  }
1593
1592
 
1594
- function parseAnthropicWireUsage(value: unknown): AnthropicWireUsage | undefined {
1595
- if (!isRecord(value)) return undefined;
1596
- const cacheCreation = isRecord(value.cache_creation)
1597
- ? {
1598
- ...(typeof value.cache_creation.ephemeral_5m_input_tokens === "number"
1599
- ? { ephemeral_5m_input_tokens: value.cache_creation.ephemeral_5m_input_tokens }
1600
- : {}),
1601
- ...(typeof value.cache_creation.ephemeral_1h_input_tokens === "number"
1602
- ? { ephemeral_1h_input_tokens: value.cache_creation.ephemeral_1h_input_tokens }
1603
- : {}),
1604
- }
1605
- : undefined;
1606
- return {
1607
- ...(typeof value.input_tokens === "number" ? { input_tokens: value.input_tokens } : {}),
1608
- ...(typeof value.output_tokens === "number" ? { output_tokens: value.output_tokens } : {}),
1609
- ...(typeof value.cache_read_input_tokens === "number"
1610
- ? { cache_read_input_tokens: value.cache_read_input_tokens }
1611
- : {}),
1612
- ...(typeof value.cache_creation_input_tokens === "number"
1613
- ? { cache_creation_input_tokens: value.cache_creation_input_tokens }
1614
- : {}),
1615
- ...(cacheCreation === undefined ? {} : { cache_creation: cacheCreation }),
1616
- };
1617
- }
1618
-
1619
1593
  function parseAnthropicFallbackWireBlock(value: unknown): AnthropicFallbackContent | undefined {
1620
1594
  if (!isRecord(value) || value.type !== "fallback") return undefined;
1621
1595
  const from = isRecord(value.from) && typeof value.from.model === "string" ? value.from.model : undefined;
@@ -2072,7 +2046,6 @@ const streamAnthropicOnce = (
2072
2046
  });
2073
2047
  }
2074
2048
 
2075
- const zeroOutputCacheRefresh = options?.anthropicCacheRefreshRequest === true;
2076
2049
  let client: AnthropicMessagesClientLike;
2077
2050
  let isOAuthToken: boolean;
2078
2051
  // Retained so a Claude Code version bump can rebuild the client's fingerprint headers.
@@ -2208,7 +2181,7 @@ const streamAnthropicOnce = (
2208
2181
  model,
2209
2182
  apiKey,
2210
2183
  extraBetas,
2211
- stream: !zeroOutputCacheRefresh,
2184
+ stream: true,
2212
2185
  interleavedThinking: options?.interleavedThinking ?? true,
2213
2186
  headers: options?.headers,
2214
2187
  dynamicHeaders: copilotDynamicHeaders?.headers,
@@ -2322,89 +2295,6 @@ const streamAnthropicOnce = (
2322
2295
  const requestTimeoutMs =
2323
2296
  firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
2324
2297
 
2325
- if (zeroOutputCacheRefresh) {
2326
- const refreshParams: MessageCreateParams = { ...params, max_tokens: 0, stream: false };
2327
- // Anthropic rejects `tool_choice: {type:"tool"|"any"}` with `max_tokens: 0`
2328
- // ("tool_choice ... cannot be used when max_tokens is 0", #12597). A refresh
2329
- // replays the captured turn's payload, which can carry a forced selector
2330
- // (e.g. a forced yield). A zero-output keep-alive produces no tokens, so the
2331
- // forced choice is meaningless here — drop it so the request is accepted.
2332
- const refreshChoiceType = refreshParams.tool_choice?.type;
2333
- if (refreshChoiceType === "tool" || refreshChoiceType === "any") {
2334
- delete refreshParams.tool_choice;
2335
- }
2336
- rawRequestDump = {
2337
- provider: model.provider,
2338
- api: output.api,
2339
- model: model.id,
2340
- method: "POST",
2341
- url: `${baseUrl}/v1/messages${isOAuthToken ? "?beta=true" : ""}`,
2342
- body: refreshParams,
2343
- };
2344
- const { requestSignal } = activeAbortTracker;
2345
- // A replayed compaction block needs the beta on injected clients too.
2346
- // Route by the client's own endpoint when it exposes one.
2347
- const refreshBetaRouteUrl =
2348
- options?.client !== undefined ? (injectedClientBaseUrl(options.client) ?? baseUrl) : baseUrl;
2349
- let refreshHeaders: Record<string, string> | undefined;
2350
- if (options?.client !== undefined && !isVertexRawPredictUrl(refreshBetaRouteUrl)) {
2351
- if (carriesSignedCompaction(refreshParams)) {
2352
- refreshHeaders = mergeAnthropicBetaHeader(refreshHeaders ?? mergedCallerHeaders, COMPACTION_BETA);
2353
- }
2354
- if (carriesLegacyCompactionEdit(refreshParams)) {
2355
- refreshHeaders = mergeAnthropicBetaHeader(
2356
- refreshHeaders ?? mergedCallerHeaders,
2357
- LEGACY_COMPACTION_BETA,
2358
- );
2359
- }
2360
- }
2361
- const requestOptions = {
2362
- ...createSdkStreamRequestOptions(requestSignal, requestTimeoutMs),
2363
- maxRetries: 0,
2364
- ...(refreshHeaders ? { headers: refreshHeaders } : {}),
2365
- };
2366
- const request: unknown =
2367
- isOAuthToken && client.beta
2368
- ? client.beta.messages.create(refreshParams, requestOptions)
2369
- : client.messages.create(refreshParams, requestOptions);
2370
- if (!hasAnthropicRawResponseRequest(request)) {
2371
- throw new AIError.AnthropicStreamEnvelopeError(
2372
- "Anthropic cache refresh request did not expose a raw response",
2373
- );
2374
- }
2375
- const response = await request.asResponse();
2376
- await notifyProviderResponse(options, response, model, response.headers.get("request-id"));
2377
- const body: unknown = await response.json();
2378
- if (!isRecord(body)) {
2379
- throw new AIError.AnthropicStreamEnvelopeError("Anthropic cache refresh returned a malformed response");
2380
- }
2381
- const wireUsage = parseAnthropicWireUsage(body.usage);
2382
- if (!wireUsage) {
2383
- throw new AIError.AnthropicStreamEnvelopeError("Anthropic cache refresh response omitted usage");
2384
- }
2385
- if (typeof body.id === "string") output.responseId = body.id;
2386
- applyReportedInputTransformations(
2387
- output,
2388
- params,
2389
- providerSessionState,
2390
- body.input_transformations,
2391
- seenInputTransformations,
2392
- );
2393
- output.usage.input = wireUsage.input_tokens ?? 0;
2394
- output.usage.output = wireUsage.output_tokens ?? 0;
2395
- output.usage.cacheRead = wireUsage.cache_read_input_tokens ?? 0;
2396
- output.usage.cacheWrite = wireUsage.cache_creation_input_tokens ?? 0;
2397
- applyAnthropicUsageExtras(output.usage, wireUsage);
2398
- output.usage.totalTokens =
2399
- output.usage.input + output.usage.output + output.usage.cacheRead + output.usage.cacheWrite;
2400
- calculateCost(model, output.usage, output.timestamp);
2401
- output.duration = performance.now() - startTime;
2402
- stream.push({ type: "start", partial: output });
2403
- stream.push({ type: "done", reason: "stop", message: output });
2404
- stream.end();
2405
- return;
2406
- }
2407
-
2408
2298
  // Opt-in flag: the response parser only honors `fallback` content
2409
2299
  // blocks and `usage.iterations` when the current request opted into
2410
2300
  // server-side-fallback beta chain. Leaving `fallbacks` unset preserves
@@ -4255,6 +4145,9 @@ function extractClaudeCodeFirstUserMessageText(messages: readonly Message[]): st
4255
4145
  */
4256
4146
  const ANTHROPIC_CONTROL = Symbol("anthropicControl");
4257
4147
 
4148
+ /** `max_tokens` requested when the catalog has no output ceiling for the model. */
4149
+ const UNKNOWN_MODEL_MAX_OUTPUT_TOKENS = 64_000;
4150
+
4258
4151
  /** One control message: tool changes (removals first) and an optional per-message effort. */
4259
4152
  type AnthropicControlSpec = { toolChanges: AnthropicToolChange[]; effort?: AnthropicOutputEffort };
4260
4153
 
@@ -4738,11 +4631,9 @@ function buildParams(
4738
4631
  }
4739
4632
  const outputConfig = Object.keys(outputConfigEntries).length ? outputConfigEntries : undefined;
4740
4633
 
4741
- // Claude Code requests at most 64k output tokens; clamp only OAuth requests,
4742
- // where the wire fingerprint must match. API-key callers keep the full model
4743
- // ceiling (e.g. 128k on Opus 4.8).
4744
- const modelMaxTokens = model.maxTokens ?? CLAUDE_CODE_MAX_OUTPUT_TOKENS;
4745
- const maxOutputTokens = isOAuthToken ? Math.min(CLAUDE_CODE_MAX_OUTPUT_TOKENS, modelMaxTokens) : modelMaxTokens;
4634
+ // OAuth and API-key requests alike get the full model ceiling; Claude Code
4635
+ // itself requests 128k on Opus 5.5.
4636
+ const maxOutputTokens = model.maxTokens ?? UNKNOWN_MODEL_MAX_OUTPUT_TOKENS;
4746
4637
 
4747
4638
  // A caller-owned client targets its own endpoint: route body betas by the
4748
4639
  // client's URL when it exposes one, not the model's routing. Otherwise the
@@ -4773,7 +4664,7 @@ function buildParams(
4773
4664
  ...(systemBlocks && { system: systemBlocks }),
4774
4665
  ...(tools !== undefined && { tools }),
4775
4666
  ...(metadata && { metadata }),
4776
- max_tokens: Math.min(maxOutputTokens, options?.maxTokens ?? modelMaxTokens),
4667
+ max_tokens: Math.min(maxOutputTokens, options?.maxTokens ?? maxOutputTokens),
4777
4668
  ...(thinking && { thinking }),
4778
4669
  ...(contextManagement && { context_management: contextManagement }),
4779
4670
  ...(compactionRequest && {
@@ -70,5 +70,3 @@ export function adoptRequiredClaudeCodeVersion(error: unknown): boolean {
70
70
  export const claudeToolPrefix: string = "_";
71
71
  /** Identity block prepended by Claude Code's CLI runtime. */
72
72
  export const claudeCodeSystemInstruction = "You are Claude Code, Anthropic's official CLI for Claude.";
73
- /** Claude Code's per-request output-token ceiling. */
74
- export const CLAUDE_CODE_MAX_OUTPUT_TOKENS = 64000;
package/src/stream.ts CHANGED
@@ -24,7 +24,6 @@ import { isConcurrencyCapExclusion, isUsageLimitOutcome } from "./error/rate-lim
24
24
  import type { BedrockOptions } from "./providers/amazon-bedrock";
25
25
  import type { AnthropicOptions } from "./providers/anthropic";
26
26
  import type { AppleFoundationModelsOptions } from "./providers/apple-foundation-models";
27
- import type { MessageCreateParamsStreaming } from "./providers/anthropic-wire";
28
27
  import type { CursorOptions } from "./providers/cursor";
29
28
  import type { DevinOptions } from "./providers/devin";
30
29
  import { streamGitLabDuo } from "./providers/gitlab-duo";
@@ -62,7 +61,6 @@ import type {
62
61
  FetchImpl,
63
62
  Model,
64
63
  OptionsForApi,
65
- ProviderSessionState,
66
64
  SimpleStreamOptions,
67
65
  StreamOptions,
68
66
  ThinkingBudgets,
@@ -1205,285 +1203,6 @@ function emitBufferedEvents(stream: AssistantMessageEventStream, events: Assista
1205
1203
  stream.push(event);
1206
1204
  }
1207
1205
  }
1208
-
1209
- const ANTHROPIC_CACHE_TTL_MS = 5 * 60_000;
1210
- const ANTHROPIC_CACHE_REFRESH_LEAD_MS = 15_000;
1211
- const ANTHROPIC_CACHE_REFRESH_LIMIT = 3;
1212
- const ANTHROPIC_CACHE_REFRESH_STATE_KEY = "anthropic-cache-refresh";
1213
-
1214
- interface AnthropicCacheRefreshPlan {
1215
- refresh(controller: AbortController): Promise<number | undefined>;
1216
- }
1217
-
1218
- class AnthropicCacheRefreshState implements ProviderSessionState {
1219
- #controller: AbortController | undefined;
1220
- #generation = 0;
1221
- #plan: AnthropicCacheRefreshPlan | undefined;
1222
- #refreshesRemaining = 0;
1223
- #timer: NodeJS.Timeout | undefined;
1224
-
1225
- cancel(): void {
1226
- this.#generation++;
1227
- if (this.#timer !== undefined) {
1228
- clearTimeout(this.#timer);
1229
- this.#timer = undefined;
1230
- }
1231
- this.#controller?.abort();
1232
- this.#controller = undefined;
1233
- this.#plan = undefined;
1234
- this.#refreshesRemaining = 0;
1235
- }
1236
-
1237
- arm(plan: AnthropicCacheRefreshPlan, cacheTouchedAtMs: number): void {
1238
- this.cancel();
1239
- this.#plan = plan;
1240
- this.#refreshesRemaining = ANTHROPIC_CACHE_REFRESH_LIMIT;
1241
- this.#schedule(cacheTouchedAtMs, this.#generation);
1242
- }
1243
-
1244
- close(): void {
1245
- this.cancel();
1246
- }
1247
-
1248
- #schedule(cacheTouchedAtMs: number, generation: number): void {
1249
- const refreshAtMs = cacheTouchedAtMs + ANTHROPIC_CACHE_TTL_MS - ANTHROPIC_CACHE_REFRESH_LEAD_MS;
1250
- this.#timer = setTimeout(
1251
- () => {
1252
- this.#timer = undefined;
1253
- void this.#refresh(generation);
1254
- },
1255
- Math.max(0, refreshAtMs - Date.now()),
1256
- );
1257
- this.#timer.unref?.();
1258
- }
1259
-
1260
- async #refresh(generation: number): Promise<void> {
1261
- const plan = this.#plan;
1262
- if (generation !== this.#generation || !plan || this.#refreshesRemaining <= 0) return;
1263
-
1264
- const controller = new AbortController();
1265
- this.#controller = controller;
1266
- let cacheTouchedAtMs: number | undefined;
1267
- try {
1268
- cacheTouchedAtMs = await plan.refresh(controller);
1269
- } catch (error) {
1270
- if (generation === this.#generation && !controller.signal.aborted) {
1271
- logger.debug("Anthropic prompt-cache refresh failed", { error: String(error) });
1272
- }
1273
- }
1274
- if (generation !== this.#generation) return;
1275
-
1276
- this.#controller = undefined;
1277
- if (cacheTouchedAtMs === undefined) {
1278
- this.#plan = undefined;
1279
- this.#refreshesRemaining = 0;
1280
- return;
1281
- }
1282
-
1283
- this.#refreshesRemaining--;
1284
- if (this.#refreshesRemaining <= 0) {
1285
- this.#plan = undefined;
1286
- return;
1287
- }
1288
- this.#schedule(cacheTouchedAtMs, generation);
1289
- }
1290
- }
1291
-
1292
- function supportsAnthropicCacheRefresh<TApi extends Api>(model: Model<TApi>): boolean {
1293
- return (
1294
- model.api === "anthropic-messages" &&
1295
- model.provider === "anthropic" &&
1296
- model.transport !== "pi-native" &&
1297
- isLeakedThinkingHealExempt(model)
1298
- );
1299
- }
1300
-
1301
- function isAnthropicRefreshPayload(payload: unknown): payload is MessageCreateParamsStreaming {
1302
- return (
1303
- typeof payload === "object" &&
1304
- payload !== null &&
1305
- "messages" in payload &&
1306
- Array.isArray(payload.messages) &&
1307
- "max_tokens" in payload &&
1308
- typeof payload.max_tokens === "number"
1309
- );
1310
- }
1311
-
1312
- function isShortAnthropicCacheControl(cacheControl: unknown): boolean {
1313
- return (
1314
- typeof cacheControl === "object" &&
1315
- cacheControl !== null &&
1316
- "type" in cacheControl &&
1317
- cacheControl.type === "ephemeral" &&
1318
- (!("ttl" in cacheControl) || cacheControl.ttl !== "1h")
1319
- );
1320
- }
1321
-
1322
- function hasShortAnthropicMessageBreakpoint(payload: MessageCreateParamsStreaming): boolean {
1323
- for (const message of payload.messages) {
1324
- if (!Array.isArray(message.content)) continue;
1325
- for (const block of message.content) {
1326
- if ("cache_control" in block && isShortAnthropicCacheControl(block.cache_control)) return true;
1327
- }
1328
- }
1329
- return false;
1330
- }
1331
-
1332
- function isAnthropicGenerationEvent(event: AssistantMessageEvent): boolean {
1333
- switch (event.type) {
1334
- case "text_start":
1335
- case "thinking_start":
1336
- case "toolcall_start":
1337
- case "image_end":
1338
- return true;
1339
- case "text_delta":
1340
- case "thinking_delta":
1341
- case "toolcall_delta":
1342
- return event.delta.length > 0;
1343
- default:
1344
- return false;
1345
- }
1346
- }
1347
-
1348
- function isAnthropicThinkingActive(model: Model<Api>, payload: MessageCreateParamsStreaming): boolean {
1349
- if (payload.thinking) return payload.thinking.type !== "disabled";
1350
- return model.thinking?.mode === "anthropic-adaptive" && payload.output_config?.effort != null;
1351
- }
1352
-
1353
- function createAnthropicCacheRefreshPlan<TApi extends Api>(
1354
- model: Model<TApi>,
1355
- context: Context,
1356
- options: SimpleStreamOptions | undefined,
1357
- payload: MessageCreateParamsStreaming,
1358
- ): AnthropicCacheRefreshPlan {
1359
- const thinkingEnabled = isAnthropicThinkingActive(model, payload);
1360
- return {
1361
- async refresh(controller) {
1362
- let cacheRead = 0;
1363
- let cacheWrite = 0;
1364
- let cacheTouchedAtMs: number | undefined;
1365
- let canceledAfterGenerationStarted = false;
1366
- const response = streamSimpleRequest(model, context, {
1367
- ...options,
1368
- acceptEmptyResponse: true,
1369
- anthropicCacheRefreshRequest: !thinkingEnabled,
1370
- cacheRetention: "short",
1371
- maxTokens: thinkingEnabled ? options?.maxTokens : 0,
1372
- onPayload: () => ({
1373
- ...payload,
1374
- max_tokens: thinkingEnabled ? payload.max_tokens : 0,
1375
- }),
1376
- onResponse: () => {
1377
- cacheTouchedAtMs = Date.now();
1378
- },
1379
- onSseEvent: undefined,
1380
- signal: controller.signal,
1381
- });
1382
-
1383
- for await (const event of response) {
1384
- if ("partial" in event) {
1385
- cacheRead = event.partial.usage.cacheRead;
1386
- cacheWrite = event.partial.usage.cacheWrite;
1387
- }
1388
- if (event.type === "error") return undefined;
1389
- if (event.type === "done") {
1390
- cacheRead = event.message.usage.cacheRead;
1391
- cacheWrite = event.message.usage.cacheWrite;
1392
- return cacheTouchedAtMs !== undefined && cacheRead > 0 && cacheWrite === 0
1393
- ? cacheTouchedAtMs
1394
- : undefined;
1395
- }
1396
- if (thinkingEnabled && isAnthropicGenerationEvent(event)) {
1397
- canceledAfterGenerationStarted = true;
1398
- controller.abort();
1399
- break;
1400
- }
1401
- }
1402
-
1403
- if (canceledAfterGenerationStarted) {
1404
- try {
1405
- await response.result();
1406
- } catch (error) {
1407
- if (!controller.signal.aborted) throw error;
1408
- }
1409
- }
1410
- return cacheTouchedAtMs !== undefined && cacheRead > 0 && cacheWrite === 0 ? cacheTouchedAtMs : undefined;
1411
- },
1412
- };
1413
- }
1414
-
1415
- function streamSimpleWithAnthropicCacheRefresh<TApi extends Api>(
1416
- model: Model<TApi>,
1417
- context: Context,
1418
- options: SimpleStreamOptions | undefined,
1419
- ): AssistantMessageEventStream {
1420
- const providerSessionState = options?.providerSessionState;
1421
- if (!options?.anthropicCacheRefresh || !providerSessionState) {
1422
- return streamSimpleRequest(model, context, options);
1423
- }
1424
-
1425
- const existingState = providerSessionState.get(ANTHROPIC_CACHE_REFRESH_STATE_KEY);
1426
- if (existingState instanceof AnthropicCacheRefreshState) {
1427
- existingState.cancel();
1428
- } else if (existingState) {
1429
- return streamSimpleRequest(model, context, options);
1430
- }
1431
- if (!supportsAnthropicCacheRefresh(model) || resolveCacheRetention(options.cacheRetention) !== "short") {
1432
- return streamSimpleRequest(model, context, options);
1433
- }
1434
-
1435
- const refreshState = existingState ?? new AnthropicCacheRefreshState();
1436
- if (!existingState) providerSessionState.set(ANTHROPIC_CACHE_REFRESH_STATE_KEY, refreshState);
1437
-
1438
- let cacheTouchedAtMs: number | undefined;
1439
- let capturedPayload: MessageCreateParamsStreaming | undefined;
1440
- const inner = streamSimpleRequest(model, context, {
1441
- ...options,
1442
- onPayload: async (payload, payloadModel) => {
1443
- const replacement = await options?.onPayload?.(payload, payloadModel);
1444
- const finalPayload = replacement ?? payload;
1445
- if (isAnthropicRefreshPayload(finalPayload)) capturedPayload = finalPayload;
1446
- return replacement;
1447
- },
1448
- onResponse: async (response, responseModel) => {
1449
- cacheTouchedAtMs = Date.now();
1450
- await options?.onResponse?.(response, responseModel);
1451
- },
1452
- });
1453
- const outer = new AssistantMessageEventStream();
1454
- const armRefresh = (message: AssistantMessage): void => {
1455
- if (
1456
- message.stopReason === "error" ||
1457
- message.stopReason === "aborted" ||
1458
- message.usage.cacheRead + message.usage.cacheWrite <= 0 ||
1459
- cacheTouchedAtMs === undefined ||
1460
- capturedPayload === undefined ||
1461
- !hasShortAnthropicMessageBreakpoint(capturedPayload)
1462
- ) {
1463
- return;
1464
- }
1465
- refreshState.arm(createAnthropicCacheRefreshPlan(model, context, options, capturedPayload), cacheTouchedAtMs);
1466
- };
1467
-
1468
- void (async () => {
1469
- try {
1470
- for await (const event of inner) {
1471
- if (event.type === "done") armRefresh(event.message);
1472
- outer.push(event);
1473
- if (outer.done) return;
1474
- }
1475
- if (!outer.done) {
1476
- const result = await inner.result();
1477
- armRefresh(result);
1478
- outer.end(result);
1479
- }
1480
- } catch (error) {
1481
- outer.fail(error);
1482
- }
1483
- })();
1484
- return outer;
1485
- }
1486
-
1487
1206
  function withInferenceSessionId(options?: SimpleStreamOptions): SimpleStreamOptions {
1488
1207
  if (options?.sessionId) return options;
1489
1208
  return { ...options, sessionId: crypto.randomUUID() };
@@ -1496,7 +1215,7 @@ export function streamSimple<TApi extends Api>(
1496
1215
  ): AssistantMessageEventStream {
1497
1216
  const sessionOptions = withInferenceSessionId(options);
1498
1217
  if (!model.requiresGlyphTokenization) {
1499
- return streamSimpleWithAnthropicCacheRefresh(model, context, sessionOptions);
1218
+ return streamSimpleRequest(model, context, sessionOptions);
1500
1219
  }
1501
1220
  const codec = applyGlyphCodec(context);
1502
1221
  const execHandlers = sessionOptions.cursorExecHandlers ?? sessionOptions.execHandlers;
@@ -1509,7 +1228,7 @@ export function streamSimple<TApi extends Api>(
1509
1228
  execHandlers: wrappedExecHandlers,
1510
1229
  cursorExecHandlers: wrappedExecHandlers,
1511
1230
  };
1512
- return codec.wrap(streamSimpleWithAnthropicCacheRefresh(model, codec.context, wireOptions));
1231
+ return codec.wrap(streamSimpleRequest(model, codec.context, wireOptions));
1513
1232
  }
1514
1233
 
1515
1234
  /**
@@ -2093,7 +1812,6 @@ function mapOptionsForApi<TApi extends Api>(
2093
1812
  fetch: options?.fetch,
2094
1813
  fallbacks: options?.fallbacks,
2095
1814
  acceptEmptyResponse: options?.acceptEmptyResponse,
2096
- anthropicCacheRefreshRequest: options?.anthropicCacheRefreshRequest,
2097
1815
  anthropicPrefixMismatchBehavior: options?.anthropicPrefixMismatchBehavior,
2098
1816
  anthropicCompaction: options?.anthropicCompaction,
2099
1817
  anthropicSlowMode: options?.anthropicSlowMode,
package/src/types.ts CHANGED
@@ -60,8 +60,7 @@ export type { AssistantMessageEventStream } from "./utils/event-stream";
60
60
 
61
61
  /**
62
62
  * Ceiling on the output-token count omp requests from any OpenAI-family endpoint
63
- * (openai-responses, azure/xai responses, and openai-completions). Mirrors
64
- * Anthropic's {@link CLAUDE_CODE_MAX_OUTPUT_TOKENS}.
63
+ * (openai-responses, azure/xai responses, and openai-completions).
65
64
  *
66
65
  * Catalog `maxTokens` frequently reflects a model's context window rather than a
67
66
  * given upstream's real per-request output cap. OpenRouter, for instance,
@@ -429,21 +428,11 @@ export interface StreamOptions {
429
428
  /** @internal Stored credential row serving this request, when known. */
430
429
  credentialId?: number;
431
430
  cacheRetention?: CacheRetention;
432
- /**
433
- * Keep Anthropic's 5-minute prompt cache warm across bounded idle gaps.
434
- *
435
- * This is an ownership flag, not a general provider default: exactly one
436
- * primary agent loop sharing `providerSessionState` should enable it.
437
- * Side-channel and advisor requests must leave it unset.
438
- */
439
- anthropicCacheRefresh?: boolean;
440
431
  /**
441
432
  * Anthropic preserved-thinking behavior when a signed block no longer matches
442
433
  * its conversation prefix. Binding-capable models default to `"drop_block"`.
443
434
  */
444
435
  anthropicPrefixMismatchBehavior?: "drop_block" | "error";
445
- /** @internal Marks a replay-only Anthropic request that must use non-streaming `max_tokens: 0`. */
446
- anthropicCacheRefreshRequest?: boolean;
447
436
  /**
448
437
  * Anthropic on-demand compaction (`compact-2026-09-04` beta). Sends a
449
438
  * top-level `compaction: { type: "summarize", instructions? }` request; the