@oh-my-pi/pi-ai 18.3.4 → 18.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/types.d.ts +0 -10
- package/package.json +6 -6
- package/src/index.ts +1 -0
- package/src/providers/anthropic.ts +1 -110
- package/src/stream.ts +2 -284
- package/src/types.ts +0 -10
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.3.5] - 2026-09-27
|
|
6
|
+
|
|
7
|
+
### Breaking Changes
|
|
8
|
+
|
|
9
|
+
- Removed the stream-level Anthropic prompt-cache keep-alive: `StreamOptions.anthropicCacheRefresh`, `StreamOptions.anthropicCacheRefreshRequest`, and the zero-output refresh request path. Prompt-cache warming now lives in the coding agent's session-level cache warmer ([#12699](https://github.com/can1357/oh-my-pi/pull/12699) by [@KamijoToma](https://github.com/KamijoToma)).
|
|
10
|
+
|
|
5
11
|
## [18.3.4] - 2026-09-27
|
|
6
12
|
|
|
7
13
|
### Fixed
|
package/dist/types/index.d.ts
CHANGED
|
@@ -38,6 +38,7 @@ export type * from "./providers/openai-completions.js";
|
|
|
38
38
|
export type * from "./providers/openai-responses.js";
|
|
39
39
|
export type * from "./providers/synthetic.js";
|
|
40
40
|
export * from "./registry/index.js";
|
|
41
|
+
export { resolveCacheRetention } from "./utils.js";
|
|
41
42
|
export * from "./stream.js";
|
|
42
43
|
export * from "./types.js";
|
|
43
44
|
export * from "./usage.js";
|
package/dist/types/types.d.ts
CHANGED
|
@@ -237,21 +237,11 @@ export interface StreamOptions {
|
|
|
237
237
|
/** @internal Stored credential row serving this request, when known. */
|
|
238
238
|
credentialId?: number;
|
|
239
239
|
cacheRetention?: CacheRetention;
|
|
240
|
-
/**
|
|
241
|
-
* Keep Anthropic's 5-minute prompt cache warm across bounded idle gaps.
|
|
242
|
-
*
|
|
243
|
-
* This is an ownership flag, not a general provider default: exactly one
|
|
244
|
-
* primary agent loop sharing `providerSessionState` should enable it.
|
|
245
|
-
* Side-channel and advisor requests must leave it unset.
|
|
246
|
-
*/
|
|
247
|
-
anthropicCacheRefresh?: boolean;
|
|
248
240
|
/**
|
|
249
241
|
* Anthropic preserved-thinking behavior when a signed block no longer matches
|
|
250
242
|
* its conversation prefix. Binding-capable models default to `"drop_block"`.
|
|
251
243
|
*/
|
|
252
244
|
anthropicPrefixMismatchBehavior?: "drop_block" | "error";
|
|
253
|
-
/** @internal Marks a replay-only Anthropic request that must use non-streaming `max_tokens: 0`. */
|
|
254
|
-
anthropicCacheRefreshRequest?: boolean;
|
|
255
245
|
/**
|
|
256
246
|
* Anthropic on-demand compaction (`compact-2026-09-04` beta). Sends a
|
|
257
247
|
* top-level `compaction: { type: "summarize", instructions? }` request; the
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@oh-my-pi/pi-ai",
|
|
3
|
-
"version": "18.3.
|
|
3
|
+
"version": "18.3.5",
|
|
4
4
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"ai",
|
|
@@ -155,11 +155,11 @@
|
|
|
155
155
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
156
156
|
},
|
|
157
157
|
"dependencies": {
|
|
158
|
-
"@oh-my-pi/omptype": "18.3.
|
|
159
|
-
"@oh-my-pi/pi-catalog": "18.3.
|
|
160
|
-
"@oh-my-pi/pi-natives": "18.3.
|
|
161
|
-
"@oh-my-pi/pi-utils": "18.3.
|
|
162
|
-
"@oh-my-pi/pi-wire": "18.3.
|
|
158
|
+
"@oh-my-pi/omptype": "18.3.5",
|
|
159
|
+
"@oh-my-pi/pi-catalog": "18.3.5",
|
|
160
|
+
"@oh-my-pi/pi-natives": "18.3.5",
|
|
161
|
+
"@oh-my-pi/pi-utils": "18.3.5",
|
|
162
|
+
"@oh-my-pi/pi-wire": "18.3.5"
|
|
163
163
|
},
|
|
164
164
|
"devDependencies": {
|
|
165
165
|
"@types/bun": "^1.3.14"
|
package/src/index.ts
CHANGED
|
@@ -39,6 +39,7 @@ export type * from "./providers/openai-completions";
|
|
|
39
39
|
export type * from "./providers/openai-responses";
|
|
40
40
|
export type * from "./providers/synthetic";
|
|
41
41
|
export * from "./registry";
|
|
42
|
+
export { resolveCacheRetention } from "./utils";
|
|
42
43
|
export * from "./stream";
|
|
43
44
|
export * from "./types";
|
|
44
45
|
export * from "./usage";
|
|
@@ -1590,31 +1590,6 @@ export function applyAnthropicUsageExtras(usage: Usage, source: AnthropicUsageLi
|
|
|
1590
1590
|
}
|
|
1591
1591
|
}
|
|
1592
1592
|
|
|
1593
|
-
function parseAnthropicWireUsage(value: unknown): AnthropicWireUsage | undefined {
|
|
1594
|
-
if (!isRecord(value)) return undefined;
|
|
1595
|
-
const cacheCreation = isRecord(value.cache_creation)
|
|
1596
|
-
? {
|
|
1597
|
-
...(typeof value.cache_creation.ephemeral_5m_input_tokens === "number"
|
|
1598
|
-
? { ephemeral_5m_input_tokens: value.cache_creation.ephemeral_5m_input_tokens }
|
|
1599
|
-
: {}),
|
|
1600
|
-
...(typeof value.cache_creation.ephemeral_1h_input_tokens === "number"
|
|
1601
|
-
? { ephemeral_1h_input_tokens: value.cache_creation.ephemeral_1h_input_tokens }
|
|
1602
|
-
: {}),
|
|
1603
|
-
}
|
|
1604
|
-
: undefined;
|
|
1605
|
-
return {
|
|
1606
|
-
...(typeof value.input_tokens === "number" ? { input_tokens: value.input_tokens } : {}),
|
|
1607
|
-
...(typeof value.output_tokens === "number" ? { output_tokens: value.output_tokens } : {}),
|
|
1608
|
-
...(typeof value.cache_read_input_tokens === "number"
|
|
1609
|
-
? { cache_read_input_tokens: value.cache_read_input_tokens }
|
|
1610
|
-
: {}),
|
|
1611
|
-
...(typeof value.cache_creation_input_tokens === "number"
|
|
1612
|
-
? { cache_creation_input_tokens: value.cache_creation_input_tokens }
|
|
1613
|
-
: {}),
|
|
1614
|
-
...(cacheCreation === undefined ? {} : { cache_creation: cacheCreation }),
|
|
1615
|
-
};
|
|
1616
|
-
}
|
|
1617
|
-
|
|
1618
1593
|
function parseAnthropicFallbackWireBlock(value: unknown): AnthropicFallbackContent | undefined {
|
|
1619
1594
|
if (!isRecord(value) || value.type !== "fallback") return undefined;
|
|
1620
1595
|
const from = isRecord(value.from) && typeof value.from.model === "string" ? value.from.model : undefined;
|
|
@@ -2071,7 +2046,6 @@ const streamAnthropicOnce = (
|
|
|
2071
2046
|
});
|
|
2072
2047
|
}
|
|
2073
2048
|
|
|
2074
|
-
const zeroOutputCacheRefresh = options?.anthropicCacheRefreshRequest === true;
|
|
2075
2049
|
let client: AnthropicMessagesClientLike;
|
|
2076
2050
|
let isOAuthToken: boolean;
|
|
2077
2051
|
// Retained so a Claude Code version bump can rebuild the client's fingerprint headers.
|
|
@@ -2207,7 +2181,7 @@ const streamAnthropicOnce = (
|
|
|
2207
2181
|
model,
|
|
2208
2182
|
apiKey,
|
|
2209
2183
|
extraBetas,
|
|
2210
|
-
stream:
|
|
2184
|
+
stream: true,
|
|
2211
2185
|
interleavedThinking: options?.interleavedThinking ?? true,
|
|
2212
2186
|
headers: options?.headers,
|
|
2213
2187
|
dynamicHeaders: copilotDynamicHeaders?.headers,
|
|
@@ -2321,89 +2295,6 @@ const streamAnthropicOnce = (
|
|
|
2321
2295
|
const requestTimeoutMs =
|
|
2322
2296
|
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined;
|
|
2323
2297
|
|
|
2324
|
-
if (zeroOutputCacheRefresh) {
|
|
2325
|
-
const refreshParams: MessageCreateParams = { ...params, max_tokens: 0, stream: false };
|
|
2326
|
-
// Anthropic rejects `tool_choice: {type:"tool"|"any"}` with `max_tokens: 0`
|
|
2327
|
-
// ("tool_choice ... cannot be used when max_tokens is 0", #12597). A refresh
|
|
2328
|
-
// replays the captured turn's payload, which can carry a forced selector
|
|
2329
|
-
// (e.g. a forced yield). A zero-output keep-alive produces no tokens, so the
|
|
2330
|
-
// forced choice is meaningless here — drop it so the request is accepted.
|
|
2331
|
-
const refreshChoiceType = refreshParams.tool_choice?.type;
|
|
2332
|
-
if (refreshChoiceType === "tool" || refreshChoiceType === "any") {
|
|
2333
|
-
delete refreshParams.tool_choice;
|
|
2334
|
-
}
|
|
2335
|
-
rawRequestDump = {
|
|
2336
|
-
provider: model.provider,
|
|
2337
|
-
api: output.api,
|
|
2338
|
-
model: model.id,
|
|
2339
|
-
method: "POST",
|
|
2340
|
-
url: `${baseUrl}/v1/messages${isOAuthToken ? "?beta=true" : ""}`,
|
|
2341
|
-
body: refreshParams,
|
|
2342
|
-
};
|
|
2343
|
-
const { requestSignal } = activeAbortTracker;
|
|
2344
|
-
// A replayed compaction block needs the beta on injected clients too.
|
|
2345
|
-
// Route by the client's own endpoint when it exposes one.
|
|
2346
|
-
const refreshBetaRouteUrl =
|
|
2347
|
-
options?.client !== undefined ? (injectedClientBaseUrl(options.client) ?? baseUrl) : baseUrl;
|
|
2348
|
-
let refreshHeaders: Record<string, string> | undefined;
|
|
2349
|
-
if (options?.client !== undefined && !isVertexRawPredictUrl(refreshBetaRouteUrl)) {
|
|
2350
|
-
if (carriesSignedCompaction(refreshParams)) {
|
|
2351
|
-
refreshHeaders = mergeAnthropicBetaHeader(refreshHeaders ?? mergedCallerHeaders, COMPACTION_BETA);
|
|
2352
|
-
}
|
|
2353
|
-
if (carriesLegacyCompactionEdit(refreshParams)) {
|
|
2354
|
-
refreshHeaders = mergeAnthropicBetaHeader(
|
|
2355
|
-
refreshHeaders ?? mergedCallerHeaders,
|
|
2356
|
-
LEGACY_COMPACTION_BETA,
|
|
2357
|
-
);
|
|
2358
|
-
}
|
|
2359
|
-
}
|
|
2360
|
-
const requestOptions = {
|
|
2361
|
-
...createSdkStreamRequestOptions(requestSignal, requestTimeoutMs),
|
|
2362
|
-
maxRetries: 0,
|
|
2363
|
-
...(refreshHeaders ? { headers: refreshHeaders } : {}),
|
|
2364
|
-
};
|
|
2365
|
-
const request: unknown =
|
|
2366
|
-
isOAuthToken && client.beta
|
|
2367
|
-
? client.beta.messages.create(refreshParams, requestOptions)
|
|
2368
|
-
: client.messages.create(refreshParams, requestOptions);
|
|
2369
|
-
if (!hasAnthropicRawResponseRequest(request)) {
|
|
2370
|
-
throw new AIError.AnthropicStreamEnvelopeError(
|
|
2371
|
-
"Anthropic cache refresh request did not expose a raw response",
|
|
2372
|
-
);
|
|
2373
|
-
}
|
|
2374
|
-
const response = await request.asResponse();
|
|
2375
|
-
await notifyProviderResponse(options, response, model, response.headers.get("request-id"));
|
|
2376
|
-
const body: unknown = await response.json();
|
|
2377
|
-
if (!isRecord(body)) {
|
|
2378
|
-
throw new AIError.AnthropicStreamEnvelopeError("Anthropic cache refresh returned a malformed response");
|
|
2379
|
-
}
|
|
2380
|
-
const wireUsage = parseAnthropicWireUsage(body.usage);
|
|
2381
|
-
if (!wireUsage) {
|
|
2382
|
-
throw new AIError.AnthropicStreamEnvelopeError("Anthropic cache refresh response omitted usage");
|
|
2383
|
-
}
|
|
2384
|
-
if (typeof body.id === "string") output.responseId = body.id;
|
|
2385
|
-
applyReportedInputTransformations(
|
|
2386
|
-
output,
|
|
2387
|
-
params,
|
|
2388
|
-
providerSessionState,
|
|
2389
|
-
body.input_transformations,
|
|
2390
|
-
seenInputTransformations,
|
|
2391
|
-
);
|
|
2392
|
-
output.usage.input = wireUsage.input_tokens ?? 0;
|
|
2393
|
-
output.usage.output = wireUsage.output_tokens ?? 0;
|
|
2394
|
-
output.usage.cacheRead = wireUsage.cache_read_input_tokens ?? 0;
|
|
2395
|
-
output.usage.cacheWrite = wireUsage.cache_creation_input_tokens ?? 0;
|
|
2396
|
-
applyAnthropicUsageExtras(output.usage, wireUsage);
|
|
2397
|
-
output.usage.totalTokens =
|
|
2398
|
-
output.usage.input + output.usage.output + output.usage.cacheRead + output.usage.cacheWrite;
|
|
2399
|
-
calculateCost(model, output.usage, output.timestamp);
|
|
2400
|
-
output.duration = performance.now() - startTime;
|
|
2401
|
-
stream.push({ type: "start", partial: output });
|
|
2402
|
-
stream.push({ type: "done", reason: "stop", message: output });
|
|
2403
|
-
stream.end();
|
|
2404
|
-
return;
|
|
2405
|
-
}
|
|
2406
|
-
|
|
2407
2298
|
// Opt-in flag: the response parser only honors `fallback` content
|
|
2408
2299
|
// blocks and `usage.iterations` when the current request opted into
|
|
2409
2300
|
// server-side-fallback beta chain. Leaving `fallbacks` unset preserves
|
package/src/stream.ts
CHANGED
|
@@ -24,7 +24,6 @@ import { isConcurrencyCapExclusion, isUsageLimitOutcome } from "./error/rate-lim
|
|
|
24
24
|
import type { BedrockOptions } from "./providers/amazon-bedrock";
|
|
25
25
|
import type { AnthropicOptions } from "./providers/anthropic";
|
|
26
26
|
import type { AppleFoundationModelsOptions } from "./providers/apple-foundation-models";
|
|
27
|
-
import type { MessageCreateParamsStreaming } from "./providers/anthropic-wire";
|
|
28
27
|
import type { CursorOptions } from "./providers/cursor";
|
|
29
28
|
import type { DevinOptions } from "./providers/devin";
|
|
30
29
|
import { streamGitLabDuo } from "./providers/gitlab-duo";
|
|
@@ -62,7 +61,6 @@ import type {
|
|
|
62
61
|
FetchImpl,
|
|
63
62
|
Model,
|
|
64
63
|
OptionsForApi,
|
|
65
|
-
ProviderSessionState,
|
|
66
64
|
SimpleStreamOptions,
|
|
67
65
|
StreamOptions,
|
|
68
66
|
ThinkingBudgets,
|
|
@@ -1205,285 +1203,6 @@ function emitBufferedEvents(stream: AssistantMessageEventStream, events: Assista
|
|
|
1205
1203
|
stream.push(event);
|
|
1206
1204
|
}
|
|
1207
1205
|
}
|
|
1208
|
-
|
|
1209
|
-
const ANTHROPIC_CACHE_TTL_MS = 5 * 60_000;
|
|
1210
|
-
const ANTHROPIC_CACHE_REFRESH_LEAD_MS = 15_000;
|
|
1211
|
-
const ANTHROPIC_CACHE_REFRESH_LIMIT = 3;
|
|
1212
|
-
const ANTHROPIC_CACHE_REFRESH_STATE_KEY = "anthropic-cache-refresh";
|
|
1213
|
-
|
|
1214
|
-
interface AnthropicCacheRefreshPlan {
|
|
1215
|
-
refresh(controller: AbortController): Promise<number | undefined>;
|
|
1216
|
-
}
|
|
1217
|
-
|
|
1218
|
-
class AnthropicCacheRefreshState implements ProviderSessionState {
|
|
1219
|
-
#controller: AbortController | undefined;
|
|
1220
|
-
#generation = 0;
|
|
1221
|
-
#plan: AnthropicCacheRefreshPlan | undefined;
|
|
1222
|
-
#refreshesRemaining = 0;
|
|
1223
|
-
#timer: NodeJS.Timeout | undefined;
|
|
1224
|
-
|
|
1225
|
-
cancel(): void {
|
|
1226
|
-
this.#generation++;
|
|
1227
|
-
if (this.#timer !== undefined) {
|
|
1228
|
-
clearTimeout(this.#timer);
|
|
1229
|
-
this.#timer = undefined;
|
|
1230
|
-
}
|
|
1231
|
-
this.#controller?.abort();
|
|
1232
|
-
this.#controller = undefined;
|
|
1233
|
-
this.#plan = undefined;
|
|
1234
|
-
this.#refreshesRemaining = 0;
|
|
1235
|
-
}
|
|
1236
|
-
|
|
1237
|
-
arm(plan: AnthropicCacheRefreshPlan, cacheTouchedAtMs: number): void {
|
|
1238
|
-
this.cancel();
|
|
1239
|
-
this.#plan = plan;
|
|
1240
|
-
this.#refreshesRemaining = ANTHROPIC_CACHE_REFRESH_LIMIT;
|
|
1241
|
-
this.#schedule(cacheTouchedAtMs, this.#generation);
|
|
1242
|
-
}
|
|
1243
|
-
|
|
1244
|
-
close(): void {
|
|
1245
|
-
this.cancel();
|
|
1246
|
-
}
|
|
1247
|
-
|
|
1248
|
-
#schedule(cacheTouchedAtMs: number, generation: number): void {
|
|
1249
|
-
const refreshAtMs = cacheTouchedAtMs + ANTHROPIC_CACHE_TTL_MS - ANTHROPIC_CACHE_REFRESH_LEAD_MS;
|
|
1250
|
-
this.#timer = setTimeout(
|
|
1251
|
-
() => {
|
|
1252
|
-
this.#timer = undefined;
|
|
1253
|
-
void this.#refresh(generation);
|
|
1254
|
-
},
|
|
1255
|
-
Math.max(0, refreshAtMs - Date.now()),
|
|
1256
|
-
);
|
|
1257
|
-
this.#timer.unref?.();
|
|
1258
|
-
}
|
|
1259
|
-
|
|
1260
|
-
async #refresh(generation: number): Promise<void> {
|
|
1261
|
-
const plan = this.#plan;
|
|
1262
|
-
if (generation !== this.#generation || !plan || this.#refreshesRemaining <= 0) return;
|
|
1263
|
-
|
|
1264
|
-
const controller = new AbortController();
|
|
1265
|
-
this.#controller = controller;
|
|
1266
|
-
let cacheTouchedAtMs: number | undefined;
|
|
1267
|
-
try {
|
|
1268
|
-
cacheTouchedAtMs = await plan.refresh(controller);
|
|
1269
|
-
} catch (error) {
|
|
1270
|
-
if (generation === this.#generation && !controller.signal.aborted) {
|
|
1271
|
-
logger.debug("Anthropic prompt-cache refresh failed", { error: String(error) });
|
|
1272
|
-
}
|
|
1273
|
-
}
|
|
1274
|
-
if (generation !== this.#generation) return;
|
|
1275
|
-
|
|
1276
|
-
this.#controller = undefined;
|
|
1277
|
-
if (cacheTouchedAtMs === undefined) {
|
|
1278
|
-
this.#plan = undefined;
|
|
1279
|
-
this.#refreshesRemaining = 0;
|
|
1280
|
-
return;
|
|
1281
|
-
}
|
|
1282
|
-
|
|
1283
|
-
this.#refreshesRemaining--;
|
|
1284
|
-
if (this.#refreshesRemaining <= 0) {
|
|
1285
|
-
this.#plan = undefined;
|
|
1286
|
-
return;
|
|
1287
|
-
}
|
|
1288
|
-
this.#schedule(cacheTouchedAtMs, generation);
|
|
1289
|
-
}
|
|
1290
|
-
}
|
|
1291
|
-
|
|
1292
|
-
function supportsAnthropicCacheRefresh<TApi extends Api>(model: Model<TApi>): boolean {
|
|
1293
|
-
return (
|
|
1294
|
-
model.api === "anthropic-messages" &&
|
|
1295
|
-
model.provider === "anthropic" &&
|
|
1296
|
-
model.transport !== "pi-native" &&
|
|
1297
|
-
isLeakedThinkingHealExempt(model)
|
|
1298
|
-
);
|
|
1299
|
-
}
|
|
1300
|
-
|
|
1301
|
-
function isAnthropicRefreshPayload(payload: unknown): payload is MessageCreateParamsStreaming {
|
|
1302
|
-
return (
|
|
1303
|
-
typeof payload === "object" &&
|
|
1304
|
-
payload !== null &&
|
|
1305
|
-
"messages" in payload &&
|
|
1306
|
-
Array.isArray(payload.messages) &&
|
|
1307
|
-
"max_tokens" in payload &&
|
|
1308
|
-
typeof payload.max_tokens === "number"
|
|
1309
|
-
);
|
|
1310
|
-
}
|
|
1311
|
-
|
|
1312
|
-
function isShortAnthropicCacheControl(cacheControl: unknown): boolean {
|
|
1313
|
-
return (
|
|
1314
|
-
typeof cacheControl === "object" &&
|
|
1315
|
-
cacheControl !== null &&
|
|
1316
|
-
"type" in cacheControl &&
|
|
1317
|
-
cacheControl.type === "ephemeral" &&
|
|
1318
|
-
(!("ttl" in cacheControl) || cacheControl.ttl !== "1h")
|
|
1319
|
-
);
|
|
1320
|
-
}
|
|
1321
|
-
|
|
1322
|
-
function hasShortAnthropicMessageBreakpoint(payload: MessageCreateParamsStreaming): boolean {
|
|
1323
|
-
for (const message of payload.messages) {
|
|
1324
|
-
if (!Array.isArray(message.content)) continue;
|
|
1325
|
-
for (const block of message.content) {
|
|
1326
|
-
if ("cache_control" in block && isShortAnthropicCacheControl(block.cache_control)) return true;
|
|
1327
|
-
}
|
|
1328
|
-
}
|
|
1329
|
-
return false;
|
|
1330
|
-
}
|
|
1331
|
-
|
|
1332
|
-
function isAnthropicGenerationEvent(event: AssistantMessageEvent): boolean {
|
|
1333
|
-
switch (event.type) {
|
|
1334
|
-
case "text_start":
|
|
1335
|
-
case "thinking_start":
|
|
1336
|
-
case "toolcall_start":
|
|
1337
|
-
case "image_end":
|
|
1338
|
-
return true;
|
|
1339
|
-
case "text_delta":
|
|
1340
|
-
case "thinking_delta":
|
|
1341
|
-
case "toolcall_delta":
|
|
1342
|
-
return event.delta.length > 0;
|
|
1343
|
-
default:
|
|
1344
|
-
return false;
|
|
1345
|
-
}
|
|
1346
|
-
}
|
|
1347
|
-
|
|
1348
|
-
function isAnthropicThinkingActive(model: Model<Api>, payload: MessageCreateParamsStreaming): boolean {
|
|
1349
|
-
if (payload.thinking) return payload.thinking.type !== "disabled";
|
|
1350
|
-
return model.thinking?.mode === "anthropic-adaptive" && payload.output_config?.effort != null;
|
|
1351
|
-
}
|
|
1352
|
-
|
|
1353
|
-
function createAnthropicCacheRefreshPlan<TApi extends Api>(
|
|
1354
|
-
model: Model<TApi>,
|
|
1355
|
-
context: Context,
|
|
1356
|
-
options: SimpleStreamOptions | undefined,
|
|
1357
|
-
payload: MessageCreateParamsStreaming,
|
|
1358
|
-
): AnthropicCacheRefreshPlan {
|
|
1359
|
-
const thinkingEnabled = isAnthropicThinkingActive(model, payload);
|
|
1360
|
-
return {
|
|
1361
|
-
async refresh(controller) {
|
|
1362
|
-
let cacheRead = 0;
|
|
1363
|
-
let cacheWrite = 0;
|
|
1364
|
-
let cacheTouchedAtMs: number | undefined;
|
|
1365
|
-
let canceledAfterGenerationStarted = false;
|
|
1366
|
-
const response = streamSimpleRequest(model, context, {
|
|
1367
|
-
...options,
|
|
1368
|
-
acceptEmptyResponse: true,
|
|
1369
|
-
anthropicCacheRefreshRequest: !thinkingEnabled,
|
|
1370
|
-
cacheRetention: "short",
|
|
1371
|
-
maxTokens: thinkingEnabled ? options?.maxTokens : 0,
|
|
1372
|
-
onPayload: () => ({
|
|
1373
|
-
...payload,
|
|
1374
|
-
max_tokens: thinkingEnabled ? payload.max_tokens : 0,
|
|
1375
|
-
}),
|
|
1376
|
-
onResponse: () => {
|
|
1377
|
-
cacheTouchedAtMs = Date.now();
|
|
1378
|
-
},
|
|
1379
|
-
onSseEvent: undefined,
|
|
1380
|
-
signal: controller.signal,
|
|
1381
|
-
});
|
|
1382
|
-
|
|
1383
|
-
for await (const event of response) {
|
|
1384
|
-
if ("partial" in event) {
|
|
1385
|
-
cacheRead = event.partial.usage.cacheRead;
|
|
1386
|
-
cacheWrite = event.partial.usage.cacheWrite;
|
|
1387
|
-
}
|
|
1388
|
-
if (event.type === "error") return undefined;
|
|
1389
|
-
if (event.type === "done") {
|
|
1390
|
-
cacheRead = event.message.usage.cacheRead;
|
|
1391
|
-
cacheWrite = event.message.usage.cacheWrite;
|
|
1392
|
-
return cacheTouchedAtMs !== undefined && cacheRead > 0 && cacheWrite === 0
|
|
1393
|
-
? cacheTouchedAtMs
|
|
1394
|
-
: undefined;
|
|
1395
|
-
}
|
|
1396
|
-
if (thinkingEnabled && isAnthropicGenerationEvent(event)) {
|
|
1397
|
-
canceledAfterGenerationStarted = true;
|
|
1398
|
-
controller.abort();
|
|
1399
|
-
break;
|
|
1400
|
-
}
|
|
1401
|
-
}
|
|
1402
|
-
|
|
1403
|
-
if (canceledAfterGenerationStarted) {
|
|
1404
|
-
try {
|
|
1405
|
-
await response.result();
|
|
1406
|
-
} catch (error) {
|
|
1407
|
-
if (!controller.signal.aborted) throw error;
|
|
1408
|
-
}
|
|
1409
|
-
}
|
|
1410
|
-
return cacheTouchedAtMs !== undefined && cacheRead > 0 && cacheWrite === 0 ? cacheTouchedAtMs : undefined;
|
|
1411
|
-
},
|
|
1412
|
-
};
|
|
1413
|
-
}
|
|
1414
|
-
|
|
1415
|
-
function streamSimpleWithAnthropicCacheRefresh<TApi extends Api>(
|
|
1416
|
-
model: Model<TApi>,
|
|
1417
|
-
context: Context,
|
|
1418
|
-
options: SimpleStreamOptions | undefined,
|
|
1419
|
-
): AssistantMessageEventStream {
|
|
1420
|
-
const providerSessionState = options?.providerSessionState;
|
|
1421
|
-
if (!options?.anthropicCacheRefresh || !providerSessionState) {
|
|
1422
|
-
return streamSimpleRequest(model, context, options);
|
|
1423
|
-
}
|
|
1424
|
-
|
|
1425
|
-
const existingState = providerSessionState.get(ANTHROPIC_CACHE_REFRESH_STATE_KEY);
|
|
1426
|
-
if (existingState instanceof AnthropicCacheRefreshState) {
|
|
1427
|
-
existingState.cancel();
|
|
1428
|
-
} else if (existingState) {
|
|
1429
|
-
return streamSimpleRequest(model, context, options);
|
|
1430
|
-
}
|
|
1431
|
-
if (!supportsAnthropicCacheRefresh(model) || resolveCacheRetention(options.cacheRetention) !== "short") {
|
|
1432
|
-
return streamSimpleRequest(model, context, options);
|
|
1433
|
-
}
|
|
1434
|
-
|
|
1435
|
-
const refreshState = existingState ?? new AnthropicCacheRefreshState();
|
|
1436
|
-
if (!existingState) providerSessionState.set(ANTHROPIC_CACHE_REFRESH_STATE_KEY, refreshState);
|
|
1437
|
-
|
|
1438
|
-
let cacheTouchedAtMs: number | undefined;
|
|
1439
|
-
let capturedPayload: MessageCreateParamsStreaming | undefined;
|
|
1440
|
-
const inner = streamSimpleRequest(model, context, {
|
|
1441
|
-
...options,
|
|
1442
|
-
onPayload: async (payload, payloadModel) => {
|
|
1443
|
-
const replacement = await options?.onPayload?.(payload, payloadModel);
|
|
1444
|
-
const finalPayload = replacement ?? payload;
|
|
1445
|
-
if (isAnthropicRefreshPayload(finalPayload)) capturedPayload = finalPayload;
|
|
1446
|
-
return replacement;
|
|
1447
|
-
},
|
|
1448
|
-
onResponse: async (response, responseModel) => {
|
|
1449
|
-
cacheTouchedAtMs = Date.now();
|
|
1450
|
-
await options?.onResponse?.(response, responseModel);
|
|
1451
|
-
},
|
|
1452
|
-
});
|
|
1453
|
-
const outer = new AssistantMessageEventStream();
|
|
1454
|
-
const armRefresh = (message: AssistantMessage): void => {
|
|
1455
|
-
if (
|
|
1456
|
-
message.stopReason === "error" ||
|
|
1457
|
-
message.stopReason === "aborted" ||
|
|
1458
|
-
message.usage.cacheRead + message.usage.cacheWrite <= 0 ||
|
|
1459
|
-
cacheTouchedAtMs === undefined ||
|
|
1460
|
-
capturedPayload === undefined ||
|
|
1461
|
-
!hasShortAnthropicMessageBreakpoint(capturedPayload)
|
|
1462
|
-
) {
|
|
1463
|
-
return;
|
|
1464
|
-
}
|
|
1465
|
-
refreshState.arm(createAnthropicCacheRefreshPlan(model, context, options, capturedPayload), cacheTouchedAtMs);
|
|
1466
|
-
};
|
|
1467
|
-
|
|
1468
|
-
void (async () => {
|
|
1469
|
-
try {
|
|
1470
|
-
for await (const event of inner) {
|
|
1471
|
-
if (event.type === "done") armRefresh(event.message);
|
|
1472
|
-
outer.push(event);
|
|
1473
|
-
if (outer.done) return;
|
|
1474
|
-
}
|
|
1475
|
-
if (!outer.done) {
|
|
1476
|
-
const result = await inner.result();
|
|
1477
|
-
armRefresh(result);
|
|
1478
|
-
outer.end(result);
|
|
1479
|
-
}
|
|
1480
|
-
} catch (error) {
|
|
1481
|
-
outer.fail(error);
|
|
1482
|
-
}
|
|
1483
|
-
})();
|
|
1484
|
-
return outer;
|
|
1485
|
-
}
|
|
1486
|
-
|
|
1487
1206
|
function withInferenceSessionId(options?: SimpleStreamOptions): SimpleStreamOptions {
|
|
1488
1207
|
if (options?.sessionId) return options;
|
|
1489
1208
|
return { ...options, sessionId: crypto.randomUUID() };
|
|
@@ -1496,7 +1215,7 @@ export function streamSimple<TApi extends Api>(
|
|
|
1496
1215
|
): AssistantMessageEventStream {
|
|
1497
1216
|
const sessionOptions = withInferenceSessionId(options);
|
|
1498
1217
|
if (!model.requiresGlyphTokenization) {
|
|
1499
|
-
return
|
|
1218
|
+
return streamSimpleRequest(model, context, sessionOptions);
|
|
1500
1219
|
}
|
|
1501
1220
|
const codec = applyGlyphCodec(context);
|
|
1502
1221
|
const execHandlers = sessionOptions.cursorExecHandlers ?? sessionOptions.execHandlers;
|
|
@@ -1509,7 +1228,7 @@ export function streamSimple<TApi extends Api>(
|
|
|
1509
1228
|
execHandlers: wrappedExecHandlers,
|
|
1510
1229
|
cursorExecHandlers: wrappedExecHandlers,
|
|
1511
1230
|
};
|
|
1512
|
-
return codec.wrap(
|
|
1231
|
+
return codec.wrap(streamSimpleRequest(model, codec.context, wireOptions));
|
|
1513
1232
|
}
|
|
1514
1233
|
|
|
1515
1234
|
/**
|
|
@@ -2093,7 +1812,6 @@ function mapOptionsForApi<TApi extends Api>(
|
|
|
2093
1812
|
fetch: options?.fetch,
|
|
2094
1813
|
fallbacks: options?.fallbacks,
|
|
2095
1814
|
acceptEmptyResponse: options?.acceptEmptyResponse,
|
|
2096
|
-
anthropicCacheRefreshRequest: options?.anthropicCacheRefreshRequest,
|
|
2097
1815
|
anthropicPrefixMismatchBehavior: options?.anthropicPrefixMismatchBehavior,
|
|
2098
1816
|
anthropicCompaction: options?.anthropicCompaction,
|
|
2099
1817
|
anthropicSlowMode: options?.anthropicSlowMode,
|
package/src/types.ts
CHANGED
|
@@ -428,21 +428,11 @@ export interface StreamOptions {
|
|
|
428
428
|
/** @internal Stored credential row serving this request, when known. */
|
|
429
429
|
credentialId?: number;
|
|
430
430
|
cacheRetention?: CacheRetention;
|
|
431
|
-
/**
|
|
432
|
-
* Keep Anthropic's 5-minute prompt cache warm across bounded idle gaps.
|
|
433
|
-
*
|
|
434
|
-
* This is an ownership flag, not a general provider default: exactly one
|
|
435
|
-
* primary agent loop sharing `providerSessionState` should enable it.
|
|
436
|
-
* Side-channel and advisor requests must leave it unset.
|
|
437
|
-
*/
|
|
438
|
-
anthropicCacheRefresh?: boolean;
|
|
439
431
|
/**
|
|
440
432
|
* Anthropic preserved-thinking behavior when a signed block no longer matches
|
|
441
433
|
* its conversation prefix. Binding-capable models default to `"drop_block"`.
|
|
442
434
|
*/
|
|
443
435
|
anthropicPrefixMismatchBehavior?: "drop_block" | "error";
|
|
444
|
-
/** @internal Marks a replay-only Anthropic request that must use non-streaming `max_tokens: 0`. */
|
|
445
|
-
anthropicCacheRefreshRequest?: boolean;
|
|
446
436
|
/**
|
|
447
437
|
* Anthropic on-demand compaction (`compact-2026-09-04` beta). Sends a
|
|
448
438
|
* top-level `compaction: { type: "summarize", instructions? }` request; the
|