@deepstrike/sdk 0.2.11 → 0.2.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -0
- package/dist/index.d.ts +4 -3
- package/dist/index.js +4 -3
- package/dist/providers/anthropic.d.ts +7 -0
- package/dist/providers/anthropic.js +129 -10
- package/dist/providers/base.d.ts +38 -0
- package/dist/providers/base.js +0 -0
- package/dist/providers/catalog.js +20 -8
- package/dist/providers/deepseek.d.ts +12 -0
- package/dist/providers/deepseek.js +23 -4
- package/dist/providers/gemini.js +23 -4
- package/dist/providers/glm.d.ts +12 -0
- package/dist/providers/glm.js +19 -2
- package/dist/providers/kimi.d.ts +12 -0
- package/dist/providers/kimi.js +19 -2
- package/dist/providers/minimax.js +4 -2
- package/dist/providers/ollama.js +2 -2
- package/dist/providers/openai-chat.js +3 -2
- package/dist/providers/openai-responses.js +13 -0
- package/dist/providers/openai.d.ts +7 -0
- package/dist/providers/openai.js +15 -2
- package/dist/providers/profiles.d.ts +52 -28
- package/dist/providers/profiles.js +52 -28
- package/dist/providers/qwen.d.ts +12 -0
- package/dist/providers/qwen.js +23 -4
- package/dist/runtime/runner.d.ts +3 -1
- package/dist/runtime/runner.js +39 -3
- package/dist/runtime/session-log.d.ts +2 -1
- package/dist/types.d.ts +34 -1
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -145,6 +145,44 @@ Spool, page-out, signals, processes, budgets, and memory events land in `Session
|
|
|
145
145
|
|
|
146
146
|
---
|
|
147
147
|
|
|
148
|
+
## Dynamic workflows
|
|
149
|
+
|
|
150
|
+
Instead of planning **and** executing a hard task in one long context window, hand the kernel a declarative DAG and let it spawn a fresh-context sub-agent per node. The kernel owns the control flow (gate · budget · suspend-on-join · resume); your SDK runs the agents. See the [top-level overview](../README.md#the-six-harness-patterns-as-first-class-kernel-nodes) for the full pattern catalog.
|
|
151
|
+
|
|
152
|
+
```ts
|
|
153
|
+
// One fresh-context verifier per rule (no inherited author context → can't rubber-stamp),
|
|
154
|
+
// then a skeptic that reviews their flags. The kernel spawns the 3 verifiers as one gated
|
|
155
|
+
// batch, suspends on the join, and runs the skeptic once they complete.
|
|
156
|
+
const outcome = await runner.runWorkflow({
|
|
157
|
+
nodes: [
|
|
158
|
+
{ task: "Rule: money is integer cents — violated?", role: "verify" },
|
|
159
|
+
{ task: "Rule: all errors propagate — violated?", role: "verify" },
|
|
160
|
+
{ task: "Rule: timestamps are UTC — violated?", role: "verify" },
|
|
161
|
+
{ task: "Skeptic: which flags are real violations?", role: "verify", dependsOn: [0, 1, 2] },
|
|
162
|
+
],
|
|
163
|
+
})
|
|
164
|
+
// → { completed: ["wf-node0", … ], failed: [] }
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
A node's `kind` selects the control-flow shape; the same executor drives them all, every spawn passing the syscall gate:
|
|
168
|
+
|
|
169
|
+
| Node `kind` | Behavior |
|
|
170
|
+
|---|---|
|
|
171
|
+
| `{ type: "spawn" }` (default) | Run the node's agent once |
|
|
172
|
+
| `{ type: "loop", maxIters }` | Re-run until the agent signals it's done, capped at `maxIters` |
|
|
173
|
+
| `{ type: "classify", branches }` | The classifier's result selects one branch; the rest are pruned |
|
|
174
|
+
| `{ type: "tournament", entrants }` | Generate N entrants, then a pairwise-judge bracket to one winner |
|
|
175
|
+
| `{ type: "reduce", reducer }` | **Tokenless host-compute** — a pure function (`dedupe_lines` / `merge_json_arrays` / `concat` / `count`, or your own via the `reducers` runner option) over the node's dependency outputs |
|
|
176
|
+
|
|
177
|
+
### 0.2.11 capabilities
|
|
178
|
+
|
|
179
|
+
- **Runtime fan-out** — give a node the `submitWorkflowNodesTool` and its agent can append nodes to the live DAG mid-run (true loop-until-done; one verifier per claim it discovers). Recorded and replayed on `resumeWorkflow`.
|
|
180
|
+
- **Quarantine, no escape** — set `trust: "quarantined"` on a node that reads untrusted content; it's denied write-capable isolation in-kernel, and any nodes it submits are coerced to quarantined too (no privilege escalation).
|
|
181
|
+
- **Structured output** — set `outputSchema` on a node; the runner instructs the agent, validates the result against the JSON-Schema subset, and re-runs once with the errors on mismatch. A node that never conforms fails (its dependents starve).
|
|
182
|
+
- **Budget as signal** — with a `maxWorkflowNodes` / `maxConcurrentSubagents` quota installed, each spawned node's goal carries its remaining headroom so a coordinator can size its fan-out to fit.
|
|
183
|
+
|
|
184
|
+
---
|
|
185
|
+
|
|
148
186
|
## Providers
|
|
149
187
|
|
|
150
188
|
| Class | Backend | Notes |
|
package/dist/index.d.ts
CHANGED
|
@@ -29,9 +29,10 @@ export { RemoteVpcPlane } from "./runtime/remote-vpc-plane.js";
|
|
|
29
29
|
export type { RemoteVpcOptions } from "./runtime/remote-vpc-plane.js";
|
|
30
30
|
export { AnthropicProvider } from "./providers/anthropic.js";
|
|
31
31
|
export { OpenAIChatProvider, OpenAIProvider } from "./providers/openai.js";
|
|
32
|
-
export { DeepSeekProvider } from "./providers/deepseek.js";
|
|
33
|
-
export { KimiProvider } from "./providers/kimi.js";
|
|
34
|
-
export { QwenProvider } from "./providers/qwen.js";
|
|
32
|
+
export { DeepSeekProvider, DeepSeekAnthropicProvider } from "./providers/deepseek.js";
|
|
33
|
+
export { KimiProvider, KimiAnthropicProvider } from "./providers/kimi.js";
|
|
34
|
+
export { QwenProvider, QwenAnthropicProvider } from "./providers/qwen.js";
|
|
35
|
+
export { GLMProvider, GLMAnthropicProvider } from "./providers/glm.js";
|
|
35
36
|
export { GeminiProvider } from "./providers/gemini.js";
|
|
36
37
|
export { MiniMaxAnthropicProvider, MiniMaxOpenAIProvider } from "./providers/minimax.js";
|
|
37
38
|
export { OllamaProvider } from "./providers/ollama.js";
|
package/dist/index.js
CHANGED
|
@@ -17,9 +17,10 @@ export { RemoteVpcPlane } from "./runtime/remote-vpc-plane.js";
|
|
|
17
17
|
// ── Providers ─────────────────────────────────────────────────────────────
|
|
18
18
|
export { AnthropicProvider } from "./providers/anthropic.js";
|
|
19
19
|
export { OpenAIChatProvider, OpenAIProvider } from "./providers/openai.js";
|
|
20
|
-
export { DeepSeekProvider } from "./providers/deepseek.js";
|
|
21
|
-
export { KimiProvider } from "./providers/kimi.js";
|
|
22
|
-
export { QwenProvider } from "./providers/qwen.js";
|
|
20
|
+
export { DeepSeekProvider, DeepSeekAnthropicProvider } from "./providers/deepseek.js";
|
|
21
|
+
export { KimiProvider, KimiAnthropicProvider } from "./providers/kimi.js";
|
|
22
|
+
export { QwenProvider, QwenAnthropicProvider } from "./providers/qwen.js";
|
|
23
|
+
export { GLMProvider, GLMAnthropicProvider } from "./providers/glm.js";
|
|
23
24
|
export { GeminiProvider } from "./providers/gemini.js";
|
|
24
25
|
export { MiniMaxAnthropicProvider, MiniMaxOpenAIProvider } from "./providers/minimax.js";
|
|
25
26
|
export { OllamaProvider } from "./providers/ollama.js";
|
|
@@ -20,6 +20,13 @@ export declare class AnthropicProvider implements LLMProvider {
|
|
|
20
20
|
descriptor(): ProviderDescriptor;
|
|
21
21
|
peekProviderReplay(message: Pick<Message, "content" | "toolCalls">): ProviderReplay | undefined;
|
|
22
22
|
seedProviderReplay(message: Pick<Message, "content" | "toolCalls">, replay: ProviderReplay): void;
|
|
23
|
+
/**
|
|
24
|
+
* Build tool definitions. A cache breakpoint is anchored on the final tool
|
|
25
|
+
* only when the system blocks won't carry one (`anchorCache`). When structured
|
|
26
|
+
* system blocks are present, their breakpoints already cache the tools prefix
|
|
27
|
+
* (tools render before system), so a redundant tool breakpoint would only burn
|
|
28
|
+
* one of Anthropic's 4 cache_control slots — slots the message history needs.
|
|
29
|
+
*/
|
|
23
30
|
private buildTools;
|
|
24
31
|
complete(context: RenderedContext, tools: ToolSchema[], extensions?: Record<string, unknown>): Promise<Message>;
|
|
25
32
|
stream(context: RenderedContext, tools: ToolSchema[], extensions?: Record<string, unknown>): AsyncIterable<StreamEvent>;
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import Anthropic from "@anthropic-ai/sdk";
|
|
2
2
|
import { assistantReplayKey } from "../runtime/provider-replay.js";
|
|
3
3
|
import { withServerRuntimeGuard } from "../runtime/server.js";
|
|
4
|
-
import { CircuitBreaker, normalizeToolCall, omitExtensionKeys, toAnthropicMessages } from "./base.js";
|
|
4
|
+
import { CircuitBreaker, normalizeToolCall, omitExtensionKeys, toAnthropicContent, toAnthropicMessages } from "./base.js";
|
|
5
5
|
const CLAUDE_POLICIES = {
|
|
6
6
|
"claude-opus-4-1": { maxTurns: 50 },
|
|
7
7
|
"claude-opus-4-7": { maxTurns: 50 },
|
|
@@ -70,12 +70,19 @@ export class AnthropicProvider {
|
|
|
70
70
|
if (blocks.length)
|
|
71
71
|
this.nativeAssistantBlocks.set(assistantReplayKey(message), blocks);
|
|
72
72
|
}
|
|
73
|
-
|
|
73
|
+
/**
|
|
74
|
+
* Build tool definitions. A cache breakpoint is anchored on the final tool
|
|
75
|
+
* only when the system blocks won't carry one (`anchorCache`). When structured
|
|
76
|
+
* system blocks are present, their breakpoints already cache the tools prefix
|
|
77
|
+
* (tools render before system), so a redundant tool breakpoint would only burn
|
|
78
|
+
* one of Anthropic's 4 cache_control slots — slots the message history needs.
|
|
79
|
+
*/
|
|
80
|
+
buildTools(tools, anchorCache) {
|
|
74
81
|
return tools.map((t, i) => ({
|
|
75
82
|
name: t.name,
|
|
76
83
|
description: t.description,
|
|
77
84
|
input_schema: JSON.parse(t.parameters),
|
|
78
|
-
...(i === tools.length - 1 ? { cache_control: { type: "ephemeral" } } : {}),
|
|
85
|
+
...(anchorCache && i === tools.length - 1 ? { cache_control: { type: "ephemeral" } } : {}),
|
|
79
86
|
}));
|
|
80
87
|
}
|
|
81
88
|
async complete(context, tools, extensions) {
|
|
@@ -83,6 +90,7 @@ export class AnthropicProvider {
|
|
|
83
90
|
throw new Error("Circuit breaker open");
|
|
84
91
|
const system = this.buildSystem(context);
|
|
85
92
|
const msgs = this.buildMessages(context);
|
|
93
|
+
assertCacheBudget(system, tools.length);
|
|
86
94
|
const requestExtensions = this.requestExtensions(extensions);
|
|
87
95
|
let lastErr;
|
|
88
96
|
for (let i = 0; i < this.maxRetries; i++) {
|
|
@@ -93,7 +101,7 @@ export class AnthropicProvider {
|
|
|
93
101
|
max_tokens: typeof extensions?.max_tokens === "number" ? extensions.max_tokens : 8096,
|
|
94
102
|
...(system ? { system } : {}),
|
|
95
103
|
messages: msgs,
|
|
96
|
-
...(tools.length ? { tools: this.buildTools(tools) } : {}),
|
|
104
|
+
...(tools.length ? { tools: this.buildTools(tools, !Array.isArray(system)) } : {}),
|
|
97
105
|
}, extensions);
|
|
98
106
|
this.circuit.recordSuccess();
|
|
99
107
|
let content = "";
|
|
@@ -123,6 +131,7 @@ export class AnthropicProvider {
|
|
|
123
131
|
async *stream(context, tools, extensions) {
|
|
124
132
|
const system = this.buildSystem(context);
|
|
125
133
|
const msgs = this.buildMessages(context);
|
|
134
|
+
assertCacheBudget(system, tools.length);
|
|
126
135
|
const requestExtensions = this.requestExtensions(extensions);
|
|
127
136
|
const toolBlocks = {};
|
|
128
137
|
const nativeBlocks = {};
|
|
@@ -134,17 +143,36 @@ export class AnthropicProvider {
|
|
|
134
143
|
max_tokens: typeof extensions?.max_tokens === "number" ? extensions.max_tokens : 8096,
|
|
135
144
|
...(system ? { system } : {}),
|
|
136
145
|
messages: msgs,
|
|
137
|
-
...(tools.length ? { tools: this.buildTools(tools) } : {}),
|
|
146
|
+
...(tools.length ? { tools: this.buildTools(tools, !Array.isArray(system)) } : {}),
|
|
138
147
|
}, extensions);
|
|
139
|
-
let
|
|
148
|
+
let uncachedInput = 0;
|
|
149
|
+
let cacheReadTokens = 0;
|
|
150
|
+
let cacheCreationTokens = 0;
|
|
151
|
+
let outputTokens = 0;
|
|
140
152
|
for await (const evt of stream) {
|
|
141
153
|
if (evt.type === "message_start" || evt.type === "message_delta") {
|
|
142
154
|
const usage = evt.usage ?? evt.message?.usage;
|
|
143
155
|
if (usage) {
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
156
|
+
// input + cache counts are cumulative and pinned at message_start; a
|
|
157
|
+
// later message_delta may omit them (null), so Math.max keeps the
|
|
158
|
+
// running totals from being clobbered back to zero.
|
|
159
|
+
uncachedInput = Math.max(uncachedInput, usage.input_tokens ?? 0);
|
|
160
|
+
cacheReadTokens = Math.max(cacheReadTokens, usage.cache_read_input_tokens ?? 0);
|
|
161
|
+
cacheCreationTokens = Math.max(cacheCreationTokens, usage.cache_creation_input_tokens ?? 0);
|
|
162
|
+
outputTokens = Math.max(outputTokens, usage.output_tokens ?? 0);
|
|
163
|
+
// inputTokens is the FULL prompt size (uncached + cache read + cache
|
|
164
|
+
// write). The kernel reads it as the authoritative prompt size for
|
|
165
|
+
// context-pressure/compaction — excluding cached tokens would make a
|
|
166
|
+
// cache-heavy turn look tiny and suppress compaction until a 413.
|
|
167
|
+
const inputTokens = uncachedInput + cacheReadTokens + cacheCreationTokens;
|
|
168
|
+
yield {
|
|
169
|
+
type: "usage",
|
|
170
|
+
totalTokens: inputTokens + outputTokens,
|
|
171
|
+
inputTokens,
|
|
172
|
+
outputTokens,
|
|
173
|
+
cacheReadInputTokens: cacheReadTokens,
|
|
174
|
+
cacheCreationInputTokens: cacheCreationTokens,
|
|
175
|
+
};
|
|
148
176
|
}
|
|
149
177
|
}
|
|
150
178
|
else if (evt.type === "content_block_start") {
|
|
@@ -206,6 +234,12 @@ export class AnthropicProvider {
|
|
|
206
234
|
: this.client.messages.stream(params));
|
|
207
235
|
}
|
|
208
236
|
buildSystem(context) {
|
|
237
|
+
// B3 note: the system shape is content-driven — 0 blocks (string), 1 block
|
|
238
|
+
// (stable only), or 2 blocks (stable + knowledge). The first turn `systemKnowledge`
|
|
239
|
+
// appears, the block count rises 1→2, which is a one-time prompt-cache invalidation
|
|
240
|
+
// (the knowledge prefix didn't exist to cache before). It is byte-stable thereafter;
|
|
241
|
+
// dynamic per-turn knowledge belongs in the uncached tail, not this block. An empty
|
|
242
|
+
// knowledge string is intentionally never emitted (the API rejects empty text blocks).
|
|
209
243
|
if (!context.systemStable && !context.systemKnowledge) {
|
|
210
244
|
return context.systemText || undefined;
|
|
211
245
|
}
|
|
@@ -220,6 +254,18 @@ export class AnthropicProvider {
|
|
|
220
254
|
}
|
|
221
255
|
buildMessages(context) {
|
|
222
256
|
const msgs = toAnthropicMessages(context.turns, message => this.nativeAssistantBlocks.get(assistantReplayKey(message)));
|
|
257
|
+
// Cache breakpoints anchor on the stable history; the volatile State turn is
|
|
258
|
+
// appended AFTER them as the uncached tail (so the history prefix re-reads
|
|
259
|
+
// across turns). On un-rebuilt bindings stateTurn is absent and the state is
|
|
260
|
+
// already inside `turns` — rendered as-is above. `frozenPrefixLen` (P1-E) pins
|
|
261
|
+
// the deep breakpoint at the compaction boundary; absent ⇒ rolling-pair fallback.
|
|
262
|
+
applyMessageCacheControl(msgs, context.frozenPrefixLen);
|
|
263
|
+
if (context.stateTurn) {
|
|
264
|
+
msgs.push({
|
|
265
|
+
role: context.stateTurn.role === "assistant" ? "assistant" : "user",
|
|
266
|
+
content: toAnthropicContent(context.stateTurn),
|
|
267
|
+
});
|
|
268
|
+
}
|
|
223
269
|
if (msgs.length === 0) {
|
|
224
270
|
msgs.push({ role: "user", content: "Proceed." });
|
|
225
271
|
}
|
|
@@ -233,6 +279,79 @@ export class AnthropicProvider {
|
|
|
233
279
|
this.nativeAssistantBlocks.set(assistantReplayKey(message), blocks);
|
|
234
280
|
}
|
|
235
281
|
}
|
|
282
|
+
/** Anthropic accepts at most this many cache_control breakpoints per request. */
|
|
283
|
+
const MAX_CACHE_BREAKPOINTS = 4;
|
|
284
|
+
/**
|
|
285
|
+
* Number of rolling cache breakpoints to spend on the message history. Anthropic
|
|
286
|
+
* allows 4 cache_control breakpoints total; the static system/tools prefix
|
|
287
|
+
* consumes up to 2 (systemStable + systemKnowledge), leaving 2 for the history.
|
|
288
|
+
*/
|
|
289
|
+
const MESSAGE_CACHE_BREAKPOINTS = 2;
|
|
290
|
+
/**
|
|
291
|
+
* Regression guard: fail loudly if the static (system + tools) breakpoints plus
|
|
292
|
+
* the rolling message budget could exceed Anthropic's hard limit, instead of
|
|
293
|
+
* letting the API reject the request with an opaque 400. Uses the worst-case
|
|
294
|
+
* message count (`MESSAGE_CACHE_BREAKPOINTS`), so it can only fire if a future
|
|
295
|
+
* change adds a system partition or raises the message budget.
|
|
296
|
+
*/
|
|
297
|
+
function assertCacheBudget(system, toolCount) {
|
|
298
|
+
const systemBreakpoints = Array.isArray(system) ? system.length : 0;
|
|
299
|
+
const toolBreakpoints = toolCount > 0 && !Array.isArray(system) ? 1 : 0;
|
|
300
|
+
const worstCase = systemBreakpoints + toolBreakpoints + MESSAGE_CACHE_BREAKPOINTS;
|
|
301
|
+
if (worstCase > MAX_CACHE_BREAKPOINTS) {
|
|
302
|
+
throw new Error(`Anthropic cache_control budget exceeded: ${systemBreakpoints} system + ${toolBreakpoints} tool + ${MESSAGE_CACHE_BREAKPOINTS} message > ${MAX_CACHE_BREAKPOINTS}`);
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
/**
|
|
306
|
+
* Place the (≤2) message-history cache breakpoints. The final message always gets
|
|
307
|
+
* one — it writes the current full prefix for the next turn to read. The second is
|
|
308
|
+
* placed by one of two strategies:
|
|
309
|
+
*
|
|
310
|
+
* • **Deep anchor (P1-E)** — when `frozenPrefixLen` marks a distinct frozen prefix
|
|
311
|
+
* (the compaction boundary), pin the second breakpoint there. It is byte-stable
|
|
312
|
+
* across turns, so `[0..frozen]` is re-read cheaply every turn and is immune to
|
|
313
|
+
* the 20-block lookback miss that strikes heavy tool turns (>20 blocks/turn); the
|
|
314
|
+
* tail breakpoint then writes only the incremental `[frozen..tail]`.
|
|
315
|
+
* • **Rolling fallback** — otherwise (older binding / no compaction yet / whole
|
|
316
|
+
* render hot), roll the second breakpoint to the nearest preceding user turn, the
|
|
317
|
+
* previous turn's read anchor (Anthropic's 20-block lookback bridges light turns).
|
|
318
|
+
*
|
|
319
|
+
* Without any of this the cached prefix stops at the end of `system` and every turn
|
|
320
|
+
* re-bills the entire tool-result history at full price (~quadratic cumulative cost).
|
|
321
|
+
* cache_control attaches to the last content block of each target, promoting a bare
|
|
322
|
+
* string body to a text block.
|
|
323
|
+
*/
|
|
324
|
+
function applyMessageCacheControl(msgs, frozenPrefixLen) {
|
|
325
|
+
if (!msgs.length)
|
|
326
|
+
return;
|
|
327
|
+
const targets = new Set([msgs.length - 1]);
|
|
328
|
+
if (typeof frozenPrefixLen === "number" && frozenPrefixLen >= 1 && frozenPrefixLen < msgs.length) {
|
|
329
|
+
// Deep anchor at the frozen-prefix boundary (last frozen turn). Fixed between compactions.
|
|
330
|
+
targets.add(frozenPrefixLen - 1);
|
|
331
|
+
}
|
|
332
|
+
else {
|
|
333
|
+
for (let i = msgs.length - 2; i >= 0 && targets.size < MESSAGE_CACHE_BREAKPOINTS; i--) {
|
|
334
|
+
if (msgs[i].role === "user")
|
|
335
|
+
targets.add(i);
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
for (const idx of targets)
|
|
339
|
+
markLastBlockCacheable(msgs[idx]);
|
|
340
|
+
}
|
|
341
|
+
/** Attach an ephemeral cache breakpoint to a message's final content block. */
|
|
342
|
+
function markLastBlockCacheable(msg) {
|
|
343
|
+
const cache_control = { type: "ephemeral" };
|
|
344
|
+
if (typeof msg.content === "string") {
|
|
345
|
+
if (!msg.content)
|
|
346
|
+
return; // don't synthesize an empty (API-rejected) text block
|
|
347
|
+
msg.content = [{ type: "text", text: msg.content, cache_control }];
|
|
348
|
+
return;
|
|
349
|
+
}
|
|
350
|
+
if (Array.isArray(msg.content) && msg.content.length) {
|
|
351
|
+
const last = msg.content[msg.content.length - 1];
|
|
352
|
+
last.cache_control = cache_control;
|
|
353
|
+
}
|
|
354
|
+
}
|
|
236
355
|
/**
|
|
237
356
|
* Reconstruct Anthropic assistant content blocks from a neutral transcript when
|
|
238
357
|
* no provider replay was persisted. Only meaningful for tool-use turns: a plain
|
package/dist/providers/base.d.ts
CHANGED
|
@@ -16,12 +16,50 @@ export declare class CircuitBreaker {
|
|
|
16
16
|
*/
|
|
17
17
|
export declare const INTERNAL_EXTENSION_KEYS: readonly string[];
|
|
18
18
|
export declare function omitExtensionKeys(extensions: Record<string, unknown> | undefined, keys: readonly string[]): Record<string, unknown>;
|
|
19
|
+
/**
|
|
20
|
+
* Cached-prompt-token count from an OpenAI-compatible usage object. Covers the
|
|
21
|
+
* standard `prompt_tokens_details.cached_tokens` (OpenAI, Qwen, MiniMax, GLM,
|
|
22
|
+
* Kimi) and DeepSeek's `prompt_cache_hit_tokens`. These caches bill reads only,
|
|
23
|
+
* so there is no separate cache-creation count. The figure is a subset of
|
|
24
|
+
* `prompt_tokens` (the full prompt), surfaced for cost visibility — it must not
|
|
25
|
+
* be subtracted from the input count the kernel uses for context accounting.
|
|
26
|
+
*/
|
|
27
|
+
export declare function openAICachedPromptTokens(usage: unknown): number;
|
|
28
|
+
/**
|
|
29
|
+
* Prompt-cache hit rate for one usage record: the fraction of the full prompt
|
|
30
|
+
* served from cache this request (`cacheReadInputTokens / inputTokens`, clamped to
|
|
31
|
+
* [0,1]). Returns 0 when the prompt size is unknown. This is the headline metric
|
|
32
|
+
* for the prefix-cache work (P0-A) — across a long, append-only session it should
|
|
33
|
+
* climb and stay high; a sustained drop means the cacheable prefix is drifting.
|
|
34
|
+
*/
|
|
35
|
+
export declare function cacheHitRate(usage: {
|
|
36
|
+
inputTokens?: number;
|
|
37
|
+
cacheReadInputTokens?: number;
|
|
38
|
+
}): number;
|
|
39
|
+
/**
|
|
40
|
+
* Deterministic short key for OpenAI's `prompt_cache_key` — groups requests that
|
|
41
|
+
* share a cacheable prefix (same system prompt + tool set) onto the same cache
|
|
42
|
+
* routing, improving automatic prefix-cache hit rates without any caller input.
|
|
43
|
+
* FNV-1a over the parts; stable across processes, no crypto dependency.
|
|
44
|
+
*/
|
|
45
|
+
export declare function stablePromptCacheKey(parts: string[]): string;
|
|
19
46
|
export declare function normalizeToolCall(id: string, name: string, args: unknown): {
|
|
20
47
|
id: string;
|
|
21
48
|
name: string;
|
|
22
49
|
arguments: string;
|
|
23
50
|
} | null;
|
|
24
51
|
export declare function toAnthropicContent(msg: Message): string | Array<Record<string, unknown>>;
|
|
52
|
+
/**
|
|
53
|
+
* History turns with the volatile State turn appended as the latest turn, for
|
|
54
|
+
* providers that render it inline (OpenAI-family, Gemini, Ollama). Appending
|
|
55
|
+
* (rather than prepending) keeps the history a byte-stable prefix so these
|
|
56
|
+
* providers' automatic prefix caches (OpenAI / Gemini implicit / Ollama KV) hit
|
|
57
|
+
* across turns — the volatile state is the uncached tail. Anthropic does the
|
|
58
|
+
* equivalent explicitly (append after the cache breakpoint — see
|
|
59
|
+
* AnthropicProvider.buildMessages). When `stateTurn` is absent (un-rebuilt
|
|
60
|
+
* binding) the State turn is still inside `turns`, so this returns `turns` as-is.
|
|
61
|
+
*/
|
|
62
|
+
export declare function turnsWithStateAppended(context: RenderedContext): Message[];
|
|
25
63
|
/** Convert RenderedContext.turns to Anthropic messages array.
|
|
26
64
|
* `turns` contains only user / assistant / tool roles — no system filtering needed. */
|
|
27
65
|
export declare function toAnthropicMessages(turns: Message[], nativeReplay?: (message: Message) => Array<Record<string, unknown>> | undefined): Array<Record<string, unknown>>;
|
package/dist/providers/base.js
CHANGED
|
Binary file
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { AnthropicProvider } from "./anthropic.js";
|
|
2
2
|
import { OpenAIChatProvider } from "./openai.js";
|
|
3
|
-
import { DeepSeekProvider } from "./deepseek.js";
|
|
4
|
-
import { KimiProvider } from "./kimi.js";
|
|
3
|
+
import { DeepSeekProvider, DeepSeekAnthropicProvider } from "./deepseek.js";
|
|
4
|
+
import { KimiProvider, KimiAnthropicProvider } from "./kimi.js";
|
|
5
5
|
import { OpenAIResponsesProvider } from "./openai-responses.js";
|
|
6
6
|
import { MiniMaxAnthropicProvider, MiniMaxOpenAIProvider } from "./minimax.js";
|
|
7
|
-
import { QwenProvider } from "./qwen.js";
|
|
7
|
+
import { QwenProvider, QwenAnthropicProvider } from "./qwen.js";
|
|
8
8
|
import { GeminiProvider } from "./gemini.js";
|
|
9
|
-
import { GLMProvider } from "./glm.js";
|
|
9
|
+
import { GLMProvider, GLMAnthropicProvider } from "./glm.js";
|
|
10
10
|
import { endpointProfiles, getModelProfile, modelProfiles } from "./profiles.js";
|
|
11
11
|
export function createProvider(options) {
|
|
12
12
|
const profile = isModelProfileId(options.model) ? getModelProfile(options.model) : undefined;
|
|
@@ -48,18 +48,30 @@ export function createProvider(options) {
|
|
|
48
48
|
if (providerId === "minimax" && endpoint.protocol === "openai-chat") {
|
|
49
49
|
return new MiniMaxOpenAIProvider(options.apiKey, model, options.retry, baseURL);
|
|
50
50
|
}
|
|
51
|
+
if (providerId === "deepseek" && endpoint.protocol === "anthropic-messages") {
|
|
52
|
+
return new DeepSeekAnthropicProvider(options.apiKey, model, options.retry, baseURL);
|
|
53
|
+
}
|
|
51
54
|
if (providerId === "deepseek" && endpoint.protocol === "openai-chat") {
|
|
52
55
|
return new DeepSeekProvider(options.apiKey, model, options.retry, baseURL);
|
|
53
56
|
}
|
|
57
|
+
if (providerId === "kimi" && endpoint.protocol === "anthropic-messages") {
|
|
58
|
+
return new KimiAnthropicProvider(options.apiKey, model, options.retry, baseURL);
|
|
59
|
+
}
|
|
54
60
|
if (providerId === "kimi" && endpoint.protocol === "openai-chat") {
|
|
55
61
|
return new KimiProvider(options.apiKey, model, options.retry, baseURL);
|
|
56
62
|
}
|
|
63
|
+
if (providerId === "qwen" && endpoint.protocol === "anthropic-messages") {
|
|
64
|
+
return new QwenAnthropicProvider(options.apiKey, model, options.retry, baseURL);
|
|
65
|
+
}
|
|
57
66
|
if (providerId === "qwen" && endpoint.protocol === "openai-chat") {
|
|
58
67
|
return new QwenProvider(options.apiKey, model, options.retry, baseURL);
|
|
59
68
|
}
|
|
60
69
|
if (providerId === "gemini" && endpoint.protocol === "gemini") {
|
|
61
70
|
return new GeminiProvider(options.apiKey, model, options.retry, baseURL);
|
|
62
71
|
}
|
|
72
|
+
if (providerId === "glm" && endpoint.protocol === "anthropic-messages") {
|
|
73
|
+
return new GLMAnthropicProvider(options.apiKey, model, options.retry, baseURL);
|
|
74
|
+
}
|
|
63
75
|
if (providerId === "glm" && endpoint.protocol === "openai-chat") {
|
|
64
76
|
return new GLMProvider(options.apiKey, model, options.retry, baseURL);
|
|
65
77
|
}
|
|
@@ -82,11 +94,11 @@ function defaultEndpointForProvider(providerId) {
|
|
|
82
94
|
anthropic: "anthropic.messages",
|
|
83
95
|
openai: "openai.chat",
|
|
84
96
|
minimax: "minimax.anthropic",
|
|
85
|
-
deepseek: "deepseek.
|
|
86
|
-
kimi: "kimi.
|
|
87
|
-
qwen: "qwen.
|
|
97
|
+
deepseek: "deepseek.anthropic",
|
|
98
|
+
kimi: "kimi.anthropic",
|
|
99
|
+
qwen: "qwen.anthropic",
|
|
88
100
|
gemini: "gemini.google",
|
|
89
|
-
glm: "glm.
|
|
101
|
+
glm: "glm.anthropic",
|
|
90
102
|
};
|
|
91
103
|
return defaults[providerId];
|
|
92
104
|
}
|
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
import type { Message, ProviderDescriptor, RenderedContext, ToolSchema, StreamEvent, RuntimePolicy } from "../types.js";
|
|
2
|
+
import { AnthropicProvider } from "./anthropic.js";
|
|
2
3
|
import { OpenAIChatProvider } from "./openai.js";
|
|
4
|
+
/**
|
|
5
|
+
* DeepSeek over its Anthropic-compatible endpoint.
|
|
6
|
+
*/
|
|
7
|
+
export declare class DeepSeekAnthropicProvider extends AnthropicProvider {
|
|
8
|
+
constructor(apiKey: string, model?: string, retry?: {
|
|
9
|
+
maxRetries: number;
|
|
10
|
+
baseDelay: number;
|
|
11
|
+
}, baseURL?: string);
|
|
12
|
+
protected providerName(): string;
|
|
13
|
+
runtimePolicy(): RuntimePolicy;
|
|
14
|
+
}
|
|
3
15
|
export declare class DeepSeekProvider extends OpenAIChatProvider {
|
|
4
16
|
constructor(apiKey: string, model?: string, retry?: {
|
|
5
17
|
maxRetries: number;
|
|
@@ -1,15 +1,32 @@
|
|
|
1
|
+
import { AnthropicProvider } from "./anthropic.js";
|
|
1
2
|
import { OpenAIChatProvider } from "./openai.js";
|
|
2
3
|
import { endpointProfiles } from "./profiles.js";
|
|
3
|
-
import { omitExtensionKeys } from "./base.js";
|
|
4
|
-
const DEEPSEEK_BASE = endpointProfiles["deepseek.openai"].baseURL;
|
|
4
|
+
import { omitExtensionKeys, openAICachedPromptTokens } from "./base.js";
|
|
5
5
|
const DEEPSEEK_POLICIES = {
|
|
6
6
|
"deepseek-chat": { maxTurns: 25 },
|
|
7
7
|
"deepseek-reasoner": { maxTurns: 50 },
|
|
8
8
|
"deepseek-v4-flash": { maxTurns: 20 },
|
|
9
9
|
"deepseek-v4-pro": { maxTurns: 35 },
|
|
10
10
|
};
|
|
11
|
+
/**
|
|
12
|
+
* DeepSeek over its Anthropic-compatible endpoint.
|
|
13
|
+
*/
|
|
14
|
+
export class DeepSeekAnthropicProvider extends AnthropicProvider {
|
|
15
|
+
constructor(apiKey, model = "deepseek-v4-flash", retry, baseURL = endpointProfiles["deepseek.anthropic"].baseURL) {
|
|
16
|
+
super(apiKey, model, retry, {
|
|
17
|
+
baseURL,
|
|
18
|
+
authMode: "api-key",
|
|
19
|
+
});
|
|
20
|
+
}
|
|
21
|
+
providerName() {
|
|
22
|
+
return "deepseek";
|
|
23
|
+
}
|
|
24
|
+
runtimePolicy() {
|
|
25
|
+
return DEEPSEEK_POLICIES[this.model] ?? {};
|
|
26
|
+
}
|
|
27
|
+
}
|
|
11
28
|
export class DeepSeekProvider extends OpenAIChatProvider {
|
|
12
|
-
constructor(apiKey, model = "deepseek-v4-flash", retry, baseURL =
|
|
29
|
+
constructor(apiKey, model = "deepseek-v4-flash", retry, baseURL = endpointProfiles["deepseek.openai"].baseURL) {
|
|
13
30
|
super(apiKey, model, retry, baseURL);
|
|
14
31
|
}
|
|
15
32
|
runtimePolicy() {
|
|
@@ -103,11 +120,13 @@ export class DeepSeekProvider extends OpenAIChatProvider {
|
|
|
103
120
|
let totalTokens = 0;
|
|
104
121
|
let inputTokens = 0;
|
|
105
122
|
let outputTokens = 0;
|
|
123
|
+
let cacheReadTokens = 0;
|
|
106
124
|
for await (const chunk of stream) {
|
|
107
125
|
if (chunk.usage) {
|
|
108
126
|
totalTokens = chunk.usage.total_tokens;
|
|
109
127
|
inputTokens = chunk.usage.prompt_tokens ?? 0;
|
|
110
128
|
outputTokens = chunk.usage.completion_tokens ?? 0;
|
|
129
|
+
cacheReadTokens = openAICachedPromptTokens(chunk.usage);
|
|
111
130
|
continue;
|
|
112
131
|
}
|
|
113
132
|
const choice = chunk.choices[0];
|
|
@@ -173,7 +192,7 @@ export class DeepSeekProvider extends OpenAIChatProvider {
|
|
|
173
192
|
yield { type: "tool_call", id: tb.id, name: tb.name, arguments: args };
|
|
174
193
|
}
|
|
175
194
|
if (totalTokens > 0)
|
|
176
|
-
yield { type: "usage", totalTokens, inputTokens, outputTokens };
|
|
195
|
+
yield { type: "usage", totalTokens, inputTokens, outputTokens, ...(cacheReadTokens > 0 ? { cacheReadInputTokens: cacheReadTokens } : {}) };
|
|
177
196
|
}
|
|
178
197
|
rememberDeepSeekReplay(content, toolCalls, reasoningContent, nativeToolCalls) {
|
|
179
198
|
if (typeof reasoningContent !== "string" || !reasoningContent.trim())
|
package/dist/providers/gemini.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { GoogleGenerativeAI } from "@google/generative-ai";
|
|
2
2
|
import { withServerRuntimeGuard } from "../runtime/server.js";
|
|
3
|
-
import { CircuitBreaker, normalizeToolCall } from "./base.js";
|
|
3
|
+
import { CircuitBreaker, normalizeToolCall, turnsWithStateAppended } from "./base.js";
|
|
4
4
|
import { endpointProfiles } from "./profiles.js";
|
|
5
5
|
const GEMINI_BASE = endpointProfiles["gemini.google"].baseURL;
|
|
6
6
|
const GEMINI_POLICIES = {
|
|
@@ -56,8 +56,23 @@ export function buildContents(turns) {
|
|
|
56
56
|
parts.push({ functionCall: { name: tc.name, args } });
|
|
57
57
|
}
|
|
58
58
|
}
|
|
59
|
-
|
|
59
|
+
// Multimodal: render contentParts (text + image) when present, else the plain
|
|
60
|
+
// text body. Without this, image inputs to Gemini were silently dropped.
|
|
61
|
+
if (msg.contentParts?.length) {
|
|
62
|
+
for (const p of msg.contentParts) {
|
|
63
|
+
if (p.type === "text")
|
|
64
|
+
parts.push({ text: p.text });
|
|
65
|
+
else if (p.type === "image") {
|
|
66
|
+
if (p.data)
|
|
67
|
+
parts.push({ inlineData: { mimeType: p.mediaType ?? "image/png", data: p.data } });
|
|
68
|
+
else if (p.url)
|
|
69
|
+
parts.push({ fileData: { mimeType: p.mediaType ?? "image/png", fileUri: p.url } });
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
else if (msg.content) {
|
|
60
74
|
parts.push({ text: msg.content });
|
|
75
|
+
}
|
|
61
76
|
if (parts.length)
|
|
62
77
|
contents.push({ role, parts });
|
|
63
78
|
}
|
|
@@ -96,7 +111,7 @@ export class GeminiProvider {
|
|
|
96
111
|
if (this.circuit.isOpen())
|
|
97
112
|
throw new Error("Circuit breaker open");
|
|
98
113
|
const system = context.systemText || undefined;
|
|
99
|
-
const contents = buildContents(context
|
|
114
|
+
const contents = buildContents(turnsWithStateAppended(context));
|
|
100
115
|
const geminiTools = buildTools(tools);
|
|
101
116
|
let lastErr;
|
|
102
117
|
for (let i = 0; i < this.maxRetries; i++) {
|
|
@@ -140,7 +155,7 @@ export class GeminiProvider {
|
|
|
140
155
|
}
|
|
141
156
|
async *stream(context, tools, extensions) {
|
|
142
157
|
const system = context.systemText || undefined;
|
|
143
|
-
const contents = buildContents(context
|
|
158
|
+
const contents = buildContents(turnsWithStateAppended(context));
|
|
144
159
|
const geminiTools = buildTools(tools);
|
|
145
160
|
const m = this.genAI.getGenerativeModel({
|
|
146
161
|
...this.modelExtensions(extensions),
|
|
@@ -165,11 +180,15 @@ export class GeminiProvider {
|
|
|
165
180
|
}
|
|
166
181
|
const usage = (await result.response).usageMetadata;
|
|
167
182
|
if (usage?.totalTokenCount) {
|
|
183
|
+
// Gemini implicit/explicit cache hits are reported as cachedContentTokenCount,
|
|
184
|
+
// a subset of promptTokenCount (which stays the full prompt for accounting).
|
|
185
|
+
const cachedTokens = usage.cachedContentTokenCount ?? 0;
|
|
168
186
|
yield {
|
|
169
187
|
type: "usage",
|
|
170
188
|
totalTokens: usage.totalTokenCount,
|
|
171
189
|
inputTokens: usage.promptTokenCount ?? 0,
|
|
172
190
|
outputTokens: usage.candidatesTokenCount ?? 0,
|
|
191
|
+
...(cachedTokens > 0 ? { cacheReadInputTokens: cachedTokens } : {}),
|
|
173
192
|
};
|
|
174
193
|
}
|
|
175
194
|
}
|
package/dist/providers/glm.d.ts
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
import type { ProviderDescriptor, RuntimePolicy } from "../types.js";
|
|
2
|
+
import { AnthropicProvider } from "./anthropic.js";
|
|
2
3
|
import { OpenAIChatProvider } from "./openai.js";
|
|
4
|
+
/**
|
|
5
|
+
* GLM over its Anthropic-compatible endpoint.
|
|
6
|
+
*/
|
|
7
|
+
export declare class GLMAnthropicProvider extends AnthropicProvider {
|
|
8
|
+
constructor(apiKey: string, model?: string, retry?: {
|
|
9
|
+
maxRetries: number;
|
|
10
|
+
baseDelay: number;
|
|
11
|
+
}, baseURL?: string);
|
|
12
|
+
protected providerName(): string;
|
|
13
|
+
runtimePolicy(): RuntimePolicy;
|
|
14
|
+
}
|
|
3
15
|
export declare class GLMProvider extends OpenAIChatProvider {
|
|
4
16
|
constructor(apiKey: string, model?: string, retry?: {
|
|
5
17
|
maxRetries: number;
|
package/dist/providers/glm.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
+
import { AnthropicProvider } from "./anthropic.js";
|
|
1
2
|
import { OpenAIChatProvider } from "./openai.js";
|
|
2
3
|
import { endpointProfiles } from "./profiles.js";
|
|
3
|
-
const GLM_BASE = endpointProfiles["glm.openai"].baseURL;
|
|
4
4
|
const GLM_POLICIES = {
|
|
5
5
|
"glm-5.1": { maxTurns: 50 },
|
|
6
6
|
"glm/glm-5.1": { maxTurns: 50 },
|
|
@@ -11,8 +11,25 @@ const GLM_POLICIES = {
|
|
|
11
11
|
"glm-4-air": { maxTurns: 20 },
|
|
12
12
|
"glm/glm-4-air": { maxTurns: 20 },
|
|
13
13
|
};
|
|
14
|
+
/**
|
|
15
|
+
* GLM over its Anthropic-compatible endpoint.
|
|
16
|
+
*/
|
|
17
|
+
export class GLMAnthropicProvider extends AnthropicProvider {
|
|
18
|
+
constructor(apiKey, model = "glm-5.1", retry, baseURL = endpointProfiles["glm.anthropic"].baseURL) {
|
|
19
|
+
super(apiKey, model, retry, {
|
|
20
|
+
baseURL,
|
|
21
|
+
authMode: "api-key",
|
|
22
|
+
});
|
|
23
|
+
}
|
|
24
|
+
providerName() {
|
|
25
|
+
return "glm";
|
|
26
|
+
}
|
|
27
|
+
runtimePolicy() {
|
|
28
|
+
return GLM_POLICIES[this.model] ?? {};
|
|
29
|
+
}
|
|
30
|
+
}
|
|
14
31
|
export class GLMProvider extends OpenAIChatProvider {
|
|
15
|
-
constructor(apiKey, model = "glm-5.1", retry, baseURL =
|
|
32
|
+
constructor(apiKey, model = "glm-5.1", retry, baseURL = endpointProfiles["glm.openai"].baseURL) {
|
|
16
33
|
super(apiKey, model, retry, baseURL);
|
|
17
34
|
}
|
|
18
35
|
runtimePolicy() {
|
package/dist/providers/kimi.d.ts
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
1
|
import type { ProviderDescriptor, RuntimePolicy } from "../types.js";
|
|
2
|
+
import { AnthropicProvider } from "./anthropic.js";
|
|
2
3
|
import { OpenAIChatProvider } from "./openai.js";
|
|
4
|
+
/**
|
|
5
|
+
* Kimi over its Anthropic-compatible endpoint.
|
|
6
|
+
*/
|
|
7
|
+
export declare class KimiAnthropicProvider extends AnthropicProvider {
|
|
8
|
+
constructor(apiKey: string, model?: string, retry?: {
|
|
9
|
+
maxRetries: number;
|
|
10
|
+
baseDelay: number;
|
|
11
|
+
}, baseURL?: string);
|
|
12
|
+
protected providerName(): string;
|
|
13
|
+
runtimePolicy(): RuntimePolicy;
|
|
14
|
+
}
|
|
3
15
|
export declare class KimiProvider extends OpenAIChatProvider {
|
|
4
16
|
constructor(apiKey: string, model?: string, retry?: {
|
|
5
17
|
maxRetries: number;
|