@deepstrike/sdk 0.2.10 → 0.2.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +38 -0
  2. package/dist/harness/harness.d.ts +5 -0
  3. package/dist/harness/harness.js +12 -1
  4. package/dist/index.d.ts +7 -4
  5. package/dist/index.js +6 -4
  6. package/dist/providers/anthropic.d.ts +7 -0
  7. package/dist/providers/anthropic.js +129 -10
  8. package/dist/providers/base.d.ts +38 -0
  9. package/dist/providers/base.js +0 -0
  10. package/dist/providers/catalog.js +20 -8
  11. package/dist/providers/deepseek.d.ts +12 -0
  12. package/dist/providers/deepseek.js +23 -4
  13. package/dist/providers/gemini.js +23 -4
  14. package/dist/providers/glm.d.ts +12 -0
  15. package/dist/providers/glm.js +19 -2
  16. package/dist/providers/kimi.d.ts +12 -0
  17. package/dist/providers/kimi.js +19 -2
  18. package/dist/providers/minimax.js +4 -2
  19. package/dist/providers/ollama.js +2 -2
  20. package/dist/providers/openai-chat.js +3 -2
  21. package/dist/providers/openai-responses.js +13 -0
  22. package/dist/providers/openai.d.ts +7 -0
  23. package/dist/providers/openai.js +15 -2
  24. package/dist/providers/profiles.d.ts +52 -28
  25. package/dist/providers/profiles.js +52 -28
  26. package/dist/providers/qwen.d.ts +12 -0
  27. package/dist/providers/qwen.js +23 -4
  28. package/dist/runtime/output-schema.d.ts +14 -0
  29. package/dist/runtime/output-schema.js +113 -0
  30. package/dist/runtime/reducers.d.ts +15 -0
  31. package/dist/runtime/reducers.js +57 -0
  32. package/dist/runtime/runner.d.ts +24 -1
  33. package/dist/runtime/runner.js +185 -15
  34. package/dist/runtime/session-log.d.ts +9 -1
  35. package/dist/runtime/session-repair.d.ts +14 -0
  36. package/dist/runtime/session-repair.js +15 -0
  37. package/dist/runtime/sub-agent-orchestrator.js +14 -1
  38. package/dist/types/agent.d.ts +43 -1
  39. package/dist/types/agent.js +105 -18
  40. package/dist/types.d.ts +42 -1
  41. package/package.json +2 -2
package/README.md CHANGED
@@ -145,6 +145,44 @@ Spool, page-out, signals, processes, budgets, and memory events land in `Session
145
145
 
146
146
  ---
147
147
 
148
+ ## Dynamic workflows
149
+
150
+ Instead of planning **and** executing a hard task in one long context window, hand the kernel a declarative DAG and let it spawn a fresh-context sub-agent per node. The kernel owns the control flow (gate · budget · suspend-on-join · resume); your SDK runs the agents. See the [top-level overview](../README.md#the-six-harness-patterns-as-first-class-kernel-nodes) for the full pattern catalog.
151
+
152
+ ```ts
153
+ // One fresh-context verifier per rule (no inherited author context → can't rubber-stamp),
154
+ // then a skeptic that reviews their flags. The kernel spawns the 3 verifiers as one gated
155
+ // batch, suspends on the join, and runs the skeptic once they complete.
156
+ const outcome = await runner.runWorkflow({
157
+ nodes: [
158
+ { task: "Rule: money is integer cents — violated?", role: "verify" },
159
+ { task: "Rule: all errors propagate — violated?", role: "verify" },
160
+ { task: "Rule: timestamps are UTC — violated?", role: "verify" },
161
+ { task: "Skeptic: which flags are real violations?", role: "verify", dependsOn: [0, 1, 2] },
162
+ ],
163
+ })
164
+ // → { completed: ["wf-node0", … ], failed: [] }
165
+ ```
166
+
167
+ A node's `kind` selects the control-flow shape; the same executor drives them all, every spawn passing the syscall gate:
168
+
169
+ | Node `kind` | Behavior |
170
+ |---|---|
171
+ | `{ type: "spawn" }` (default) | Run the node's agent once |
172
+ | `{ type: "loop", maxIters }` | Re-run until the agent signals it's done, capped at `maxIters` |
173
+ | `{ type: "classify", branches }` | The classifier's result selects one branch; the rest are pruned |
174
+ | `{ type: "tournament", entrants }` | Generate N entrants, then a pairwise-judge bracket to one winner |
175
+ | `{ type: "reduce", reducer }` | **Tokenless host-compute** — a pure function (`dedupe_lines` / `merge_json_arrays` / `concat` / `count`, or your own via the `reducers` runner option) over the node's dependency outputs |
176
+
177
+ ### 0.2.11 capabilities
178
+
179
+ - **Runtime fan-out** — give a node the `submitWorkflowNodesTool` and its agent can append nodes to the live DAG mid-run (true loop-until-done; one verifier per claim it discovers). Recorded and replayed on `resumeWorkflow`.
180
+ - **Quarantine, no escape** — set `trust: "quarantined"` on a node that reads untrusted content; it's denied write-capable isolation in-kernel, and any nodes it submits are coerced to quarantined too (no privilege escalation).
181
+ - **Structured output** — set `outputSchema` on a node; the runner instructs the agent, validates the result against the JSON-Schema subset, and re-runs once with the errors on mismatch. A node that never conforms fails (its dependents starve).
182
+ - **Budget as signal** — with a `maxWorkflowNodes` / `maxConcurrentSubagents` quota installed, each spawned node's goal carries its remaining headroom so a coordinator can size its fan-out to fit.
183
+
184
+ ---
185
+
148
186
  ## Providers
149
187
 
150
188
  | Class | Backend | Notes |
@@ -26,6 +26,8 @@ export interface HarnessOutcome {
26
26
  overallScore?: number;
27
27
  feedback?: string;
28
28
  details?: CriterionResult[];
29
+ /** R3-1: nodes the agent submitted via `submit_workflow_nodes` while running under the harness. */
30
+ submittedNodes?: import("../types/agent.js").WorkflowNodeSpec[];
29
31
  }
30
32
  export interface Verdict {
31
33
  passed: boolean;
@@ -55,6 +57,9 @@ export type HarnessEvent = {
55
57
  callId: string;
56
58
  content: string;
57
59
  isError: boolean;
60
+ } | {
61
+ type: "workflow_nodes_submitted";
62
+ nodes: import("../types/agent.js").WorkflowNodeSpec[];
58
63
  } | {
59
64
  type: "supervising";
60
65
  } | {
@@ -61,8 +61,14 @@ export class HarnessLoop {
61
61
  }
62
62
  async run(request) {
63
63
  let last;
64
- for await (const evt of this.stream(request))
64
+ // R3-1: collect nodes the agent submitted while running under the harness, so dynamic fan-out
65
+ // works in harness mode too (not just the plain streaming path).
66
+ const submittedNodes = [];
67
+ for await (const evt of this.stream(request)) {
65
68
  last = evt;
69
+ if (evt.type === "workflow_nodes_submitted")
70
+ submittedNodes.push(...evt.nodes);
71
+ }
66
72
  const done = last?.type === "done" ? last : undefined;
67
73
  return {
68
74
  result: "",
@@ -73,6 +79,7 @@ export class HarnessLoop {
73
79
  overallScore: done?.verdict.overallScore,
74
80
  feedback: done?.verdict.feedback,
75
81
  details: done?.verdict.details,
82
+ ...(submittedNodes.length ? { submittedNodes } : {}),
76
83
  };
77
84
  }
78
85
  async *stream(request) {
@@ -107,6 +114,10 @@ export class HarnessLoop {
107
114
  const tr = evt;
108
115
  yield { type: "tool_result", callId: tr.callId, content: tr.content, isError: tr.isError };
109
116
  }
117
+ else if (evt.type === "workflow_nodes_submitted") {
118
+ const ws = evt;
119
+ yield { type: "workflow_nodes_submitted", nodes: ws.nodes };
120
+ }
110
121
  else if (evt.type === "done") {
111
122
  const d = evt;
112
123
  lastIterations = d.iterations;
package/dist/index.d.ts CHANGED
@@ -1,5 +1,7 @@
1
1
  export { RuntimeRunner, collectText } from "./runtime/runner.js";
2
2
  export type { RuntimeOptions, SchedulerBudget } from "./runtime/runner.js";
3
+ export { builtinReducers, resolveReducer } from "./runtime/reducers.js";
4
+ export type { Reducer, ReducerRegistry, ReducerInput } from "./runtime/reducers.js";
3
5
  export type { MemoryPolicy, MemoryWriteRateLimit, ResourceQuota } from "./kernel.js";
4
6
  export { KernelPrimitivesDashboard } from "./runtime/kernel-primitives-dashboard.js";
5
7
  export { FilteredExecutionPlane } from "./runtime/filtered-plane.js";
@@ -27,9 +29,10 @@ export { RemoteVpcPlane } from "./runtime/remote-vpc-plane.js";
27
29
  export type { RemoteVpcOptions } from "./runtime/remote-vpc-plane.js";
28
30
  export { AnthropicProvider } from "./providers/anthropic.js";
29
31
  export { OpenAIChatProvider, OpenAIProvider } from "./providers/openai.js";
30
- export { DeepSeekProvider } from "./providers/deepseek.js";
31
- export { KimiProvider } from "./providers/kimi.js";
32
- export { QwenProvider } from "./providers/qwen.js";
32
+ export { DeepSeekProvider, DeepSeekAnthropicProvider } from "./providers/deepseek.js";
33
+ export { KimiProvider, KimiAnthropicProvider } from "./providers/kimi.js";
34
+ export { QwenProvider, QwenAnthropicProvider } from "./providers/qwen.js";
35
+ export { GLMProvider, GLMAnthropicProvider } from "./providers/glm.js";
33
36
  export { GeminiProvider } from "./providers/gemini.js";
34
37
  export { MiniMaxAnthropicProvider, MiniMaxOpenAIProvider } from "./providers/minimax.js";
35
38
  export { OllamaProvider } from "./providers/ollama.js";
@@ -61,7 +64,7 @@ export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harne
61
64
  export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions, QualityGate } from "./harness/harness.js";
62
65
  export type { Message, ToolCall, ToolResult, ToolSchema, ContentPart, TextPart, ImagePart, AudioPart, StreamEvent, TextDelta, ThinkingDelta, ToolCallEvent, ToolChunk, ToolDeltaEvent, ToolSuspendEvent, ToolResultEvent, DoneEvent, ErrorEvent, PermissionRequestEvent, PermissionResolvedEvent, PermissionResponse, LLMProvider, RetryConfig, TokenUsage, ProviderToolSpec, ProviderRunState, ProviderReplay, RenderedContext, ReplayabilityAssessment, } from "./types.js";
63
66
  export type { AgentCapabilityFilter, AgentIdentity, AgentIsolation, AgentRunSpec, AgentProcessChangedObservation, ContextInheritance, KernelAgentRole, LoopResult, MilestoneCheckResult, MilestoneContract, MilestonePhase, MilestonePolicy, SubAgentResult, TerminationReason, WorkflowSpec, WorkflowNodeSpec, WorkflowTaskSpec, WorkflowSpawnInfo, } from "./types/agent.js";
64
- export { agentIdentitySub, agentRunSpecToKernel, milestoneCheckFail, milestoneCheckPass, milestoneCheckResultToKernel, subAgentResultToKernel, workflowSpecToKernel, fanoutSynthesize, generateAndFilter, verifyRules, } from "./types/agent.js";
67
+ export { agentIdentitySub, agentRunSpecToKernel, milestoneCheckFail, milestoneCheckPass, milestoneCheckResultToKernel, subAgentResultToKernel, workflowSpecToKernel, workflowNodeSpecToKernel, submitWorkflowNodesToKernel, submitWorkflowNodesTool, fanoutSynthesize, generateAndFilter, verifyRules, } from "./types/agent.js";
65
68
  export type { AcceptanceCriterion, VerificationContract, ContractCheckResult, } from "./collaboration/contract.js";
66
69
  export { ContractBuilder, formatContractForSystemPrompt, contractToCriteriaStrings, } from "./collaboration/contract.js";
67
70
  export { AgentPool } from "./collaboration/pool.js";
package/dist/index.js CHANGED
@@ -1,5 +1,6 @@
1
1
  // ── Runtime (Layer 1.5) ────────────────────────────────────────────────────
2
2
  export { RuntimeRunner, collectText } from "./runtime/runner.js";
3
+ export { builtinReducers, resolveReducer } from "./runtime/reducers.js";
3
4
  export { KernelPrimitivesDashboard } from "./runtime/kernel-primitives-dashboard.js";
4
5
  export { FilteredExecutionPlane } from "./runtime/filtered-plane.js";
5
6
  export { SubAgentOrchestrator, defaultSubAgentOrchestrator, spawnStandalone } from "./runtime/sub-agent-orchestrator.js";
@@ -16,9 +17,10 @@ export { RemoteVpcPlane } from "./runtime/remote-vpc-plane.js";
16
17
  // ── Providers ─────────────────────────────────────────────────────────────
17
18
  export { AnthropicProvider } from "./providers/anthropic.js";
18
19
  export { OpenAIChatProvider, OpenAIProvider } from "./providers/openai.js";
19
- export { DeepSeekProvider } from "./providers/deepseek.js";
20
- export { KimiProvider } from "./providers/kimi.js";
21
- export { QwenProvider } from "./providers/qwen.js";
20
+ export { DeepSeekProvider, DeepSeekAnthropicProvider } from "./providers/deepseek.js";
21
+ export { KimiProvider, KimiAnthropicProvider } from "./providers/kimi.js";
22
+ export { QwenProvider, QwenAnthropicProvider } from "./providers/qwen.js";
23
+ export { GLMProvider, GLMAnthropicProvider } from "./providers/glm.js";
22
24
  export { GeminiProvider } from "./providers/gemini.js";
23
25
  export { MiniMaxAnthropicProvider, MiniMaxOpenAIProvider } from "./providers/minimax.js";
24
26
  export { OllamaProvider } from "./providers/ollama.js";
@@ -41,7 +43,7 @@ export { PermissionManager, PermissionMode } from "./safety/permissions.js";
41
43
  export { Governance, governancePolicyToKernelEvent } from "./governance.js";
42
44
  // ── Harness ────────────────────────────────────────────────────────────────
43
45
  export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
44
- export { agentIdentitySub, agentRunSpecToKernel, milestoneCheckFail, milestoneCheckPass, milestoneCheckResultToKernel, subAgentResultToKernel, workflowSpecToKernel, fanoutSynthesize, generateAndFilter, verifyRules, } from "./types/agent.js";
46
+ export { agentIdentitySub, agentRunSpecToKernel, milestoneCheckFail, milestoneCheckPass, milestoneCheckResultToKernel, subAgentResultToKernel, workflowSpecToKernel, workflowNodeSpecToKernel, submitWorkflowNodesToKernel, submitWorkflowNodesTool, fanoutSynthesize, generateAndFilter, verifyRules, } from "./types/agent.js";
45
47
  export { ContractBuilder, formatContractForSystemPrompt, contractToCriteriaStrings, } from "./collaboration/contract.js";
46
48
  export { AgentPool } from "./collaboration/pool.js";
47
49
  export { KERNEL_ROLE_MAP } from "./collaboration/pool.js";
@@ -20,6 +20,13 @@ export declare class AnthropicProvider implements LLMProvider {
20
20
  descriptor(): ProviderDescriptor;
21
21
  peekProviderReplay(message: Pick<Message, "content" | "toolCalls">): ProviderReplay | undefined;
22
22
  seedProviderReplay(message: Pick<Message, "content" | "toolCalls">, replay: ProviderReplay): void;
23
+ /**
24
+ * Build tool definitions. A cache breakpoint is anchored on the final tool
25
+ * only when the system blocks won't carry one (`anchorCache`). When structured
26
+ * system blocks are present, their breakpoints already cache the tools prefix
27
+ * (tools render before system), so a redundant tool breakpoint would only burn
28
+ * one of Anthropic's 4 cache_control slots — slots the message history needs.
29
+ */
23
30
  private buildTools;
24
31
  complete(context: RenderedContext, tools: ToolSchema[], extensions?: Record<string, unknown>): Promise<Message>;
25
32
  stream(context: RenderedContext, tools: ToolSchema[], extensions?: Record<string, unknown>): AsyncIterable<StreamEvent>;
@@ -1,7 +1,7 @@
1
1
  import Anthropic from "@anthropic-ai/sdk";
2
2
  import { assistantReplayKey } from "../runtime/provider-replay.js";
3
3
  import { withServerRuntimeGuard } from "../runtime/server.js";
4
- import { CircuitBreaker, normalizeToolCall, omitExtensionKeys, toAnthropicMessages } from "./base.js";
4
+ import { CircuitBreaker, normalizeToolCall, omitExtensionKeys, toAnthropicContent, toAnthropicMessages } from "./base.js";
5
5
  const CLAUDE_POLICIES = {
6
6
  "claude-opus-4-1": { maxTurns: 50 },
7
7
  "claude-opus-4-7": { maxTurns: 50 },
@@ -70,12 +70,19 @@ export class AnthropicProvider {
70
70
  if (blocks.length)
71
71
  this.nativeAssistantBlocks.set(assistantReplayKey(message), blocks);
72
72
  }
73
- buildTools(tools) {
73
+ /**
74
+ * Build tool definitions. A cache breakpoint is anchored on the final tool
75
+ * only when the system blocks won't carry one (`anchorCache`). When structured
76
+ * system blocks are present, their breakpoints already cache the tools prefix
77
+ * (tools render before system), so a redundant tool breakpoint would only burn
78
+ * one of Anthropic's 4 cache_control slots — slots the message history needs.
79
+ */
80
+ buildTools(tools, anchorCache) {
74
81
  return tools.map((t, i) => ({
75
82
  name: t.name,
76
83
  description: t.description,
77
84
  input_schema: JSON.parse(t.parameters),
78
- ...(i === tools.length - 1 ? { cache_control: { type: "ephemeral" } } : {}),
85
+ ...(anchorCache && i === tools.length - 1 ? { cache_control: { type: "ephemeral" } } : {}),
79
86
  }));
80
87
  }
81
88
  async complete(context, tools, extensions) {
@@ -83,6 +90,7 @@ export class AnthropicProvider {
83
90
  throw new Error("Circuit breaker open");
84
91
  const system = this.buildSystem(context);
85
92
  const msgs = this.buildMessages(context);
93
+ assertCacheBudget(system, tools.length);
86
94
  const requestExtensions = this.requestExtensions(extensions);
87
95
  let lastErr;
88
96
  for (let i = 0; i < this.maxRetries; i++) {
@@ -93,7 +101,7 @@ export class AnthropicProvider {
93
101
  max_tokens: typeof extensions?.max_tokens === "number" ? extensions.max_tokens : 8096,
94
102
  ...(system ? { system } : {}),
95
103
  messages: msgs,
96
- ...(tools.length ? { tools: this.buildTools(tools) } : {}),
104
+ ...(tools.length ? { tools: this.buildTools(tools, !Array.isArray(system)) } : {}),
97
105
  }, extensions);
98
106
  this.circuit.recordSuccess();
99
107
  let content = "";
@@ -123,6 +131,7 @@ export class AnthropicProvider {
123
131
  async *stream(context, tools, extensions) {
124
132
  const system = this.buildSystem(context);
125
133
  const msgs = this.buildMessages(context);
134
+ assertCacheBudget(system, tools.length);
126
135
  const requestExtensions = this.requestExtensions(extensions);
127
136
  const toolBlocks = {};
128
137
  const nativeBlocks = {};
@@ -134,17 +143,36 @@ export class AnthropicProvider {
134
143
  max_tokens: typeof extensions?.max_tokens === "number" ? extensions.max_tokens : 8096,
135
144
  ...(system ? { system } : {}),
136
145
  messages: msgs,
137
- ...(tools.length ? { tools: this.buildTools(tools) } : {}),
146
+ ...(tools.length ? { tools: this.buildTools(tools, !Array.isArray(system)) } : {}),
138
147
  }, extensions);
139
- let totalTokens = 0;
148
+ let uncachedInput = 0;
149
+ let cacheReadTokens = 0;
150
+ let cacheCreationTokens = 0;
151
+ let outputTokens = 0;
140
152
  for await (const evt of stream) {
141
153
  if (evt.type === "message_start" || evt.type === "message_delta") {
142
154
  const usage = evt.usage ?? evt.message?.usage;
143
155
  if (usage) {
144
- const inputTokens = usage.input_tokens ?? 0;
145
- const outputTokens = usage.output_tokens ?? 0;
146
- totalTokens = inputTokens + outputTokens;
147
- yield { type: "usage", totalTokens, inputTokens, outputTokens };
156
+ // input + cache counts are cumulative and pinned at message_start; a
157
+ // later message_delta may omit them (null), so Math.max keeps the
158
+ // running totals from being clobbered back to zero.
159
+ uncachedInput = Math.max(uncachedInput, usage.input_tokens ?? 0);
160
+ cacheReadTokens = Math.max(cacheReadTokens, usage.cache_read_input_tokens ?? 0);
161
+ cacheCreationTokens = Math.max(cacheCreationTokens, usage.cache_creation_input_tokens ?? 0);
162
+ outputTokens = Math.max(outputTokens, usage.output_tokens ?? 0);
163
+ // inputTokens is the FULL prompt size (uncached + cache read + cache
164
+ // write). The kernel reads it as the authoritative prompt size for
165
+ // context-pressure/compaction — excluding cached tokens would make a
166
+ // cache-heavy turn look tiny and suppress compaction until a 413.
167
+ const inputTokens = uncachedInput + cacheReadTokens + cacheCreationTokens;
168
+ yield {
169
+ type: "usage",
170
+ totalTokens: inputTokens + outputTokens,
171
+ inputTokens,
172
+ outputTokens,
173
+ cacheReadInputTokens: cacheReadTokens,
174
+ cacheCreationInputTokens: cacheCreationTokens,
175
+ };
148
176
  }
149
177
  }
150
178
  else if (evt.type === "content_block_start") {
@@ -206,6 +234,12 @@ export class AnthropicProvider {
206
234
  : this.client.messages.stream(params));
207
235
  }
208
236
  buildSystem(context) {
237
+ // B3 note: the system shape is content-driven — 0 blocks (string), 1 block
238
+ // (stable only), or 2 blocks (stable + knowledge). The first turn `systemKnowledge`
239
+ // appears, the block count rises 1→2, which is a one-time prompt-cache invalidation
240
+ // (the knowledge prefix didn't exist to cache before). It is byte-stable thereafter;
241
+ // dynamic per-turn knowledge belongs in the uncached tail, not this block. An empty
242
+ // knowledge string is intentionally never emitted (the API rejects empty text blocks).
209
243
  if (!context.systemStable && !context.systemKnowledge) {
210
244
  return context.systemText || undefined;
211
245
  }
@@ -220,6 +254,18 @@ export class AnthropicProvider {
220
254
  }
221
255
  buildMessages(context) {
222
256
  const msgs = toAnthropicMessages(context.turns, message => this.nativeAssistantBlocks.get(assistantReplayKey(message)));
257
+ // Cache breakpoints anchor on the stable history; the volatile State turn is
258
+ // appended AFTER them as the uncached tail (so the history prefix re-reads
259
+ // across turns). On un-rebuilt bindings stateTurn is absent and the state is
260
+ // already inside `turns` — rendered as-is above. `frozenPrefixLen` (P1-E) pins
261
+ // the deep breakpoint at the compaction boundary; absent ⇒ rolling-pair fallback.
262
+ applyMessageCacheControl(msgs, context.frozenPrefixLen);
263
+ if (context.stateTurn) {
264
+ msgs.push({
265
+ role: context.stateTurn.role === "assistant" ? "assistant" : "user",
266
+ content: toAnthropicContent(context.stateTurn),
267
+ });
268
+ }
223
269
  if (msgs.length === 0) {
224
270
  msgs.push({ role: "user", content: "Proceed." });
225
271
  }
@@ -233,6 +279,79 @@ export class AnthropicProvider {
233
279
  this.nativeAssistantBlocks.set(assistantReplayKey(message), blocks);
234
280
  }
235
281
  }
282
+ /** Anthropic accepts at most this many cache_control breakpoints per request. */
283
+ const MAX_CACHE_BREAKPOINTS = 4;
284
+ /**
285
+ * Number of rolling cache breakpoints to spend on the message history. Anthropic
286
+ * allows 4 cache_control breakpoints total; the static system/tools prefix
287
+ * consumes up to 2 (systemStable + systemKnowledge), leaving 2 for the history.
288
+ */
289
+ const MESSAGE_CACHE_BREAKPOINTS = 2;
290
+ /**
291
+ * Regression guard: fail loudly if the static (system + tools) breakpoints plus
292
+ * the rolling message budget could exceed Anthropic's hard limit, instead of
293
+ * letting the API reject the request with an opaque 400. Uses the worst-case
294
+ * message count (`MESSAGE_CACHE_BREAKPOINTS`), so it can only fire if a future
295
+ * change adds a system partition or raises the message budget.
296
+ */
297
+ function assertCacheBudget(system, toolCount) {
298
+ const systemBreakpoints = Array.isArray(system) ? system.length : 0;
299
+ const toolBreakpoints = toolCount > 0 && !Array.isArray(system) ? 1 : 0;
300
+ const worstCase = systemBreakpoints + toolBreakpoints + MESSAGE_CACHE_BREAKPOINTS;
301
+ if (worstCase > MAX_CACHE_BREAKPOINTS) {
302
+ throw new Error(`Anthropic cache_control budget exceeded: ${systemBreakpoints} system + ${toolBreakpoints} tool + ${MESSAGE_CACHE_BREAKPOINTS} message > ${MAX_CACHE_BREAKPOINTS}`);
303
+ }
304
+ }
305
+ /**
306
+ * Place the (≤2) message-history cache breakpoints. The final message always gets
307
+ * one — it writes the current full prefix for the next turn to read. The second is
308
+ * placed by one of two strategies:
309
+ *
310
+ * • **Deep anchor (P1-E)** — when `frozenPrefixLen` marks a distinct frozen prefix
311
+ * (the compaction boundary), pin the second breakpoint there. It is byte-stable
312
+ * across turns, so `[0..frozen]` is re-read cheaply every turn and is immune to
313
+ * the 20-block lookback miss that strikes heavy tool turns (>20 blocks/turn); the
314
+ * tail breakpoint then writes only the incremental `[frozen..tail]`.
315
+ * • **Rolling fallback** — otherwise (older binding / no compaction yet / whole
316
+ * render hot), roll the second breakpoint to the nearest preceding user turn, the
317
+ * previous turn's read anchor (Anthropic's 20-block lookback bridges light turns).
318
+ *
319
+ * Without any of this the cached prefix stops at the end of `system` and every turn
320
+ * re-bills the entire tool-result history at full price (~quadratic cumulative cost).
321
+ * cache_control attaches to the last content block of each target, promoting a bare
322
+ * string body to a text block.
323
+ */
324
+ function applyMessageCacheControl(msgs, frozenPrefixLen) {
325
+ if (!msgs.length)
326
+ return;
327
+ const targets = new Set([msgs.length - 1]);
328
+ if (typeof frozenPrefixLen === "number" && frozenPrefixLen >= 1 && frozenPrefixLen < msgs.length) {
329
+ // Deep anchor at the frozen-prefix boundary (last frozen turn). Fixed between compactions.
330
+ targets.add(frozenPrefixLen - 1);
331
+ }
332
+ else {
333
+ for (let i = msgs.length - 2; i >= 0 && targets.size < MESSAGE_CACHE_BREAKPOINTS; i--) {
334
+ if (msgs[i].role === "user")
335
+ targets.add(i);
336
+ }
337
+ }
338
+ for (const idx of targets)
339
+ markLastBlockCacheable(msgs[idx]);
340
+ }
341
+ /** Attach an ephemeral cache breakpoint to a message's final content block. */
342
+ function markLastBlockCacheable(msg) {
343
+ const cache_control = { type: "ephemeral" };
344
+ if (typeof msg.content === "string") {
345
+ if (!msg.content)
346
+ return; // don't synthesize an empty (API-rejected) text block
347
+ msg.content = [{ type: "text", text: msg.content, cache_control }];
348
+ return;
349
+ }
350
+ if (Array.isArray(msg.content) && msg.content.length) {
351
+ const last = msg.content[msg.content.length - 1];
352
+ last.cache_control = cache_control;
353
+ }
354
+ }
236
355
  /**
237
356
  * Reconstruct Anthropic assistant content blocks from a neutral transcript when
238
357
  * no provider replay was persisted. Only meaningful for tool-use turns: a plain
@@ -16,12 +16,50 @@ export declare class CircuitBreaker {
16
16
  */
17
17
  export declare const INTERNAL_EXTENSION_KEYS: readonly string[];
18
18
  export declare function omitExtensionKeys(extensions: Record<string, unknown> | undefined, keys: readonly string[]): Record<string, unknown>;
19
+ /**
20
+ * Cached-prompt-token count from an OpenAI-compatible usage object. Covers the
21
+ * standard `prompt_tokens_details.cached_tokens` (OpenAI, Qwen, MiniMax, GLM,
22
+ * Kimi) and DeepSeek's `prompt_cache_hit_tokens`. These caches bill reads only,
23
+ * so there is no separate cache-creation count. The figure is a subset of
24
+ * `prompt_tokens` (the full prompt), surfaced for cost visibility — it must not
25
+ * be subtracted from the input count the kernel uses for context accounting.
26
+ */
27
+ export declare function openAICachedPromptTokens(usage: unknown): number;
28
+ /**
29
+ * Prompt-cache hit rate for one usage record: the fraction of the full prompt
30
+ * served from cache this request (`cacheReadInputTokens / inputTokens`, clamped to
31
+ * [0,1]). Returns 0 when the prompt size is unknown. This is the headline metric
32
+ * for the prefix-cache work (P0-A) — across a long, append-only session it should
33
+ * climb and stay high; a sustained drop means the cacheable prefix is drifting.
34
+ */
35
+ export declare function cacheHitRate(usage: {
36
+ inputTokens?: number;
37
+ cacheReadInputTokens?: number;
38
+ }): number;
39
+ /**
40
+ * Deterministic short key for OpenAI's `prompt_cache_key` — groups requests that
41
+ * share a cacheable prefix (same system prompt + tool set) onto the same cache
42
+ * routing, improving automatic prefix-cache hit rates without any caller input.
43
+ * FNV-1a over the parts; stable across processes, no crypto dependency.
44
+ */
45
+ export declare function stablePromptCacheKey(parts: string[]): string;
19
46
  export declare function normalizeToolCall(id: string, name: string, args: unknown): {
20
47
  id: string;
21
48
  name: string;
22
49
  arguments: string;
23
50
  } | null;
24
51
  export declare function toAnthropicContent(msg: Message): string | Array<Record<string, unknown>>;
52
+ /**
53
+ * History turns with the volatile State turn appended as the latest turn, for
54
+ * providers that render it inline (OpenAI-family, Gemini, Ollama). Appending
55
+ * (rather than prepending) keeps the history a byte-stable prefix so these
56
+ * providers' automatic prefix caches (OpenAI / Gemini implicit / Ollama KV) hit
57
+ * across turns — the volatile state is the uncached tail. Anthropic does the
58
+ * equivalent explicitly (append after the cache breakpoint — see
59
+ * AnthropicProvider.buildMessages). When `stateTurn` is absent (un-rebuilt
60
+ * binding) the State turn is still inside `turns`, so this returns `turns` as-is.
61
+ */
62
+ export declare function turnsWithStateAppended(context: RenderedContext): Message[];
25
63
  /** Convert RenderedContext.turns to Anthropic messages array.
26
64
  * `turns` contains only user / assistant / tool roles — no system filtering needed. */
27
65
  export declare function toAnthropicMessages(turns: Message[], nativeReplay?: (message: Message) => Array<Record<string, unknown>> | undefined): Array<Record<string, unknown>>;
Binary file
@@ -1,12 +1,12 @@
1
1
  import { AnthropicProvider } from "./anthropic.js";
2
2
  import { OpenAIChatProvider } from "./openai.js";
3
- import { DeepSeekProvider } from "./deepseek.js";
4
- import { KimiProvider } from "./kimi.js";
3
+ import { DeepSeekProvider, DeepSeekAnthropicProvider } from "./deepseek.js";
4
+ import { KimiProvider, KimiAnthropicProvider } from "./kimi.js";
5
5
  import { OpenAIResponsesProvider } from "./openai-responses.js";
6
6
  import { MiniMaxAnthropicProvider, MiniMaxOpenAIProvider } from "./minimax.js";
7
- import { QwenProvider } from "./qwen.js";
7
+ import { QwenProvider, QwenAnthropicProvider } from "./qwen.js";
8
8
  import { GeminiProvider } from "./gemini.js";
9
- import { GLMProvider } from "./glm.js";
9
+ import { GLMProvider, GLMAnthropicProvider } from "./glm.js";
10
10
  import { endpointProfiles, getModelProfile, modelProfiles } from "./profiles.js";
11
11
  export function createProvider(options) {
12
12
  const profile = isModelProfileId(options.model) ? getModelProfile(options.model) : undefined;
@@ -48,18 +48,30 @@ export function createProvider(options) {
48
48
  if (providerId === "minimax" && endpoint.protocol === "openai-chat") {
49
49
  return new MiniMaxOpenAIProvider(options.apiKey, model, options.retry, baseURL);
50
50
  }
51
+ if (providerId === "deepseek" && endpoint.protocol === "anthropic-messages") {
52
+ return new DeepSeekAnthropicProvider(options.apiKey, model, options.retry, baseURL);
53
+ }
51
54
  if (providerId === "deepseek" && endpoint.protocol === "openai-chat") {
52
55
  return new DeepSeekProvider(options.apiKey, model, options.retry, baseURL);
53
56
  }
57
+ if (providerId === "kimi" && endpoint.protocol === "anthropic-messages") {
58
+ return new KimiAnthropicProvider(options.apiKey, model, options.retry, baseURL);
59
+ }
54
60
  if (providerId === "kimi" && endpoint.protocol === "openai-chat") {
55
61
  return new KimiProvider(options.apiKey, model, options.retry, baseURL);
56
62
  }
63
+ if (providerId === "qwen" && endpoint.protocol === "anthropic-messages") {
64
+ return new QwenAnthropicProvider(options.apiKey, model, options.retry, baseURL);
65
+ }
57
66
  if (providerId === "qwen" && endpoint.protocol === "openai-chat") {
58
67
  return new QwenProvider(options.apiKey, model, options.retry, baseURL);
59
68
  }
60
69
  if (providerId === "gemini" && endpoint.protocol === "gemini") {
61
70
  return new GeminiProvider(options.apiKey, model, options.retry, baseURL);
62
71
  }
72
+ if (providerId === "glm" && endpoint.protocol === "anthropic-messages") {
73
+ return new GLMAnthropicProvider(options.apiKey, model, options.retry, baseURL);
74
+ }
63
75
  if (providerId === "glm" && endpoint.protocol === "openai-chat") {
64
76
  return new GLMProvider(options.apiKey, model, options.retry, baseURL);
65
77
  }
@@ -82,11 +94,11 @@ function defaultEndpointForProvider(providerId) {
82
94
  anthropic: "anthropic.messages",
83
95
  openai: "openai.chat",
84
96
  minimax: "minimax.anthropic",
85
- deepseek: "deepseek.openai",
86
- kimi: "kimi.openai",
87
- qwen: "qwen.dashscope",
97
+ deepseek: "deepseek.anthropic",
98
+ kimi: "kimi.anthropic",
99
+ qwen: "qwen.anthropic",
88
100
  gemini: "gemini.google",
89
- glm: "glm.openai",
101
+ glm: "glm.anthropic",
90
102
  };
91
103
  return defaults[providerId];
92
104
  }
@@ -1,5 +1,17 @@
1
1
  import type { Message, ProviderDescriptor, RenderedContext, ToolSchema, StreamEvent, RuntimePolicy } from "../types.js";
2
+ import { AnthropicProvider } from "./anthropic.js";
2
3
  import { OpenAIChatProvider } from "./openai.js";
4
+ /**
5
+ * DeepSeek over its Anthropic-compatible endpoint.
6
+ */
7
+ export declare class DeepSeekAnthropicProvider extends AnthropicProvider {
8
+ constructor(apiKey: string, model?: string, retry?: {
9
+ maxRetries: number;
10
+ baseDelay: number;
11
+ }, baseURL?: string);
12
+ protected providerName(): string;
13
+ runtimePolicy(): RuntimePolicy;
14
+ }
3
15
  export declare class DeepSeekProvider extends OpenAIChatProvider {
4
16
  constructor(apiKey: string, model?: string, retry?: {
5
17
  maxRetries: number;
@@ -1,15 +1,32 @@
1
+ import { AnthropicProvider } from "./anthropic.js";
1
2
  import { OpenAIChatProvider } from "./openai.js";
2
3
  import { endpointProfiles } from "./profiles.js";
3
- import { omitExtensionKeys } from "./base.js";
4
- const DEEPSEEK_BASE = endpointProfiles["deepseek.openai"].baseURL;
4
+ import { omitExtensionKeys, openAICachedPromptTokens } from "./base.js";
5
5
  const DEEPSEEK_POLICIES = {
6
6
  "deepseek-chat": { maxTurns: 25 },
7
7
  "deepseek-reasoner": { maxTurns: 50 },
8
8
  "deepseek-v4-flash": { maxTurns: 20 },
9
9
  "deepseek-v4-pro": { maxTurns: 35 },
10
10
  };
11
+ /**
12
+ * DeepSeek over its Anthropic-compatible endpoint.
13
+ */
14
+ export class DeepSeekAnthropicProvider extends AnthropicProvider {
15
+ constructor(apiKey, model = "deepseek-v4-flash", retry, baseURL = endpointProfiles["deepseek.anthropic"].baseURL) {
16
+ super(apiKey, model, retry, {
17
+ baseURL,
18
+ authMode: "api-key",
19
+ });
20
+ }
21
+ providerName() {
22
+ return "deepseek";
23
+ }
24
+ runtimePolicy() {
25
+ return DEEPSEEK_POLICIES[this.model] ?? {};
26
+ }
27
+ }
11
28
  export class DeepSeekProvider extends OpenAIChatProvider {
12
- constructor(apiKey, model = "deepseek-v4-flash", retry, baseURL = DEEPSEEK_BASE) {
29
+ constructor(apiKey, model = "deepseek-v4-flash", retry, baseURL = endpointProfiles["deepseek.openai"].baseURL) {
13
30
  super(apiKey, model, retry, baseURL);
14
31
  }
15
32
  runtimePolicy() {
@@ -103,11 +120,13 @@ export class DeepSeekProvider extends OpenAIChatProvider {
103
120
  let totalTokens = 0;
104
121
  let inputTokens = 0;
105
122
  let outputTokens = 0;
123
+ let cacheReadTokens = 0;
106
124
  for await (const chunk of stream) {
107
125
  if (chunk.usage) {
108
126
  totalTokens = chunk.usage.total_tokens;
109
127
  inputTokens = chunk.usage.prompt_tokens ?? 0;
110
128
  outputTokens = chunk.usage.completion_tokens ?? 0;
129
+ cacheReadTokens = openAICachedPromptTokens(chunk.usage);
111
130
  continue;
112
131
  }
113
132
  const choice = chunk.choices[0];
@@ -173,7 +192,7 @@ export class DeepSeekProvider extends OpenAIChatProvider {
173
192
  yield { type: "tool_call", id: tb.id, name: tb.name, arguments: args };
174
193
  }
175
194
  if (totalTokens > 0)
176
- yield { type: "usage", totalTokens, inputTokens, outputTokens };
195
+ yield { type: "usage", totalTokens, inputTokens, outputTokens, ...(cacheReadTokens > 0 ? { cacheReadInputTokens: cacheReadTokens } : {}) };
177
196
  }
178
197
  rememberDeepSeekReplay(content, toolCalls, reasoningContent, nativeToolCalls) {
179
198
  if (typeof reasoningContent !== "string" || !reasoningContent.trim())