@deepstrike/sdk 0.2.2 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -118,6 +118,33 @@ const provider = createProvider({
118
118
 
119
119
  ---
120
120
 
121
+ ## Context model (four slots)
122
+
123
+ The kernel renders context as four LLM API slots — only **history** is compressed.
124
+
125
+ | Slot | Source | Role |
126
+ |------|--------|------|
127
+ | `systemStable` | system partition | Identity, rules — never changes within a run |
128
+ | `systemKnowledge` | knowledge partition | Preloaded memory, skill defs — low frequency |
129
+ | `turns[0]` | `task_state` + signals | Goal, plan, progress, compression log, runtime signals |
130
+ | `turns[1..N]` | history | Conversation transcript |
131
+
132
+ ```typescript
133
+ const runner = new RuntimeRunner({
134
+ // ...
135
+ initialMemory: ["User prefers chartreuse."], // → Slot 2 (systemKnowledge)
136
+ systemPrompt: "You are a helpful assistant.", // → Slot 1 (systemStable)
137
+ })
138
+ ```
139
+
140
+ - `memory(query)` / `knowledge(query)` meta-tool results → **history** (tool results)
141
+ - External signals → **Slot 3** via `push_signal()`, cleared after each render
142
+ - Anthropic: Slots 1–2 get separate `cache_control` breakpoints
143
+
144
+ Full reference: [docs/context-partition-compression.md](../docs/context-partition-compression.md)
145
+
146
+ ---
147
+
121
148
  ## Runtime options
122
149
 
123
150
  ```typescript
@@ -135,6 +162,11 @@ const runner = new RuntimeRunner({
135
162
  signalSource: rx, // SignalSource for external signals
136
163
  dreamStore: myStore, // DreamStore for long-term memory
137
164
  agentId: "my-agent", // required with dreamStore for memory meta-tool
165
+ initialMemory: ["..."], // preloaded blocks → Slot 2 (systemKnowledge)
166
+ subAgentHarness: { // optional: sub-agents run through HarnessLoop
167
+ evalProvider,
168
+ maxAttempts: 3,
169
+ },
138
170
  governance: gov, // Governance pipeline instance
139
171
  })
140
172
  ```
@@ -182,7 +214,9 @@ effort: 1
182
214
 
183
215
  ## Knowledge
184
216
 
185
- Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand.
217
+ Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand. **Runtime retrieval results land in history** as tool results.
218
+
219
+ To inject durable knowledge at startup (Slot 2, cacheable on Anthropic), use `initialMemory` or kernel `add_knowledge_message`.
186
220
 
187
221
  ```typescript
188
222
  const runner = new RuntimeRunner({
@@ -202,7 +236,9 @@ const runner = new RuntimeRunner({
202
236
 
203
237
  ## Memory
204
238
 
205
- ### WorkingMemory (in-session scratch pad)
239
+ ### WorkingMemory (SDK-side scratch pad)
240
+
241
+ `WorkingMemory` is an SDK helper — not the kernel `working` partition (removed). Kernel task state lives in `task_state` and renders into Slot 3 (`turns[0]`).
206
242
 
207
243
  ```typescript
208
244
  import { WorkingMemory } from "@deepstrike/sdk"
@@ -233,7 +269,8 @@ const runner = new RuntimeRunner({
233
269
  agentId: "my-agent", // enables `memory` meta-tool
234
270
  })
235
271
 
236
- // In-session: LLM calls memory(query) → DreamStore.search()
272
+ // In-session: LLM calls memory(query) → DreamStore.search() → history tool result
273
+ // Preload: initialMemory → Slot 2 (systemKnowledge)
237
274
  // Post-session: trigger memory consolidation
238
275
  const result = await runner.dream("my-agent", Date.now())
239
276
  ```
@@ -318,6 +355,14 @@ const harness = new EvalLoopHarness(runner, {
318
355
 
319
356
  // 3. HarnessLoop — LLM-as-judge with feedback injection + skill extraction
320
357
  const loop = new HarnessLoop(runner, evalProvider, { maxAttempts: 3, skillDir: "./skills" })
358
+
359
+ // Sub-agents: pass subAgentHarness on RuntimeRunner to auto-evaluate spawned children
360
+ const runnerWithHarness = new RuntimeRunner({
361
+ provider,
362
+ executionPlane: plane,
363
+ sessionLog,
364
+ subAgentHarness: { evalProvider, maxAttempts: 3 },
365
+ })
321
366
  for await (const event of loop.runStreaming({
322
367
  goal: "Write a haiku",
323
368
  criteria: [{ text: "Must be 3 lines", required: true }],
@@ -24,6 +24,7 @@ export declare class AnthropicProvider implements LLMProvider {
24
24
  private hasBetas;
25
25
  private createMessage;
26
26
  private streamMessage;
27
+ private buildSystem;
27
28
  private buildMessages;
28
29
  private rememberNativeBlocks;
29
30
  }
@@ -44,16 +44,17 @@ export class AnthropicProvider {
44
44
  }
45
45
  }
46
46
  buildTools(tools) {
47
- return tools.map(t => ({
47
+ return tools.map((t, i) => ({
48
48
  name: t.name,
49
49
  description: t.description,
50
50
  input_schema: JSON.parse(t.parameters),
51
+ ...(i === tools.length - 1 ? { cache_control: { type: "ephemeral" } } : {}),
51
52
  }));
52
53
  }
53
54
  async complete(context, tools, extensions) {
54
55
  if (this.circuit.isOpen())
55
56
  throw new Error("Circuit breaker open");
56
- const system = context.systemText || undefined;
57
+ const system = this.buildSystem(context);
57
58
  const msgs = this.buildMessages(context);
58
59
  const requestExtensions = this.requestExtensions(extensions);
59
60
  let lastErr;
@@ -79,7 +80,7 @@ export class AnthropicProvider {
79
80
  toolCalls.push(tc);
80
81
  }
81
82
  }
82
- const message = { role: "assistant", content, tokenCount: resp.usage.input_tokens + resp.usage.output_tokens, toolCalls };
83
+ const message = { role: "assistant", content, tokenCount: resp.usage.output_tokens, toolCalls };
83
84
  this.rememberNativeBlocks(message, resp.content);
84
85
  return message;
85
86
  }
@@ -93,7 +94,7 @@ export class AnthropicProvider {
93
94
  throw lastErr;
94
95
  }
95
96
  async *stream(context, tools, extensions) {
96
- const system = context.systemText || undefined;
97
+ const system = this.buildSystem(context);
97
98
  const msgs = this.buildMessages(context);
98
99
  const requestExtensions = this.requestExtensions(extensions);
99
100
  const toolBlocks = {};
@@ -113,8 +114,10 @@ export class AnthropicProvider {
113
114
  if (evt.type === "message_start" || evt.type === "message_delta") {
114
115
  const usage = evt.usage ?? evt.message?.usage;
115
116
  if (usage) {
116
- totalTokens = usage.input_tokens + (usage.output_tokens ?? 0);
117
- yield { type: "usage", totalTokens };
117
+ const inputTokens = usage.input_tokens ?? 0;
118
+ const outputTokens = usage.output_tokens ?? 0;
119
+ totalTokens = inputTokens + outputTokens;
120
+ yield { type: "usage", totalTokens, inputTokens, outputTokens };
118
121
  }
119
122
  }
120
123
  else if (evt.type === "content_block_start") {
@@ -175,8 +178,25 @@ export class AnthropicProvider {
175
178
  ? this.client.beta.messages.stream(params)
176
179
  : this.client.messages.stream(params));
177
180
  }
181
+ buildSystem(context) {
182
+ if (!context.systemStable && !context.systemKnowledge) {
183
+ return context.systemText || undefined;
184
+ }
185
+ const blocks = [];
186
+ if (context.systemStable) {
187
+ blocks.push({ type: "text", text: context.systemStable, cache_control: { type: "ephemeral" } });
188
+ }
189
+ if (context.systemKnowledge) {
190
+ blocks.push({ type: "text", text: context.systemKnowledge, cache_control: { type: "ephemeral" } });
191
+ }
192
+ return blocks.length ? blocks : undefined;
193
+ }
178
194
  buildMessages(context) {
179
- return toAnthropicMessages(context.turns, message => this.nativeAssistantBlocks.get(assistantReplayKey(message)));
195
+ const msgs = toAnthropicMessages(context.turns, message => this.nativeAssistantBlocks.get(assistantReplayKey(message)));
196
+ if (msgs.length === 0) {
197
+ msgs.push({ role: "user", content: "Proceed." });
198
+ }
199
+ return msgs;
180
200
  }
181
201
  rememberNativeBlocks(message, blocks) {
182
202
  if (!blocks.length)
@@ -96,12 +96,14 @@ export function toAnthropicMessages(turns, nativeReplay) {
96
96
  if (msg.role === "assistant" && msg.toolCalls?.length) {
97
97
  const replay = nativeReplay?.(msg);
98
98
  if (replay) {
99
- result.push({ role: "assistant", content: replay });
99
+ result.push({ role: "assistant", content: ensureAssistantToolText(replay) });
100
100
  continue;
101
101
  }
102
102
  const blocks = [];
103
103
  if (msg.content)
104
104
  blocks.push({ type: "text", text: msg.content });
105
+ else
106
+ blocks.push({ type: "text", text: "Tool call requested." });
105
107
  blocks.push(...msg.toolCalls.map(tc => ({
106
108
  type: "tool_use",
107
109
  id: tc.id,
@@ -115,6 +117,15 @@ export function toAnthropicMessages(turns, nativeReplay) {
115
117
  }
116
118
  return result;
117
119
  }
120
+ function ensureAssistantToolText(blocks) {
121
+ if (!blocks.some(b => b.type === "tool_use"))
122
+ return blocks;
123
+ if (blocks.some(b => b.type === "text" && String(b.text ?? "").trim()))
124
+ return blocks;
125
+ if (blocks.some(b => b.type === "thinking"))
126
+ return blocks;
127
+ return [{ type: "text", text: "Tool call requested." }, ...blocks];
128
+ }
118
129
  // ─── OpenAI-compatible message conversion ────────────────────────────────────
119
130
  export function toOpenAIContent(msg) {
120
131
  if (!msg.contentParts?.length)
@@ -35,17 +35,27 @@ export class DeepSeekProvider extends OpenAIChatProvider {
35
35
  let finalText = "";
36
36
  const stream = await this.client.chat.completions.create({
37
37
  ...omitExtensionKeys(extensions, [
38
- "model", "messages", "tools", "stream", "extra_body", "reasoning_effort",
38
+ "model", "messages", "tools", "stream", "stream_options", "extra_body", "reasoning_effort",
39
39
  "exposeReasoning", "thinking", "reasoningEffort",
40
40
  ]),
41
41
  model: this.model,
42
42
  messages: msgs,
43
43
  ...(tools.length ? { tools: this.chat.buildTools(tools) } : {}),
44
44
  stream: true,
45
+ stream_options: { include_usage: true },
45
46
  reasoning_effort: reasoningEffort,
46
47
  extra_body: { thinking: { type: thinking } },
47
48
  });
49
+ let totalTokens = 0;
50
+ let inputTokens = 0;
51
+ let outputTokens = 0;
48
52
  for await (const chunk of stream) {
53
+ if (chunk.usage) {
54
+ totalTokens = chunk.usage.total_tokens;
55
+ inputTokens = chunk.usage.prompt_tokens ?? 0;
56
+ outputTokens = chunk.usage.completion_tokens ?? 0;
57
+ continue;
58
+ }
49
59
  const choice = chunk.choices[0];
50
60
  if (!choice)
51
61
  continue;
@@ -110,5 +120,7 @@ export class DeepSeekProvider extends OpenAIChatProvider {
110
120
  emittedToolCallIndexes.add(idx);
111
121
  yield { type: "tool_call", id: tb.id, name: tb.name, arguments: args };
112
122
  }
123
+ if (totalTokens > 0)
124
+ yield { type: "usage", totalTokens, inputTokens, outputTokens };
113
125
  }
114
126
  }
@@ -125,7 +125,7 @@ export class GeminiProvider {
125
125
  return {
126
126
  role: "assistant",
127
127
  content,
128
- tokenCount: usage?.totalTokenCount,
128
+ tokenCount: usage?.candidatesTokenCount ?? usage?.totalTokenCount,
129
129
  toolCalls,
130
130
  };
131
131
  }
@@ -164,8 +164,14 @@ export class GeminiProvider {
164
164
  yield { type: "tool_call", id: tc.id, name: tc.name, arguments: tc.args };
165
165
  }
166
166
  const usage = (await result.response).usageMetadata;
167
- if (usage?.totalTokenCount)
168
- yield { type: "usage", totalTokens: usage.totalTokenCount };
167
+ if (usage?.totalTokenCount) {
168
+ yield {
169
+ type: "usage",
170
+ totalTokens: usage.totalTokenCount,
171
+ inputTokens: usage.promptTokenCount ?? 0,
172
+ outputTokens: usage.candidatesTokenCount ?? 0,
173
+ };
174
+ }
169
175
  }
170
176
  modelExtensions(extensions) {
171
177
  if (!extensions)
@@ -11,13 +11,7 @@ export declare class OpenAIChatAdapter {
11
11
  };
12
12
  }[];
13
13
  buildMessages(context: RenderedContext): OpenAI.ChatCompletionMessageParam[];
14
- normalizeToolCalls(toolCalls?: Array<{
15
- id: string;
16
- function: {
17
- name: string;
18
- arguments: string;
19
- };
20
- }>): Array<{
14
+ normalizeToolCalls(toolCalls?: OpenAI.ChatCompletionMessageToolCall[]): Array<{
21
15
  id: string;
22
16
  name: string;
23
17
  arguments: string;
@@ -29,6 +29,7 @@ export class OpenAIChatAdapter {
29
29
  }
30
30
  normalizeToolCalls(toolCalls = []) {
31
31
  return toolCalls
32
+ .filter((tc) => tc.type === "function")
32
33
  .map(tc => normalizeToolCall(tc.id, tc.function.name, tc.function.arguments))
33
34
  .filter(Boolean);
34
35
  }
@@ -158,7 +158,7 @@ export class OpenAIResponsesProvider {
158
158
  role: "assistant",
159
159
  content: decoded.content,
160
160
  toolCalls: decoded.toolCalls,
161
- tokenCount: resp.usage?.total_tokens,
161
+ tokenCount: resp.usage?.output_tokens ?? resp.usage?.total_tokens,
162
162
  };
163
163
  }
164
164
  catch (err) {
@@ -223,7 +223,12 @@ export class OpenAIResponsesProvider {
223
223
  runState.previousResponseId = evt.response.id;
224
224
  runState.coveredMessageCount = context.turns.length + 1;
225
225
  if (evt.response.usage?.total_tokens) {
226
- yield { type: "usage", totalTokens: evt.response.usage.total_tokens };
226
+ yield {
227
+ type: "usage",
228
+ totalTokens: evt.response.usage.total_tokens,
229
+ ...(evt.response.usage.input_tokens ? { inputTokens: evt.response.usage.input_tokens } : {}),
230
+ ...(evt.response.usage.output_tokens ? { outputTokens: evt.response.usage.output_tokens } : {}),
231
+ };
227
232
  }
228
233
  }
229
234
  }
@@ -69,7 +69,7 @@ export class OpenAIChatProvider {
69
69
  this.circuit.recordSuccess();
70
70
  const choice = resp.choices[0].message;
71
71
  const toolCalls = this.chat.normalizeToolCalls(choice.tool_calls ?? []);
72
- return { role: "assistant", content: choice.content ?? "", tokenCount: resp.usage?.total_tokens, toolCalls };
72
+ return { role: "assistant", content: choice.content ?? "", tokenCount: resp.usage?.completion_tokens ?? resp.usage?.total_tokens, toolCalls };
73
73
  }
74
74
  catch (err) {
75
75
  lastErr = err;
@@ -96,9 +96,13 @@ export class OpenAIChatProvider {
96
96
  stream_options: { include_usage: true },
97
97
  });
98
98
  let totalTokens = 0;
99
+ let inputTokens = 0;
100
+ let outputTokens = 0;
99
101
  for await (const chunk of stream) {
100
102
  if (chunk.usage) {
101
103
  totalTokens = chunk.usage.total_tokens;
104
+ inputTokens = chunk.usage.prompt_tokens ?? 0;
105
+ outputTokens = chunk.usage.completion_tokens ?? 0;
102
106
  continue;
103
107
  }
104
108
  const choice = chunk.choices[0];
@@ -183,7 +187,7 @@ export class OpenAIChatProvider {
183
187
  yield { type: "tool_call", id: tb.id, name: tb.name, arguments: args };
184
188
  }
185
189
  if (totalTokens > 0)
186
- yield { type: "usage", totalTokens };
190
+ yield { type: "usage", totalTokens, inputTokens, outputTokens };
187
191
  }
188
192
  requestExtensions(extensions) {
189
193
  return omitExtensionKeys(extensions, ["model", "messages", "tools", "stream", "stream_options"]);
@@ -65,7 +65,7 @@ export class QwenProvider {
65
65
  this.circuit.recordSuccess();
66
66
  const choice = resp.choices[0].message;
67
67
  const toolCalls = this.chat.normalizeToolCalls(choice.tool_calls ?? []);
68
- return { role: "assistant", content: choice.content ?? "", tokenCount: resp.usage?.total_tokens, toolCalls };
68
+ return { role: "assistant", content: choice.content ?? "", tokenCount: resp.usage?.completion_tokens ?? resp.usage?.total_tokens, toolCalls };
69
69
  }
70
70
  catch (err) {
71
71
  lastErr = err;
@@ -93,9 +93,13 @@ export class QwenProvider {
93
93
  ...(extraBody ? { extra_body: extraBody } : {}),
94
94
  });
95
95
  let totalTokens = 0;
96
+ let inputTokens = 0;
97
+ let outputTokens = 0;
96
98
  for await (const chunk of stream) {
97
99
  if (chunk.usage) {
98
100
  totalTokens = chunk.usage.total_tokens;
101
+ inputTokens = chunk.usage.prompt_tokens ?? 0;
102
+ outputTokens = chunk.usage.completion_tokens ?? 0;
99
103
  continue;
100
104
  }
101
105
  const choice = chunk.choices[0];
@@ -162,7 +166,7 @@ export class QwenProvider {
162
166
  yield { type: "tool_call", id: tb.id, name: tb.name, arguments: args };
163
167
  }
164
168
  if (totalTokens > 0)
165
- yield { type: "usage", totalTokens };
169
+ yield { type: "usage", totalTokens, inputTokens, outputTokens };
166
170
  }
167
171
  thinkingExtraBody(extensions) {
168
172
  const enableThinking = Boolean(extensions?.enableThinking ?? extensions?.enable_thinking);
@@ -194,6 +194,8 @@ function kernelMessageToSdk(raw) {
194
194
  function renderedContextToSdk(raw) {
195
195
  return {
196
196
  systemText: String(raw.system_text ?? raw.systemText ?? ""),
197
+ systemStable: String(raw.system_stable ?? raw.systemStable ?? ""),
198
+ systemKnowledge: String(raw.system_knowledge ?? raw.systemKnowledge ?? ""),
197
199
  turns: (raw.turns ?? []).map(kernelMessageToSdk),
198
200
  };
199
201
  }
@@ -1,4 +1,4 @@
1
- import type { LLMProvider, Message, ToolSchema, StreamEvent, ToolSuspendEvent } from "../types.js";
1
+ import type { LLMProvider, Message, ToolSchema, StreamEvent, ToolSuspendEvent, AsyncSummarizer } from "../types.js";
2
2
  import type { DreamStore } from "../memory/protocols.js";
3
3
  import type { KnowledgeSource } from "../knowledge/source.js";
4
4
  import type { SignalSource } from "../signals/types.js";
@@ -48,8 +48,24 @@ export interface RuntimeOptions {
48
48
  milestoneContract?: MilestoneContract;
49
49
  /** Custom sub-agent host driver; defaults to SubAgentOrchestrator. */
50
50
  subAgentOrchestrator?: SubAgentOrchestrator;
51
+ /**
52
+ * When set, sub-agents run through a HarnessLoop with this config.
53
+ * The eval provider evaluates the sub-agent's output against the criteria
54
+ * from the AgentRunSpec, retrying up to maxAttempts times.
55
+ */
56
+ subAgentHarness?: {
57
+ evalProvider: LLMProvider;
58
+ maxAttempts?: number;
59
+ };
51
60
  /** Optional system prompt injected into the dream synthesis call. */
52
61
  dreamSystemPrompt?: string;
62
+ /**
63
+ * Optional async LLM summarizer. When provided, a background call is fired
64
+ * after each compression event to produce a richer semantic summary.
65
+ * The result is written back to SessionLog as `summary_upgraded` and used
66
+ * on the next wake() in place of the rule-based summary.
67
+ */
68
+ asyncSummarizer?: AsyncSummarizer;
53
69
  }
54
70
  export declare class RuntimeRunner {
55
71
  private readonly opts;
@@ -69,8 +85,8 @@ export declare class RuntimeRunner {
69
85
  mountMarker(kind: string, id: string, description: string): void;
70
86
  /** Unmount a capability by kind + id from the active run. No-op if not running. */
71
87
  unmountCapability(kind: string, id: string): void;
72
- /** Push a large artifact into the kernel artifacts partition (not inlined in history). */
73
- pushArtifact(message: Message, tokens?: number): void;
88
+ /** Push content into the Knowledge slot (memory retrievals, skill definitions, artifacts). */
89
+ pushKnowledge(message: Message, tokens?: number): void;
74
90
  /**
75
91
  * Spawn an isolated sub-agent via the kernel, run it on the host, and feed the result back.
76
92
  * Requires an active parent run (`run()` / `wake()` in progress or paused at milestone).
@@ -92,6 +108,7 @@ export declare class RuntimeRunner {
92
108
  dream(agentId: string, nowMs?: number): AsyncIterable<StreamEvent>;
93
109
  private execute;
94
110
  private appendObservations;
111
+ private upgradeCompressedSummary;
95
112
  }
96
113
  export declare function replayMessages(events: Array<{
97
114
  seq: number;
@@ -43,14 +43,14 @@ export class RuntimeRunner {
43
43
  return;
44
44
  kernelApply(this.activeKernel, this.pendingObservations, capabilityCommandUnmount(kind, id));
45
45
  }
46
- /** Push a large artifact into the kernel artifacts partition (not inlined in history). */
47
- pushArtifact(message, tokens) {
46
+ /** Push content into the Knowledge slot (memory retrievals, skill definitions, artifacts). */
47
+ pushKnowledge(message, tokens) {
48
48
  if (!this.activeKernel)
49
49
  return;
50
50
  kernelApply(this.activeKernel, this.pendingObservations, {
51
- kind: "push_artifact",
52
- message: messageToKernelMessage(message),
53
- ...(tokens !== undefined ? { tokens } : {}),
51
+ kind: "add_knowledge_message",
52
+ content: message.content ?? "",
53
+ tokens: tokens ?? Math.max(1, Math.ceil((message.content?.length ?? 0) / 4)),
54
54
  });
55
55
  }
56
56
  /**
@@ -89,6 +89,7 @@ export class RuntimeRunner {
89
89
  spec,
90
90
  manifest,
91
91
  sessionLog: this.opts.sessionLog,
92
+ ...(this.opts.subAgentHarness ? { harness: this.opts.subAgentHarness } : {}),
92
93
  });
93
94
  kernelApply(runtime, this.pendingObservations, {
94
95
  kind: "sub_agent_completed",
@@ -238,7 +239,7 @@ export class RuntimeRunner {
238
239
  if (this.opts.initialMemory) {
239
240
  for (const mem of this.opts.initialMemory) {
240
241
  kernelApply(runtime, this.pendingObservations, {
241
- kind: "add_memory_message",
242
+ kind: "add_knowledge_message",
242
243
  content: mem,
243
244
  tokens: Math.max(1, Math.ceil(mem.length / 4)),
244
245
  });
@@ -346,11 +347,16 @@ export class RuntimeRunner {
346
347
  const context = action.context;
347
348
  const tools = action.tools;
348
349
  let turnTokens = 0;
350
+ let turnInputTokens = 0;
351
+ let turnOutputTokens = 0;
349
352
  let shouldRetry = false;
350
353
  try {
351
354
  for await (const evt of this.opts.provider.stream(context, tools, Object.keys(ext).length ? ext : undefined, providerState)) {
352
355
  if (evt.type === "usage") {
353
- turnTokens = evt.totalTokens;
356
+ const usageEvt = evt;
357
+ turnTokens = usageEvt.totalTokens;
358
+ turnInputTokens = usageEvt.inputTokens ?? 0;
359
+ turnOutputTokens = usageEvt.outputTokens ?? 0;
354
360
  continue;
355
361
  }
356
362
  yield evt;
@@ -390,17 +396,19 @@ export class RuntimeRunner {
390
396
  role: "assistant",
391
397
  content: finalText,
392
398
  toolCalls: finalToolCalls,
393
- tokenCount: turnTokens || undefined,
399
+ tokenCount: turnOutputTokens || turnTokens || undefined,
394
400
  };
395
401
  action = kernelAction(runtime, this.pendingObservations, {
396
402
  kind: "provider_result",
397
403
  message: messageToKernelMessage(assistantMessage),
404
+ ...(turnInputTokens > 0 ? { observed_input_tokens: turnInputTokens } : {}),
405
+ ...(turnOutputTokens > 0 ? { observed_output_tokens: turnOutputTokens } : {}),
398
406
  });
399
407
  const providerReplay = peekProviderReplay(this.opts.provider, finalText, finalToolCalls);
400
408
  await this.opts.sessionLog.append(sessionId, buildLlmCompletedEvent({
401
409
  turn: runtime.turn(),
402
410
  content: finalText,
403
- tokenCount: turnTokens || undefined,
411
+ tokenCount: turnOutputTokens || turnTokens || undefined,
404
412
  toolCalls: finalToolCalls,
405
413
  providerReplay,
406
414
  }));
@@ -616,6 +624,9 @@ export class RuntimeRunner {
616
624
  preserved_refs: preservedRefs,
617
625
  });
618
626
  nextArchiveStart = compressedSeq + 1;
627
+ if (this.opts.asyncSummarizer && archived && archived.length > 0) {
628
+ void this.upgradeCompressedSummary(sessionId, compressedSeq, archived, compressionAction(obs.action) ?? "auto_compact");
629
+ }
619
630
  }
620
631
  else if (obs.kind === "rollbacked") {
621
632
  await this.opts.sessionLog.append(sessionId, {
@@ -687,6 +698,19 @@ export class RuntimeRunner {
687
698
  }
688
699
  return nextArchiveStart;
689
700
  }
701
+ async upgradeCompressedSummary(sessionId, compressedSeq, archived, action) {
702
+ try {
703
+ const summary = await this.opts.asyncSummarizer.summarize(archived, action);
704
+ await this.opts.sessionLog.append(sessionId, {
705
+ kind: "summary_upgraded",
706
+ compressed_seq: compressedSeq,
707
+ summary,
708
+ });
709
+ }
710
+ catch {
711
+ // non-fatal: rule-based summary stays in place
712
+ }
713
+ }
690
714
  }
691
715
  function isMidRun(events) {
692
716
  return events.length > 0 && !events.some(e => e.event.kind === "run_terminal");
@@ -701,8 +725,14 @@ function compressionAction(action) {
701
725
  return undefined;
702
726
  }
703
727
  export function replayMessages(events, maxBytes) {
704
- const messages = [];
728
+ // Build upgraded-summary index: compressed_seq -> upgraded summary
729
+ const upgradedSummaries = new Map();
705
730
  for (const { event: e } of events) {
731
+ if (e.kind === "summary_upgraded")
732
+ upgradedSummaries.set(e.compressed_seq, e.summary);
733
+ }
734
+ const messages = [];
735
+ for (const { seq, event: e } of events) {
706
736
  if (e.kind === "run_started") {
707
737
  const userText = e.criteria.length
708
738
  ? `${e.goal}\n\nCriteria:\n${e.criteria.map((c, i) => `${i + 1}. ${c}`).join("\n")}`
@@ -715,8 +745,9 @@ export function replayMessages(events, maxBytes) {
715
745
  });
716
746
  }
717
747
  else if (e.kind === "compressed") {
718
- if (e.summary) {
719
- const systemText = `[Compressed context: turn ${e.turn}]\n${e.summary}`;
748
+ const summary = upgradedSummaries.get(seq) ?? e.summary;
749
+ if (summary) {
750
+ const systemText = `[Compressed context: turn ${e.turn}]\n${summary}`;
720
751
  messages.push({
721
752
  role: "system",
722
753
  content: systemText,
@@ -754,8 +785,14 @@ export function replayMessages(events, maxBytes) {
754
785
  return messages;
755
786
  }
756
787
  export async function replayMessagesAsync(events, maxBytes, loadArchive) {
757
- const messages = [];
788
+ // Build upgraded-summary index: compressed_seq -> upgraded summary
789
+ const upgradedSummaries = new Map();
758
790
  for (const { event: e } of events) {
791
+ if (e.kind === "summary_upgraded")
792
+ upgradedSummaries.set(e.compressed_seq, e.summary);
793
+ }
794
+ const messages = [];
795
+ for (const { seq, event: e } of events) {
759
796
  if (e.kind === "run_started") {
760
797
  const userText = e.criteria.length
761
798
  ? `${e.goal}\n\nCriteria:\n${e.criteria.map((c, i) => `${i + 1}. ${c}`).join("\n")}`
@@ -787,8 +824,9 @@ export async function replayMessagesAsync(events, maxBytes, loadArchive) {
787
824
  }
788
825
  }
789
826
  if (!loadedSuccessfully) {
790
- if (e.summary) {
791
- const systemText = `[Compressed context: turn ${e.turn}]\n${e.summary}`;
827
+ const summary = upgradedSummaries.get(seq) ?? e.summary;
828
+ if (summary) {
829
+ const systemText = `[Compressed context: turn ${e.turn}]\n${summary}`;
792
830
  messages.push({
793
831
  role: "system",
794
832
  content: systemText,
@@ -127,6 +127,10 @@ export type SessionEvent = {
127
127
  reason: string;
128
128
  turns_used: number;
129
129
  total_tokens: number;
130
+ } | {
131
+ kind: "summary_upgraded";
132
+ compressed_seq: number;
133
+ summary: string;
130
134
  };
131
135
  export interface SessionLog {
132
136
  append(sessionId: string, event: SessionEvent): Promise<number>;
@@ -8,6 +8,10 @@ export interface SubAgentRunContext {
8
8
  spec: AgentRunSpec;
9
9
  manifest: AgentSpawnedObservation;
10
10
  sessionLog: SessionLog;
11
+ harness?: {
12
+ evalProvider: import("../types.js").LLMProvider;
13
+ maxAttempts?: number;
14
+ };
11
15
  }
12
16
  /** Host-side driver for kernel-isolated sub-agent runs. */
13
17
  export declare class SubAgentOrchestrator {
@@ -46,6 +46,36 @@ export class SubAgentOrchestrator {
46
46
  });
47
47
  }
48
48
  async run(ctx) {
49
+ if (ctx.harness) {
50
+ const { RuntimeRunner } = await import("./runner.js");
51
+ const { HarnessLoop } = await import("../harness/harness.js");
52
+ const permitted = new Set(ctx.manifest.permitted_capability_ids ?? []);
53
+ const filteredPlane = new FilteredExecutionPlane(ctx.parentOpts.executionPlane, permitted);
54
+ const childRunner = new RuntimeRunner({
55
+ ...ctx.parentOpts,
56
+ executionPlane: filteredPlane,
57
+ agentId: ctx.spec.identity.agentId,
58
+ sessionLog: ctx.sessionLog,
59
+ });
60
+ const loop = new HarnessLoop(childRunner, ctx.harness.evalProvider, {
61
+ maxAttempts: ctx.harness.maxAttempts ?? 3,
62
+ });
63
+ const outcome = await loop.run({
64
+ goal: ctx.spec.goal,
65
+ criteria: (ctx.spec.milestones?.phases.flatMap(p => p.criteria) ?? [])
66
+ .filter((t) => typeof t === "string")
67
+ .map(text => ({ text, required: true })),
68
+ });
69
+ return {
70
+ agentId: ctx.spec.identity.agentId,
71
+ result: {
72
+ termination: outcome.passed ? "completed" : "error",
73
+ turnsUsed: outcome.iterations,
74
+ totalTokensUsed: outcome.totalTokens,
75
+ ...(outcome.result ? { finalMessage: { role: "assistant", content: outcome.result, toolCalls: [] } } : {}),
76
+ },
77
+ };
78
+ }
49
79
  let done;
50
80
  let finalText = "";
51
81
  for await (const evt of this.stream(ctx)) {
package/dist/types.d.ts CHANGED
@@ -180,12 +180,13 @@ export interface ProviderReplay {
180
180
  }
181
181
  /** Structured render output produced by the kernel for each LLM call. */
182
182
  export interface RenderedContext {
183
- /** Combined system text: system partition + dashboard (when non-empty).
184
- * Anthropic → `system` param · OpenAI → messages[0] system role ·
185
- * Gemini → `systemInstruction`. */
183
+ /** Identity + Knowledge combined — for providers with a single system slot (OpenAI). */
186
184
  systemText: string;
187
- /** Strictly alternating user / assistant / tool turns.
188
- * Working-partition signals are already folded into the first user turn. */
185
+ /** Identity only (system partition). Anthropic system[0] with cache_control. */
186
+ systemStable?: string;
187
+ /** Knowledge (memory retrievals, skill definitions, artifacts). Anthropic system[1] with cache_control. */
188
+ systemKnowledge?: string;
189
+ /** Turns: [0] = State (task_state + signals), [1..N] = History. */
189
190
  turns: Message[];
190
191
  }
191
192
  /**
@@ -213,6 +214,13 @@ export interface LLMProvider {
213
214
  complete(context: RenderedContext, tools: ToolSchema[], extensions?: Record<string, unknown>): Promise<Message>;
214
215
  stream(context: RenderedContext, tools: ToolSchema[], extensions?: Record<string, unknown>, state?: ProviderRunState): AsyncIterable<StreamEvent>;
215
216
  }
217
+ /**
218
+ * Optional async summarizer called after context compression.
219
+ * Produces a richer LLM-generated summary that replaces the rule-based one on next wake.
220
+ */
221
+ export interface AsyncSummarizer {
222
+ summarize(archived: Message[], action: string): Promise<string>;
223
+ }
216
224
  export interface TaskUpdate {
217
225
  plan?: string[];
218
226
  currentStep?: number;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@deepstrike/sdk",
3
- "version": "0.2.2",
3
+ "version": "0.2.4",
4
4
  "description": "DeepStrike Node.js SDK",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
@@ -12,14 +12,17 @@
12
12
  "scripts": {
13
13
  "build": "tsc",
14
14
  "test": "node --experimental-vm-modules node_modules/.bin/jest",
15
+ "test:e2e": "node --experimental-vm-modules node_modules/.bin/jest --testPathPattern e2e/run --testTimeout 300000",
16
+ "test:e2e:k": "node --experimental-vm-modules node_modules/.bin/jest --testPathPattern e2e/run --testTimeout 300000",
15
17
  "test:local-core": "cd ../crates/deepstrike-node && napi build --platform && cd ../../node && npm test",
16
- "smoke:providers": "node ../scripts/smoke-node-provider-loops.mjs"
18
+ "smoke:providers": "node ../scripts/smoke-node-provider-loops.mjs",
19
+ "e2e": "node ../scripts/run-e2e.mjs"
17
20
  },
18
21
  "dependencies": {
19
- "@anthropic-ai/sdk": "^0.39.0",
20
- "@deepstrike/core": "0.2.2",
22
+ "@anthropic-ai/sdk": "^0.99.0",
23
+ "@deepstrike/core": "0.2.4",
21
24
  "@google/generative-ai": "^0.24.1",
22
- "openai": "^4.77.0"
25
+ "openai": "^5.23.2"
23
26
  },
24
27
  "devDependencies": {
25
28
  "@types/jest": "^30.0.0",