@deepstrike/sdk 0.1.13 → 0.1.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +77 -51
  2. package/dist/collaboration/harness.js +6 -1
  3. package/dist/collaboration/modes/creator-verifier.d.ts +2 -2
  4. package/dist/collaboration/modes/creator-verifier.js +2 -2
  5. package/dist/collaboration/pool.d.ts +4 -35
  6. package/dist/collaboration/pool.js +18 -44
  7. package/dist/harness/harness.d.ts +9 -7
  8. package/dist/harness/harness.js +19 -22
  9. package/dist/index.d.ts +17 -5
  10. package/dist/index.js +14 -2
  11. package/dist/kernel.d.ts +1 -0
  12. package/dist/kernel.js +10 -1
  13. package/dist/memory/protocols.d.ts +0 -4
  14. package/dist/providers/anthropic.d.ts +2 -1
  15. package/dist/providers/anthropic.js +19 -1
  16. package/dist/providers/deepseek.d.ts +3 -2
  17. package/dist/providers/deepseek.js +9 -0
  18. package/dist/providers/gemini.d.ts +2 -1
  19. package/dist/providers/gemini.js +11 -0
  20. package/dist/providers/kimi.d.ts +3 -1
  21. package/dist/providers/kimi.js +10 -0
  22. package/dist/providers/minimax.d.ts +3 -1
  23. package/dist/providers/minimax.js +8 -0
  24. package/dist/providers/ollama.d.ts +2 -1
  25. package/dist/providers/ollama.js +22 -0
  26. package/dist/providers/openai-responses.d.ts +1 -0
  27. package/dist/providers/openai-responses.js +13 -0
  28. package/dist/providers/openai.d.ts +2 -1
  29. package/dist/providers/openai.js +17 -0
  30. package/dist/providers/profiles.d.ts +502 -6
  31. package/dist/providers/profiles.js +225 -82
  32. package/dist/providers/qwen.d.ts +2 -1
  33. package/dist/providers/qwen.js +15 -0
  34. package/dist/runtime/credential-vault.d.ts +18 -0
  35. package/dist/runtime/credential-vault.js +33 -0
  36. package/dist/runtime/execution-plane.d.ts +38 -0
  37. package/dist/runtime/execution-plane.js +146 -0
  38. package/dist/runtime/mcp-proxy-plane.d.ts +51 -0
  39. package/dist/runtime/mcp-proxy-plane.js +204 -0
  40. package/dist/runtime/process-sandbox-plane.d.ts +32 -0
  41. package/dist/runtime/process-sandbox-plane.js +109 -0
  42. package/dist/runtime/remote-vpc-plane.d.ts +48 -0
  43. package/dist/runtime/remote-vpc-plane.js +82 -0
  44. package/dist/runtime/runner.d.ts +49 -0
  45. package/dist/runtime/runner.js +364 -0
  46. package/dist/runtime/session-log.d.ts +67 -0
  47. package/dist/runtime/session-log.js +84 -0
  48. package/dist/types.d.ts +16 -0
  49. package/package.json +3 -2
  50. package/dist/agent.d.ts +0 -90
  51. package/dist/agent.js +0 -499
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # DeepStrike Node.js SDK
2
2
 
3
- Agent framework built on a Rust kernel. The kernel handles loop control, context compression, skill routing, governance, signal prioritization — the SDK handles all I/O.
3
+ Runtime framework built on a Rust kernel. The kernel handles loop control, context compression, skill routing, governance, signal prioritization — the SDK handles all I/O.
4
4
 
5
5
  ## Install
6
6
 
@@ -33,7 +33,14 @@ The correct platform package is selected and installed automatically via `option
33
33
  ## Quick start
34
34
 
35
35
  ```typescript
36
- import { Agent, OpenAIResponsesProvider, tool } from "@deepstrike/sdk"
36
+ import {
37
+ FileSessionLog,
38
+ LocalExecutionPlane,
39
+ RuntimeRunner,
40
+ OpenAIResponsesProvider,
41
+ collectText,
42
+ tool,
43
+ } from "@deepstrike/sdk"
37
44
 
38
45
  const provider = new OpenAIResponsesProvider(process.env.OPENAI_API_KEY!, "gpt-5-mini")
39
46
 
@@ -43,26 +50,34 @@ const add = tool("add", "Add two numbers.", {
43
50
  required: ["x", "y"],
44
51
  }, async ({ x, y }) => String(Number(x) + Number(y)))
45
52
 
46
- const agent = new Agent(provider, { maxTokens: 4096 })
47
- agent.register(add)
53
+ const plane = new LocalExecutionPlane().register(add)
54
+ const runner = new RuntimeRunner({
55
+ provider,
56
+ executionPlane: plane,
57
+ sessionLog: new FileSessionLog(".deepstrike/sessions"),
58
+ maxTokens: 4096,
59
+ })
48
60
 
49
- const result = await agent.run("What is 17 + 28?")
61
+ const result = await collectText(runner.run({
62
+ sessionId: "math-1",
63
+ goal: "What is 17 + 28?",
64
+ }))
50
65
  console.log(result)
51
66
  ```
52
67
 
53
68
  Same-session conversation continuity is explicit via `sessionId`:
54
69
 
55
70
  ```typescript
56
- await agent.run("My name is Ada.", undefined, undefined, "chat-1")
57
- const reply = await agent.run("What is my name?", undefined, undefined, "chat-1")
71
+ await collectText(runner.run({ sessionId: "chat-1", goal: "My name is Ada." }))
72
+ const reply = await collectText(runner.run({ sessionId: "chat-1", goal: "What is my name?" }))
58
73
  ```
59
74
 
60
- By default, the agent keeps session transcripts in memory for the lifetime of that `Agent` instance. Provide a `sessionStore` when the transcript must survive process restarts or be shared across workers.
75
+ Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when event replay should survive restarts. `wake(sessionId)` resumes from the event log without inserting a duplicate user start event.
61
76
 
62
77
  Streaming:
63
78
 
64
79
  ```typescript
65
- for await (const event of agent.runStreaming("Summarize README.md")) {
80
+ for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize README.md" })) {
66
81
  if (event.type === "text_delta") process.stdout.write(event.delta)
67
82
  else if (event.type === "tool_call") console.log(`\n[→ ${event.name}]`)
68
83
  else if (event.type === "tool_result") console.log(` = ${event.content}`)
@@ -103,10 +118,14 @@ const provider = createProvider({
103
118
 
104
119
  ---
105
120
 
106
- ## Agent options
121
+ ## Runtime options
107
122
 
108
123
  ```typescript
109
- const agent = new Agent(provider, {
124
+ const plane = new LocalExecutionPlane()
125
+ const runner = new RuntimeRunner({
126
+ provider,
127
+ executionPlane: plane,
128
+ sessionLog: new FileSessionLog(".deepstrike/sessions"),
110
129
  maxTokens: 4096, // context window size
111
130
  maxTurns: 25, // max turns (default 25)
112
131
  timeoutMs: 60_000, // timeout in ms
@@ -127,19 +146,25 @@ const agent = new Agent(provider, {
127
146
  ```typescript
128
147
  import { tool, readFile } from "@deepstrike/sdk"
129
148
 
130
- agent.register(tool("search", "Search.", schema, async (args) => ...))
131
- agent.register(readFile) // built-in: read files from disk
132
- agent.unregister("search")
149
+ plane.register(tool("search", "Search.", schema, async (args) => ...))
150
+ plane.register(readFile) // built-in: read files from disk
151
+ plane.unregister("search")
133
152
  ```
134
153
 
135
154
  ---
136
155
 
137
156
  ## Skills
138
157
 
139
- Skills are `.md` files with YAML frontmatter. Set `skillDir` on the agent — the kernel auto-injects a `skill` meta-tool, and the LLM loads skills by name on demand.
158
+ Skills are `.md` files with YAML frontmatter. Set `skillDir` on the runner — the kernel auto-injects a `skill` meta-tool, and the LLM loads skills by name on demand.
140
159
 
141
160
  ```typescript
142
- const agent = new Agent(provider, { maxTokens: 4096, skillDir: "./skills" })
161
+ const runner = new RuntimeRunner({
162
+ provider,
163
+ executionPlane: plane,
164
+ sessionLog: new FileSessionLog(".deepstrike/sessions"),
165
+ maxTokens: 4096,
166
+ skillDir: "./skills",
167
+ })
143
168
  ```
144
169
 
145
170
  ```markdown
@@ -160,7 +185,10 @@ effort: 1
160
185
  Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand.
161
186
 
162
187
  ```typescript
163
- const agent = new Agent(provider, {
188
+ const runner = new RuntimeRunner({
189
+ provider,
190
+ executionPlane: plane,
191
+ sessionLog: new FileSessionLog(".deepstrike/sessions"),
164
192
  maxTokens: 4096,
165
193
  knowledgeSource: {
166
194
  async retrieve(query: string, topK: number): Promise<string[]> {
@@ -196,7 +224,10 @@ class MyStore implements DreamStore {
196
224
  async search(agentId, query, topK) { ... }
197
225
  }
198
226
 
199
- const agent = new Agent(provider, {
227
+ const runner = new RuntimeRunner({
228
+ provider,
229
+ executionPlane: plane,
230
+ sessionLog: new FileSessionLog(".deepstrike/sessions"),
200
231
  maxTokens: 4096,
201
232
  dreamStore: new MyStore(),
202
233
  agentId: "my-agent", // enables `memory` meta-tool
@@ -204,23 +235,7 @@ const agent = new Agent(provider, {
204
235
 
205
236
  // In-session: LLM calls memory(query) → DreamStore.search()
206
237
  // Post-session: trigger memory consolidation
207
- const result = await agent.dream("my-agent", Date.now())
208
- ```
209
-
210
- ### SessionStore (same-session transcript continuity)
211
-
212
- ```typescript
213
- import type { SessionStore } from "@deepstrike/sdk"
214
-
215
- class MySessionStore implements SessionStore {
216
- async loadSession(sessionId) { return db.sessions.get(sessionId) }
217
- async saveSession(session) { await db.sessions.put(session.sessionId, session) }
218
- }
219
-
220
- const agent = new Agent(provider, {
221
- maxTokens: 4096,
222
- sessionStore: new MySessionStore(),
223
- })
238
+ const result = await runner.dream("my-agent", Date.now())
224
239
  ```
225
240
 
226
241
  ---
@@ -252,7 +267,13 @@ gov.requireParam("write_file", "path")
252
267
  gov.allowParamValues("set_mode", "mode", ["read", "write"])
253
268
  gov.limitParamRange("sleep", "seconds", 0, 10)
254
269
 
255
- const agent = new Agent(provider, { maxTokens: 4096, governance: gov })
270
+ const runner = new RuntimeRunner({
271
+ provider,
272
+ executionPlane: plane,
273
+ sessionLog: new FileSessionLog(".deepstrike/sessions"),
274
+ maxTokens: 4096,
275
+ governance: gov,
276
+ })
256
277
  // Every tool call goes through: Permission → Veto → RateLimit → Constraint → Audit
257
278
  ```
258
279
 
@@ -267,10 +288,16 @@ const gw = new SignalGateway()
267
288
  gw.schedule(new ScheduledPrompt("standup", Date.now() + 3600_000))
268
289
  gw.ingest({ kind: "interrupt", urgency: "critical", payload: {} })
269
290
 
270
- const agent = new Agent(provider, { maxTokens: 4096, signalSource: gw })
271
- // kind="interrupt" → immediately stops the running agent
291
+ const runner = new RuntimeRunner({
292
+ provider,
293
+ executionPlane: plane,
294
+ sessionLog: new FileSessionLog(".deepstrike/sessions"),
295
+ maxTokens: 4096,
296
+ signalSource: gw,
297
+ })
298
+ // kind="interrupt" → immediately stops the running runner
272
299
 
273
- agent.interrupt() // also works directly
300
+ runner.interrupt() // also works directly
274
301
  gw.destroy()
275
302
  ```
276
303
 
@@ -282,22 +309,21 @@ gw.destroy()
282
309
  import { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "@deepstrike/sdk"
283
310
 
284
311
  // 1. SinglePass — run once, always passes
285
- const outcome = await new SinglePassHarness(agent).run({ goal: "Say hello" })
312
+ const outcome = await new SinglePassHarness(runner).run({ goal: "Say hello" })
286
313
 
287
314
  // 2. EvalLoop — retry until QualityGate passes
288
- const harness = new EvalLoopHarness(agent, {
289
- gate: async (req, out) => out.result.includes("hello"),
290
- maxAttempts: 3,
291
- })
315
+ const harness = new EvalLoopHarness(runner, {
316
+ async evaluate(_req, out) { return out.result.includes("hello") },
317
+ }, 3)
292
318
 
293
319
  // 3. HarnessLoop — LLM-as-judge with feedback injection + skill extraction
294
- const loop = new HarnessLoop(agent, {
295
- evalProvider,
296
- maxAttempts: 3,
297
- skillDir: "./skills",
298
- })
299
- const out = await loop.run({ goal: "Write a haiku", criteria: ["Must be 3 lines"] })
300
- console.log(out.passed, out.feedback)
320
+ const loop = new HarnessLoop(runner, evalProvider, { maxAttempts: 3, skillDir: "./skills" })
321
+ for await (const event of loop.runStreaming({
322
+ goal: "Write a haiku",
323
+ criteria: [{ text: "Must be 3 lines", required: true }],
324
+ })) {
325
+ if (event.type === "done") console.log(event.verdict.passed, event.verdict.feedback)
326
+ }
301
327
  ```
302
328
 
303
329
  ---
@@ -1,3 +1,4 @@
1
+ import { collectText } from "../runtime/runner.js";
1
2
  import { formatContractForSystemPrompt, contractToCriteriaStrings } from "./contract.js";
2
3
  import { HandoffBus } from "./handoff.js";
3
4
  /**
@@ -42,7 +43,11 @@ export class ContractDrivenHarness {
42
43
  ? `\n\n[Previous attempt failed. Violations to fix:\n${this._formatViolationsForFeedback(checkResults)}]`
43
44
  : "";
44
45
  const executorGoal = `${contractBlock}\n\n---\n\n${currentGoal}${violationNote}`;
45
- artifact = await this.pool.get("executor").run(executorGoal, contractToCriteriaStrings(this.contract));
46
+ artifact = await collectText(this.pool.get("executor").run({
47
+ sessionId: crypto.randomUUID(),
48
+ goal: executorGoal,
49
+ criteria: contractToCriteriaStrings(this.contract),
50
+ }));
46
51
  // ── Phase 2: Verifier ──────────────────────────────────────────────────
47
52
  // Verifier sees: artifact + contract only. No executor history.
48
53
  const auditText = await this.pool.runVerifier({ contract: this.contract, artifact });
@@ -15,8 +15,8 @@ export interface CreatorVerifierMetrics {
15
15
  * Usage:
16
16
  * ```ts
17
17
  * const pool = new AgentPool()
18
- * .add("executor", new Agent(provider, { maxTokens: 32_000, skillDir }))
19
- * .add("verifier", new Agent(provider, { maxTokens: 8_000 }))
18
+ * .add("executor", executorRunner)
19
+ * .add("verifier", verifierRunner)
20
20
  *
21
21
  * const mode = new CreatorVerifierMode(pool)
22
22
  * const result = await mode.run(contract)
@@ -8,8 +8,8 @@ import { ContractDrivenHarness } from "../harness.js";
8
8
  * Usage:
9
9
  * ```ts
10
10
  * const pool = new AgentPool()
11
- * .add("executor", new Agent(provider, { maxTokens: 32_000, skillDir }))
12
- * .add("verifier", new Agent(provider, { maxTokens: 8_000 }))
11
+ * .add("executor", executorRunner)
12
+ * .add("verifier", verifierRunner)
13
13
  *
14
14
  * const mode = new CreatorVerifierMode(pool)
15
15
  * const result = await mode.run(contract)
@@ -1,46 +1,15 @@
1
- import type { Agent } from "../agent.js";
1
+ import type { RuntimeRunner } from "../runtime/runner.js";
2
2
  import type { VerificationContract } from "./contract.js";
3
- /**
4
- * Roles in a multi-agent collaboration.
5
- *
6
- * - orchestrator: strong-reasoning, produces VerificationContracts; no tools beyond planning
7
- * - executor: code/task execution, full tool access, sees goal + contract only
8
- * - verifier: adversarial auditor, no tools, low temperature, sees artifact + contract only
9
- */
10
3
  export type AgentRole = "orchestrator" | "executor" | "verifier";
11
- /** Context passed to a verifier run — intentionally minimal. */
12
4
  export interface IsolatedVerifierContext {
13
5
  contract: VerificationContract;
14
- /** The artifact produced by the executor. */
15
6
  artifact: string;
16
7
  }
17
- /**
18
- * AgentPool manages a set of role-specific Agent instances.
19
- *
20
- * Each role runs in its own Agent instance with an independent history partition,
21
- * ensuring that the verifier never sees the executor's implementation transcript.
22
- *
23
- * Usage:
24
- * ```ts
25
- * const pool = new AgentPool()
26
- * .add("executor", executorAgent)
27
- * .add("verifier", verifierAgent)
28
- * ```
29
- */
30
8
  export declare class AgentPool {
31
- private agents;
32
- add(role: AgentRole, agent: Agent): this;
9
+ private runners;
10
+ add(role: AgentRole, runner: RuntimeRunner): this;
33
11
  has(role: AgentRole): boolean;
34
- get(role: AgentRole): Agent;
35
- /**
36
- * Run the verifier with an isolated context — only the artifact and contract.
37
- * The verifier does NOT receive the executor's conversation history.
38
- * Returns a structured audit response as a plain string.
39
- */
12
+ get(role: AgentRole): RuntimeRunner;
40
13
  runVerifier(ctx: IsolatedVerifierContext): Promise<string>;
41
- /**
42
- * Run the orchestrator to decompose a high-level goal into a VerificationContract.
43
- * The orchestrator receives the goal and must produce a structured contract in JSON.
44
- */
45
14
  runOrchestrator(goal: string): Promise<string>;
46
15
  }
@@ -1,64 +1,38 @@
1
+ import { collectText } from "../runtime/runner.js";
1
2
  import { formatContractForSystemPrompt } from "./contract.js";
2
- /**
3
- * AgentPool manages a set of role-specific Agent instances.
4
- *
5
- * Each role runs in its own Agent instance with an independent history partition,
6
- * ensuring that the verifier never sees the executor's implementation transcript.
7
- *
8
- * Usage:
9
- * ```ts
10
- * const pool = new AgentPool()
11
- * .add("executor", executorAgent)
12
- * .add("verifier", verifierAgent)
13
- * ```
14
- */
15
3
  export class AgentPool {
16
- agents = new Map();
17
- add(role, agent) {
18
- this.agents.set(role, agent);
4
+ runners = new Map();
5
+ add(role, runner) {
6
+ this.runners.set(role, runner);
19
7
  return this;
20
8
  }
21
9
  has(role) {
22
- return this.agents.has(role);
10
+ return this.runners.has(role);
23
11
  }
24
12
  get(role) {
25
- const agent = this.agents.get(role);
26
- if (!agent)
27
- throw new Error(`AgentPool: no agent registered for role "${role}"`);
28
- return agent;
13
+ const runner = this.runners.get(role);
14
+ if (!runner)
15
+ throw new Error(`AgentPool: no runner registered for role "${role}"`);
16
+ return runner;
29
17
  }
30
- /**
31
- * Run the verifier with an isolated context — only the artifact and contract.
32
- * The verifier does NOT receive the executor's conversation history.
33
- * Returns a structured audit response as a plain string.
34
- */
35
18
  async runVerifier(ctx) {
36
- const agent = this.get("verifier");
19
+ const runner = this.get("verifier");
37
20
  const contractBlock = formatContractForSystemPrompt(ctx.contract);
38
21
  const auditGoal = [
39
- contractBlock,
40
- "",
41
- "---",
42
- "",
43
- "## Artifact to Audit",
44
- "",
45
- ctx.artifact,
46
- "",
47
- "---",
48
- "",
22
+ contractBlock, "",
23
+ "---", "",
24
+ "## Artifact to Audit", "",
25
+ ctx.artifact, "",
26
+ "---", "",
49
27
  "Audit the artifact against every criterion in the contract above.",
50
28
  "For each criterion, state whether it PASSED or FAILED and cite specific evidence.",
51
29
  "List any anti-patterns you detected.",
52
30
  "Conclude with an overall PASS or FAIL verdict.",
53
31
  ].join("\n");
54
- return agent.run(auditGoal);
32
+ return collectText(runner.run({ sessionId: crypto.randomUUID(), goal: auditGoal }));
55
33
  }
56
- /**
57
- * Run the orchestrator to decompose a high-level goal into a VerificationContract.
58
- * The orchestrator receives the goal and must produce a structured contract in JSON.
59
- */
60
34
  async runOrchestrator(goal) {
61
- const agent = this.get("orchestrator");
35
+ const runner = this.get("orchestrator");
62
36
  const orchestratorGoal = [
63
37
  `You are a planning orchestrator. Decompose the following goal into a VerificationContract.`,
64
38
  ``,
@@ -75,6 +49,6 @@ export class AgentPool {
75
49
  ``,
76
50
  `Output ONLY the JSON object, no prose.`,
77
51
  ].join("\n");
78
- return agent.run(orchestratorGoal);
52
+ return collectText(runner.run({ sessionId: crypto.randomUUID(), goal: orchestratorGoal }));
79
53
  }
80
54
  }
@@ -1,4 +1,5 @@
1
- import type { Agent } from "../agent.js";
1
+ import type { RuntimeRunner } from "../runtime/runner.js";
2
+ import { collectText } from "../runtime/runner.js";
2
3
  export interface Criterion {
3
4
  text: string;
4
5
  required: boolean;
@@ -71,15 +72,15 @@ export interface QualityGate {
71
72
  evaluate(request: HarnessRequest, outcome: HarnessOutcome): Promise<boolean>;
72
73
  }
73
74
  export declare class SinglePassHarness {
74
- private agent;
75
- constructor(agent: Agent);
75
+ private runner;
76
+ constructor(runner: RuntimeRunner);
76
77
  run(request: HarnessRequest): Promise<HarnessOutcome>;
77
78
  }
78
79
  export declare class EvalLoopHarness {
79
- private agent;
80
+ private runner;
80
81
  private gate;
81
82
  private maxAttempts;
82
- constructor(agent: Agent, gate: QualityGate, maxAttempts?: number);
83
+ constructor(runner: RuntimeRunner, gate: QualityGate, maxAttempts?: number);
83
84
  run(request: HarnessRequest): Promise<HarnessOutcome>;
84
85
  }
85
86
  export interface HarnessLoopOptions {
@@ -87,10 +88,11 @@ export interface HarnessLoopOptions {
87
88
  skillDir?: string;
88
89
  }
89
90
  export declare class HarnessLoop {
90
- private agent;
91
+ private runner;
91
92
  private evalProvider;
92
93
  private maxAttempts;
93
94
  private skillDir?;
94
- constructor(agent: Agent, evalProvider: import("../types.js").LLMProvider, options?: HarnessLoopOptions);
95
+ constructor(runner: RuntimeRunner, evalProvider: import("../types.js").LLMProvider, options?: HarnessLoopOptions);
95
96
  runStreaming(request: HarnessRequest): AsyncIterable<HarnessEvent>;
96
97
  }
98
+ export { collectText };
@@ -1,10 +1,12 @@
1
+ import { collectText } from "../runtime/runner.js";
1
2
  import { writeFile } from "fs/promises";
2
3
  import path from "path";
3
4
  import { getKernel } from "../kernel.js";
4
- async function runOnce(agent, req) {
5
+ async function runOnce(runner, req) {
5
6
  let text = "";
6
7
  let done;
7
- for await (const evt of agent.runStreaming(req.goal, req.criteria?.map(c => c.text), req.extensions)) {
8
+ const sessionId = crypto.randomUUID();
9
+ for await (const evt of runner.run({ sessionId, goal: req.goal, criteria: req.criteria?.map(c => c.text), extensions: req.extensions })) {
8
10
  if (evt.type === "text_delta")
9
11
  text += evt.delta;
10
12
  else if (evt.type === "done")
@@ -13,27 +15,27 @@ async function runOnce(agent, req) {
13
15
  return { result: text, passed: false, iterations: done?.iterations ?? 0, totalTokens: done?.totalTokens ?? 0, status: done?.status ?? "error" };
14
16
  }
15
17
  export class SinglePassHarness {
16
- agent;
17
- constructor(agent) {
18
- this.agent = agent;
18
+ runner;
19
+ constructor(runner) {
20
+ this.runner = runner;
19
21
  }
20
22
  async run(request) {
21
- return { ...await runOnce(this.agent, request), passed: true };
23
+ return { ...await runOnce(this.runner, request), passed: true };
22
24
  }
23
25
  }
24
26
  export class EvalLoopHarness {
25
- agent;
27
+ runner;
26
28
  gate;
27
29
  maxAttempts;
28
- constructor(agent, gate, maxAttempts = 3) {
29
- this.agent = agent;
30
+ constructor(runner, gate, maxAttempts = 3) {
31
+ this.runner = runner;
30
32
  this.gate = gate;
31
33
  this.maxAttempts = maxAttempts;
32
34
  }
33
35
  async run(request) {
34
36
  let outcome = { result: "", passed: false, iterations: 0, totalTokens: 0, status: "error" };
35
37
  for (let i = 0; i < this.maxAttempts; i++) {
36
- outcome = await runOnce(this.agent, request);
38
+ outcome = await runOnce(this.runner, request);
37
39
  if (await this.gate.evaluate(request, outcome))
38
40
  return { ...outcome, passed: true };
39
41
  }
@@ -41,12 +43,12 @@ export class EvalLoopHarness {
41
43
  }
42
44
  }
43
45
  export class HarnessLoop {
44
- agent;
46
+ runner;
45
47
  evalProvider;
46
48
  maxAttempts;
47
49
  skillDir;
48
- constructor(agent, evalProvider, options = {}) {
49
- this.agent = agent;
50
+ constructor(runner, evalProvider, options = {}) {
51
+ this.runner = runner;
50
52
  this.evalProvider = evalProvider;
51
53
  this.maxAttempts = options.maxAttempts ?? 3;
52
54
  this.skillDir = options.skillDir;
@@ -55,22 +57,19 @@ export class HarnessLoop {
55
57
  const kernel = getKernel();
56
58
  const pipeline = new kernel.EvalPipeline({ extractSkillOnPass: true });
57
59
  const criteria = request.criteria ?? [];
58
- // history tracks the full conversation; agent sees only the current goal message
59
- // but we inject feedback as user messages to guide revision without discarding prior output
60
60
  let currentGoal = request.goal;
61
61
  let lastIterations = 0;
62
62
  let lastTotalTokens = 0;
63
63
  let lastStatus = "error";
64
64
  let lastResult = "";
65
65
  for (let attempt = 1; attempt <= this.maxAttempts; attempt++) {
66
- // Stream agent output directly to caller
67
- for await (const evt of this.agent.runStreaming(currentGoal, criteria.map(c => c.text), request.extensions)) {
66
+ const sessionId = crypto.randomUUID();
67
+ for await (const evt of this.runner.run({ sessionId, goal: currentGoal, criteria: criteria.map(c => c.text), extensions: request.extensions })) {
68
68
  if (evt.type === "text_delta") {
69
69
  lastResult += evt.delta;
70
70
  yield { type: "token", text: evt.delta };
71
71
  }
72
72
  else if (evt.type === "tool_call") {
73
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
74
73
  const tc = evt;
75
74
  yield { type: "tool_call", id: tc.id, name: tc.name };
76
75
  }
@@ -83,7 +82,6 @@ export class HarnessLoop {
83
82
  yield { type: "tool_suspend", callId: ts.callId, suspensionId: ts.suspensionId, ...(ts.payload ? { payload: ts.payload } : {}) };
84
83
  }
85
84
  else if (evt.type === "tool_result") {
86
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
87
85
  const tr = evt;
88
86
  yield { type: "tool_result", callId: tr.callId, content: tr.content, isError: tr.isError };
89
87
  }
@@ -94,13 +92,11 @@ export class HarnessLoop {
94
92
  lastStatus = d.status;
95
93
  }
96
94
  }
97
- // Checkpoint: evaluator judges the completed output
98
95
  yield { type: "supervising" };
99
96
  const evalAction = pipeline.feedOutcome(request.goal, criteria, lastResult, attempt);
100
97
  if (evalAction.kind !== "evaluate")
101
98
  break;
102
99
  let evalText = "";
103
- // Wrap harness eval messages in a RenderedContext (system messages → systemText, rest → turns).
104
100
  const evalMsgs = evalAction.messages ?? [];
105
101
  const evalContext = {
106
102
  systemText: evalMsgs.filter((m) => m.role === "system").map((m) => m.content).join("\n\n"),
@@ -131,7 +127,6 @@ export class HarnessLoop {
131
127
  return;
132
128
  }
133
129
  yield { type: "revising", verdict };
134
- // Inject feedback as next turn's goal — agent sees its prior output + evaluator notes
135
130
  currentGoal = `${request.goal}\n\n[Attempt ${attempt} feedback: ${verdict.feedback}]`;
136
131
  lastResult = "";
137
132
  pipeline.reset();
@@ -139,3 +134,5 @@ export class HarnessLoop {
139
134
  yield { type: "max_attempts_reached" };
140
135
  }
141
136
  }
137
+ // Re-export collectText so harness callers can use it without knowing runner internals.
138
+ export { collectText };
package/dist/index.d.ts CHANGED
@@ -1,5 +1,17 @@
1
- export { Agent } from "./agent.js";
2
- export type { AgentOptions } from "./agent.js";
1
+ export { RuntimeRunner, collectText } from "./runtime/runner.js";
2
+ export type { RuntimeOptions } from "./runtime/runner.js";
3
+ export { LocalExecutionPlane } from "./runtime/execution-plane.js";
4
+ export type { ExecutionPlane, RunContext } from "./runtime/execution-plane.js";
5
+ export { InMemorySessionLog, FileSessionLog } from "./runtime/session-log.js";
6
+ export type { SessionLog, SessionEvent } from "./runtime/session-log.js";
7
+ export { EnvCredentialVault, InMemoryCredentialVault, ChainedCredentialVault } from "./runtime/credential-vault.js";
8
+ export type { CredentialVault } from "./runtime/credential-vault.js";
9
+ export { ProcessSandboxPlane } from "./runtime/process-sandbox-plane.js";
10
+ export type { SandboxOptions } from "./runtime/process-sandbox-plane.js";
11
+ export { McpProxyPlane } from "./runtime/mcp-proxy-plane.js";
12
+ export type { McpServerConfig } from "./runtime/mcp-proxy-plane.js";
13
+ export { RemoteVpcPlane } from "./runtime/remote-vpc-plane.js";
14
+ export type { RemoteVpcOptions } from "./runtime/remote-vpc-plane.js";
3
15
  export { AnthropicProvider } from "./providers/anthropic.js";
4
16
  export { OpenAIChatProvider, OpenAIProvider } from "./providers/openai.js";
5
17
  export { DeepSeekProvider } from "./providers/deepseek.js";
@@ -20,10 +32,8 @@ export type { RegisteredTool } from "./tools/index.js";
20
32
  export { scanSkillDir, readSkillFile } from "./skills/loader.js";
21
33
  export type { SkillMetadata } from "./skills/loader.js";
22
34
  export { WorkingMemory } from "./memory/working.js";
23
- export type { DreamStore, DreamResult, SessionStore, SessionData, SessionMessage, MemoryEntry, CurationResult, CurationStats, } from "./memory/protocols.js";
35
+ export type { DreamStore, DreamResult, SessionData, SessionMessage, MemoryEntry, CurationResult, CurationStats, } from "./memory/protocols.js";
24
36
  export type { KnowledgeSource } from "./knowledge/source.js";
25
- export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
26
- export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions, QualityGate } from "./harness/harness.js";
27
37
  export { ScheduledPrompt } from "./signals/scheduled.js";
28
38
  export { SignalGateway } from "./signals/gateway.js";
29
39
  export type { RuntimeSignal, SignalSource } from "./signals/types.js";
@@ -31,6 +41,8 @@ export { PermissionManager, PermissionMode } from "./safety/permissions.js";
31
41
  export type { PermissionDecision, Permission } from "./safety/permissions.js";
32
42
  export { Governance } from "./governance.js";
33
43
  export type { GovernanceVerdict } from "./governance.js";
44
+ export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
45
+ export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions, QualityGate } from "./harness/harness.js";
34
46
  export type { Message, ToolCall, ToolResult, ToolSchema, ContentPart, TextPart, ImagePart, AudioPart, StreamEvent, TextDelta, ThinkingDelta, ToolCallEvent, ToolChunk, ToolDeltaEvent, ToolSuspendEvent, ToolResultEvent, DoneEvent, ErrorEvent, PermissionRequestEvent, LLMProvider, RetryConfig, TokenUsage, ProviderToolSpec, ProviderRunState, RenderedContext, } from "./types.js";
35
47
  export type { AcceptanceCriterion, VerificationContract, ContractCheckResult, } from "./collaboration/contract.js";
36
48
  export { ContractBuilder, formatContractForSystemPrompt, contractToCriteriaStrings, } from "./collaboration/contract.js";
package/dist/index.js CHANGED
@@ -1,4 +1,12 @@
1
- export { Agent } from "./agent.js";
1
+ // ── Runtime (Layer 1.5) ────────────────────────────────────────────────────
2
+ export { RuntimeRunner, collectText } from "./runtime/runner.js";
3
+ export { LocalExecutionPlane } from "./runtime/execution-plane.js";
4
+ export { InMemorySessionLog, FileSessionLog } from "./runtime/session-log.js";
5
+ export { EnvCredentialVault, InMemoryCredentialVault, ChainedCredentialVault } from "./runtime/credential-vault.js";
6
+ export { ProcessSandboxPlane } from "./runtime/process-sandbox-plane.js";
7
+ export { McpProxyPlane } from "./runtime/mcp-proxy-plane.js";
8
+ export { RemoteVpcPlane } from "./runtime/remote-vpc-plane.js";
9
+ // ── Providers ─────────────────────────────────────────────────────────────
2
10
  export { AnthropicProvider } from "./providers/anthropic.js";
3
11
  export { OpenAIChatProvider, OpenAIProvider } from "./providers/openai.js";
4
12
  export { DeepSeekProvider } from "./providers/deepseek.js";
@@ -12,14 +20,18 @@ export { OpenAIChatAdapter } from "./providers/openai-chat.js";
12
20
  export { OpenAIResponsesAdapter, OpenAIResponsesProvider } from "./providers/openai-responses.js";
13
21
  export { endpointProfiles, modelProfiles, getModelProfile } from "./providers/profiles.js";
14
22
  export { createProvider } from "./providers/catalog.js";
23
+ // ── Tools & Skills ─────────────────────────────────────────────────────────
15
24
  export { tool, streamingTool, executeTools, readFile, validateToolArguments } from "./tools/index.js";
16
25
  export { scanSkillDir, readSkillFile } from "./skills/loader.js";
26
+ // ── Memory ─────────────────────────────────────────────────────────────────
17
27
  export { WorkingMemory } from "./memory/working.js";
18
- export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
19
28
  export { ScheduledPrompt } from "./signals/scheduled.js";
20
29
  export { SignalGateway } from "./signals/gateway.js";
30
+ // ── Safety & Governance ────────────────────────────────────────────────────
21
31
  export { PermissionManager, PermissionMode } from "./safety/permissions.js";
22
32
  export { Governance } from "./governance.js";
33
+ // ── Harness ────────────────────────────────────────────────────────────────
34
+ export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
23
35
  export { ContractBuilder, formatContractForSystemPrompt, contractToCriteriaStrings, } from "./collaboration/contract.js";
24
36
  export { AgentPool } from "./collaboration/pool.js";
25
37
  export { ContractDrivenHarness } from "./collaboration/harness.js";