@deepstrike/sdk 0.1.13 → 0.1.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +77 -51
- package/dist/collaboration/harness.js +6 -1
- package/dist/collaboration/modes/creator-verifier.d.ts +2 -2
- package/dist/collaboration/modes/creator-verifier.js +2 -2
- package/dist/collaboration/pool.d.ts +4 -35
- package/dist/collaboration/pool.js +18 -44
- package/dist/harness/harness.d.ts +9 -7
- package/dist/harness/harness.js +19 -22
- package/dist/index.d.ts +17 -5
- package/dist/index.js +14 -2
- package/dist/kernel.d.ts +1 -0
- package/dist/kernel.js +10 -1
- package/dist/memory/protocols.d.ts +0 -4
- package/dist/providers/anthropic.d.ts +2 -1
- package/dist/providers/anthropic.js +19 -1
- package/dist/providers/deepseek.d.ts +3 -2
- package/dist/providers/deepseek.js +9 -0
- package/dist/providers/gemini.d.ts +2 -1
- package/dist/providers/gemini.js +11 -0
- package/dist/providers/kimi.d.ts +3 -1
- package/dist/providers/kimi.js +10 -0
- package/dist/providers/minimax.d.ts +3 -1
- package/dist/providers/minimax.js +8 -0
- package/dist/providers/ollama.d.ts +2 -1
- package/dist/providers/ollama.js +22 -0
- package/dist/providers/openai-responses.d.ts +1 -0
- package/dist/providers/openai-responses.js +13 -0
- package/dist/providers/openai.d.ts +2 -1
- package/dist/providers/openai.js +17 -0
- package/dist/providers/profiles.d.ts +502 -6
- package/dist/providers/profiles.js +225 -82
- package/dist/providers/qwen.d.ts +2 -1
- package/dist/providers/qwen.js +15 -0
- package/dist/runtime/credential-vault.d.ts +18 -0
- package/dist/runtime/credential-vault.js +33 -0
- package/dist/runtime/execution-plane.d.ts +38 -0
- package/dist/runtime/execution-plane.js +146 -0
- package/dist/runtime/mcp-proxy-plane.d.ts +51 -0
- package/dist/runtime/mcp-proxy-plane.js +204 -0
- package/dist/runtime/process-sandbox-plane.d.ts +32 -0
- package/dist/runtime/process-sandbox-plane.js +109 -0
- package/dist/runtime/remote-vpc-plane.d.ts +48 -0
- package/dist/runtime/remote-vpc-plane.js +82 -0
- package/dist/runtime/runner.d.ts +49 -0
- package/dist/runtime/runner.js +364 -0
- package/dist/runtime/session-log.d.ts +67 -0
- package/dist/runtime/session-log.js +84 -0
- package/dist/types.d.ts +16 -0
- package/package.json +3 -2
- package/dist/agent.d.ts +0 -90
- package/dist/agent.js +0 -499
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# DeepStrike Node.js SDK
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Runtime framework built on a Rust kernel. The kernel handles loop control, context compression, skill routing, governance, signal prioritization — the SDK handles all I/O.
|
|
4
4
|
|
|
5
5
|
## Install
|
|
6
6
|
|
|
@@ -33,7 +33,14 @@ The correct platform package is selected and installed automatically via `option
|
|
|
33
33
|
## Quick start
|
|
34
34
|
|
|
35
35
|
```typescript
|
|
36
|
-
import {
|
|
36
|
+
import {
|
|
37
|
+
FileSessionLog,
|
|
38
|
+
LocalExecutionPlane,
|
|
39
|
+
RuntimeRunner,
|
|
40
|
+
OpenAIResponsesProvider,
|
|
41
|
+
collectText,
|
|
42
|
+
tool,
|
|
43
|
+
} from "@deepstrike/sdk"
|
|
37
44
|
|
|
38
45
|
const provider = new OpenAIResponsesProvider(process.env.OPENAI_API_KEY!, "gpt-5-mini")
|
|
39
46
|
|
|
@@ -43,26 +50,34 @@ const add = tool("add", "Add two numbers.", {
|
|
|
43
50
|
required: ["x", "y"],
|
|
44
51
|
}, async ({ x, y }) => String(Number(x) + Number(y)))
|
|
45
52
|
|
|
46
|
-
const
|
|
47
|
-
|
|
53
|
+
const plane = new LocalExecutionPlane().register(add)
|
|
54
|
+
const runner = new RuntimeRunner({
|
|
55
|
+
provider,
|
|
56
|
+
executionPlane: plane,
|
|
57
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
58
|
+
maxTokens: 4096,
|
|
59
|
+
})
|
|
48
60
|
|
|
49
|
-
const result = await
|
|
61
|
+
const result = await collectText(runner.run({
|
|
62
|
+
sessionId: "math-1",
|
|
63
|
+
goal: "What is 17 + 28?",
|
|
64
|
+
}))
|
|
50
65
|
console.log(result)
|
|
51
66
|
```
|
|
52
67
|
|
|
53
68
|
Same-session conversation continuity is explicit via `sessionId`:
|
|
54
69
|
|
|
55
70
|
```typescript
|
|
56
|
-
await
|
|
57
|
-
const reply = await
|
|
71
|
+
await collectText(runner.run({ sessionId: "chat-1", goal: "My name is Ada." }))
|
|
72
|
+
const reply = await collectText(runner.run({ sessionId: "chat-1", goal: "What is my name?" }))
|
|
58
73
|
```
|
|
59
74
|
|
|
60
|
-
|
|
75
|
+
Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when event replay should survive restarts. `wake(sessionId)` resumes from the event log without inserting a duplicate user start event.
|
|
61
76
|
|
|
62
77
|
Streaming:
|
|
63
78
|
|
|
64
79
|
```typescript
|
|
65
|
-
for await (const event of
|
|
80
|
+
for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize README.md" })) {
|
|
66
81
|
if (event.type === "text_delta") process.stdout.write(event.delta)
|
|
67
82
|
else if (event.type === "tool_call") console.log(`\n[→ ${event.name}]`)
|
|
68
83
|
else if (event.type === "tool_result") console.log(` = ${event.content}`)
|
|
@@ -103,10 +118,14 @@ const provider = createProvider({
|
|
|
103
118
|
|
|
104
119
|
---
|
|
105
120
|
|
|
106
|
-
##
|
|
121
|
+
## Runtime options
|
|
107
122
|
|
|
108
123
|
```typescript
|
|
109
|
-
const
|
|
124
|
+
const plane = new LocalExecutionPlane()
|
|
125
|
+
const runner = new RuntimeRunner({
|
|
126
|
+
provider,
|
|
127
|
+
executionPlane: plane,
|
|
128
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
110
129
|
maxTokens: 4096, // context window size
|
|
111
130
|
maxTurns: 25, // max turns (default 25)
|
|
112
131
|
timeoutMs: 60_000, // timeout in ms
|
|
@@ -127,19 +146,25 @@ const agent = new Agent(provider, {
|
|
|
127
146
|
```typescript
|
|
128
147
|
import { tool, readFile } from "@deepstrike/sdk"
|
|
129
148
|
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
149
|
+
plane.register(tool("search", "Search.", schema, async (args) => ...))
|
|
150
|
+
plane.register(readFile) // built-in: read files from disk
|
|
151
|
+
plane.unregister("search")
|
|
133
152
|
```
|
|
134
153
|
|
|
135
154
|
---
|
|
136
155
|
|
|
137
156
|
## Skills
|
|
138
157
|
|
|
139
|
-
Skills are `.md` files with YAML frontmatter. Set `skillDir` on the
|
|
158
|
+
Skills are `.md` files with YAML frontmatter. Set `skillDir` on the runner — the kernel auto-injects a `skill` meta-tool, and the LLM loads skills by name on demand.
|
|
140
159
|
|
|
141
160
|
```typescript
|
|
142
|
-
const
|
|
161
|
+
const runner = new RuntimeRunner({
|
|
162
|
+
provider,
|
|
163
|
+
executionPlane: plane,
|
|
164
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
165
|
+
maxTokens: 4096,
|
|
166
|
+
skillDir: "./skills",
|
|
167
|
+
})
|
|
143
168
|
```
|
|
144
169
|
|
|
145
170
|
```markdown
|
|
@@ -160,7 +185,10 @@ effort: 1
|
|
|
160
185
|
Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand.
|
|
161
186
|
|
|
162
187
|
```typescript
|
|
163
|
-
const
|
|
188
|
+
const runner = new RuntimeRunner({
|
|
189
|
+
provider,
|
|
190
|
+
executionPlane: plane,
|
|
191
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
164
192
|
maxTokens: 4096,
|
|
165
193
|
knowledgeSource: {
|
|
166
194
|
async retrieve(query: string, topK: number): Promise<string[]> {
|
|
@@ -196,7 +224,10 @@ class MyStore implements DreamStore {
|
|
|
196
224
|
async search(agentId, query, topK) { ... }
|
|
197
225
|
}
|
|
198
226
|
|
|
199
|
-
const
|
|
227
|
+
const runner = new RuntimeRunner({
|
|
228
|
+
provider,
|
|
229
|
+
executionPlane: plane,
|
|
230
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
200
231
|
maxTokens: 4096,
|
|
201
232
|
dreamStore: new MyStore(),
|
|
202
233
|
agentId: "my-agent", // enables `memory` meta-tool
|
|
@@ -204,23 +235,7 @@ const agent = new Agent(provider, {
|
|
|
204
235
|
|
|
205
236
|
// In-session: LLM calls memory(query) → DreamStore.search()
|
|
206
237
|
// Post-session: trigger memory consolidation
|
|
207
|
-
const result = await
|
|
208
|
-
```
|
|
209
|
-
|
|
210
|
-
### SessionStore (same-session transcript continuity)
|
|
211
|
-
|
|
212
|
-
```typescript
|
|
213
|
-
import type { SessionStore } from "@deepstrike/sdk"
|
|
214
|
-
|
|
215
|
-
class MySessionStore implements SessionStore {
|
|
216
|
-
async loadSession(sessionId) { return db.sessions.get(sessionId) }
|
|
217
|
-
async saveSession(session) { await db.sessions.put(session.sessionId, session) }
|
|
218
|
-
}
|
|
219
|
-
|
|
220
|
-
const agent = new Agent(provider, {
|
|
221
|
-
maxTokens: 4096,
|
|
222
|
-
sessionStore: new MySessionStore(),
|
|
223
|
-
})
|
|
238
|
+
const result = await runner.dream("my-agent", Date.now())
|
|
224
239
|
```
|
|
225
240
|
|
|
226
241
|
---
|
|
@@ -252,7 +267,13 @@ gov.requireParam("write_file", "path")
|
|
|
252
267
|
gov.allowParamValues("set_mode", "mode", ["read", "write"])
|
|
253
268
|
gov.limitParamRange("sleep", "seconds", 0, 10)
|
|
254
269
|
|
|
255
|
-
const
|
|
270
|
+
const runner = new RuntimeRunner({
|
|
271
|
+
provider,
|
|
272
|
+
executionPlane: plane,
|
|
273
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
274
|
+
maxTokens: 4096,
|
|
275
|
+
governance: gov,
|
|
276
|
+
})
|
|
256
277
|
// Every tool call goes through: Permission → Veto → RateLimit → Constraint → Audit
|
|
257
278
|
```
|
|
258
279
|
|
|
@@ -267,10 +288,16 @@ const gw = new SignalGateway()
|
|
|
267
288
|
gw.schedule(new ScheduledPrompt("standup", Date.now() + 3600_000))
|
|
268
289
|
gw.ingest({ kind: "interrupt", urgency: "critical", payload: {} })
|
|
269
290
|
|
|
270
|
-
const
|
|
271
|
-
|
|
291
|
+
const runner = new RuntimeRunner({
|
|
292
|
+
provider,
|
|
293
|
+
executionPlane: plane,
|
|
294
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
295
|
+
maxTokens: 4096,
|
|
296
|
+
signalSource: gw,
|
|
297
|
+
})
|
|
298
|
+
// kind="interrupt" → immediately stops the running runner
|
|
272
299
|
|
|
273
|
-
|
|
300
|
+
runner.interrupt() // also works directly
|
|
274
301
|
gw.destroy()
|
|
275
302
|
```
|
|
276
303
|
|
|
@@ -282,22 +309,21 @@ gw.destroy()
|
|
|
282
309
|
import { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "@deepstrike/sdk"
|
|
283
310
|
|
|
284
311
|
// 1. SinglePass — run once, always passes
|
|
285
|
-
const outcome = await new SinglePassHarness(
|
|
312
|
+
const outcome = await new SinglePassHarness(runner).run({ goal: "Say hello" })
|
|
286
313
|
|
|
287
314
|
// 2. EvalLoop — retry until QualityGate passes
|
|
288
|
-
const harness = new EvalLoopHarness(
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
})
|
|
315
|
+
const harness = new EvalLoopHarness(runner, {
|
|
316
|
+
async evaluate(_req, out) { return out.result.includes("hello") },
|
|
317
|
+
}, 3)
|
|
292
318
|
|
|
293
319
|
// 3. HarnessLoop — LLM-as-judge with feedback injection + skill extraction
|
|
294
|
-
const loop = new HarnessLoop(
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
})
|
|
299
|
-
|
|
300
|
-
|
|
320
|
+
const loop = new HarnessLoop(runner, evalProvider, { maxAttempts: 3, skillDir: "./skills" })
|
|
321
|
+
for await (const event of loop.runStreaming({
|
|
322
|
+
goal: "Write a haiku",
|
|
323
|
+
criteria: [{ text: "Must be 3 lines", required: true }],
|
|
324
|
+
})) {
|
|
325
|
+
if (event.type === "done") console.log(event.verdict.passed, event.verdict.feedback)
|
|
326
|
+
}
|
|
301
327
|
```
|
|
302
328
|
|
|
303
329
|
---
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { collectText } from "../runtime/runner.js";
|
|
1
2
|
import { formatContractForSystemPrompt, contractToCriteriaStrings } from "./contract.js";
|
|
2
3
|
import { HandoffBus } from "./handoff.js";
|
|
3
4
|
/**
|
|
@@ -42,7 +43,11 @@ export class ContractDrivenHarness {
|
|
|
42
43
|
? `\n\n[Previous attempt failed. Violations to fix:\n${this._formatViolationsForFeedback(checkResults)}]`
|
|
43
44
|
: "";
|
|
44
45
|
const executorGoal = `${contractBlock}\n\n---\n\n${currentGoal}${violationNote}`;
|
|
45
|
-
artifact = await this.pool.get("executor").run(
|
|
46
|
+
artifact = await collectText(this.pool.get("executor").run({
|
|
47
|
+
sessionId: crypto.randomUUID(),
|
|
48
|
+
goal: executorGoal,
|
|
49
|
+
criteria: contractToCriteriaStrings(this.contract),
|
|
50
|
+
}));
|
|
46
51
|
// ── Phase 2: Verifier ──────────────────────────────────────────────────
|
|
47
52
|
// Verifier sees: artifact + contract only. No executor history.
|
|
48
53
|
const auditText = await this.pool.runVerifier({ contract: this.contract, artifact });
|
|
@@ -15,8 +15,8 @@ export interface CreatorVerifierMetrics {
|
|
|
15
15
|
* Usage:
|
|
16
16
|
* ```ts
|
|
17
17
|
* const pool = new AgentPool()
|
|
18
|
-
* .add("executor",
|
|
19
|
-
* .add("verifier",
|
|
18
|
+
* .add("executor", executorRunner)
|
|
19
|
+
* .add("verifier", verifierRunner)
|
|
20
20
|
*
|
|
21
21
|
* const mode = new CreatorVerifierMode(pool)
|
|
22
22
|
* const result = await mode.run(contract)
|
|
@@ -8,8 +8,8 @@ import { ContractDrivenHarness } from "../harness.js";
|
|
|
8
8
|
* Usage:
|
|
9
9
|
* ```ts
|
|
10
10
|
* const pool = new AgentPool()
|
|
11
|
-
* .add("executor",
|
|
12
|
-
* .add("verifier",
|
|
11
|
+
* .add("executor", executorRunner)
|
|
12
|
+
* .add("verifier", verifierRunner)
|
|
13
13
|
*
|
|
14
14
|
* const mode = new CreatorVerifierMode(pool)
|
|
15
15
|
* const result = await mode.run(contract)
|
|
@@ -1,46 +1,15 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { RuntimeRunner } from "../runtime/runner.js";
|
|
2
2
|
import type { VerificationContract } from "./contract.js";
|
|
3
|
-
/**
|
|
4
|
-
* Roles in a multi-agent collaboration.
|
|
5
|
-
*
|
|
6
|
-
* - orchestrator: strong-reasoning, produces VerificationContracts; no tools beyond planning
|
|
7
|
-
* - executor: code/task execution, full tool access, sees goal + contract only
|
|
8
|
-
* - verifier: adversarial auditor, no tools, low temperature, sees artifact + contract only
|
|
9
|
-
*/
|
|
10
3
|
export type AgentRole = "orchestrator" | "executor" | "verifier";
|
|
11
|
-
/** Context passed to a verifier run — intentionally minimal. */
|
|
12
4
|
export interface IsolatedVerifierContext {
|
|
13
5
|
contract: VerificationContract;
|
|
14
|
-
/** The artifact produced by the executor. */
|
|
15
6
|
artifact: string;
|
|
16
7
|
}
|
|
17
|
-
/**
|
|
18
|
-
* AgentPool manages a set of role-specific Agent instances.
|
|
19
|
-
*
|
|
20
|
-
* Each role runs in its own Agent instance with an independent history partition,
|
|
21
|
-
* ensuring that the verifier never sees the executor's implementation transcript.
|
|
22
|
-
*
|
|
23
|
-
* Usage:
|
|
24
|
-
* ```ts
|
|
25
|
-
* const pool = new AgentPool()
|
|
26
|
-
* .add("executor", executorAgent)
|
|
27
|
-
* .add("verifier", verifierAgent)
|
|
28
|
-
* ```
|
|
29
|
-
*/
|
|
30
8
|
export declare class AgentPool {
|
|
31
|
-
private
|
|
32
|
-
add(role: AgentRole,
|
|
9
|
+
private runners;
|
|
10
|
+
add(role: AgentRole, runner: RuntimeRunner): this;
|
|
33
11
|
has(role: AgentRole): boolean;
|
|
34
|
-
get(role: AgentRole):
|
|
35
|
-
/**
|
|
36
|
-
* Run the verifier with an isolated context — only the artifact and contract.
|
|
37
|
-
* The verifier does NOT receive the executor's conversation history.
|
|
38
|
-
* Returns a structured audit response as a plain string.
|
|
39
|
-
*/
|
|
12
|
+
get(role: AgentRole): RuntimeRunner;
|
|
40
13
|
runVerifier(ctx: IsolatedVerifierContext): Promise<string>;
|
|
41
|
-
/**
|
|
42
|
-
* Run the orchestrator to decompose a high-level goal into a VerificationContract.
|
|
43
|
-
* The orchestrator receives the goal and must produce a structured contract in JSON.
|
|
44
|
-
*/
|
|
45
14
|
runOrchestrator(goal: string): Promise<string>;
|
|
46
15
|
}
|
|
@@ -1,64 +1,38 @@
|
|
|
1
|
+
import { collectText } from "../runtime/runner.js";
|
|
1
2
|
import { formatContractForSystemPrompt } from "./contract.js";
|
|
2
|
-
/**
|
|
3
|
-
* AgentPool manages a set of role-specific Agent instances.
|
|
4
|
-
*
|
|
5
|
-
* Each role runs in its own Agent instance with an independent history partition,
|
|
6
|
-
* ensuring that the verifier never sees the executor's implementation transcript.
|
|
7
|
-
*
|
|
8
|
-
* Usage:
|
|
9
|
-
* ```ts
|
|
10
|
-
* const pool = new AgentPool()
|
|
11
|
-
* .add("executor", executorAgent)
|
|
12
|
-
* .add("verifier", verifierAgent)
|
|
13
|
-
* ```
|
|
14
|
-
*/
|
|
15
3
|
export class AgentPool {
|
|
16
|
-
|
|
17
|
-
add(role,
|
|
18
|
-
this.
|
|
4
|
+
runners = new Map();
|
|
5
|
+
add(role, runner) {
|
|
6
|
+
this.runners.set(role, runner);
|
|
19
7
|
return this;
|
|
20
8
|
}
|
|
21
9
|
has(role) {
|
|
22
|
-
return this.
|
|
10
|
+
return this.runners.has(role);
|
|
23
11
|
}
|
|
24
12
|
get(role) {
|
|
25
|
-
const
|
|
26
|
-
if (!
|
|
27
|
-
throw new Error(`AgentPool: no
|
|
28
|
-
return
|
|
13
|
+
const runner = this.runners.get(role);
|
|
14
|
+
if (!runner)
|
|
15
|
+
throw new Error(`AgentPool: no runner registered for role "${role}"`);
|
|
16
|
+
return runner;
|
|
29
17
|
}
|
|
30
|
-
/**
|
|
31
|
-
* Run the verifier with an isolated context — only the artifact and contract.
|
|
32
|
-
* The verifier does NOT receive the executor's conversation history.
|
|
33
|
-
* Returns a structured audit response as a plain string.
|
|
34
|
-
*/
|
|
35
18
|
async runVerifier(ctx) {
|
|
36
|
-
const
|
|
19
|
+
const runner = this.get("verifier");
|
|
37
20
|
const contractBlock = formatContractForSystemPrompt(ctx.contract);
|
|
38
21
|
const auditGoal = [
|
|
39
|
-
contractBlock,
|
|
40
|
-
"",
|
|
41
|
-
"
|
|
42
|
-
"",
|
|
43
|
-
"
|
|
44
|
-
"",
|
|
45
|
-
ctx.artifact,
|
|
46
|
-
"",
|
|
47
|
-
"---",
|
|
48
|
-
"",
|
|
22
|
+
contractBlock, "",
|
|
23
|
+
"---", "",
|
|
24
|
+
"## Artifact to Audit", "",
|
|
25
|
+
ctx.artifact, "",
|
|
26
|
+
"---", "",
|
|
49
27
|
"Audit the artifact against every criterion in the contract above.",
|
|
50
28
|
"For each criterion, state whether it PASSED or FAILED and cite specific evidence.",
|
|
51
29
|
"List any anti-patterns you detected.",
|
|
52
30
|
"Conclude with an overall PASS or FAIL verdict.",
|
|
53
31
|
].join("\n");
|
|
54
|
-
return
|
|
32
|
+
return collectText(runner.run({ sessionId: crypto.randomUUID(), goal: auditGoal }));
|
|
55
33
|
}
|
|
56
|
-
/**
|
|
57
|
-
* Run the orchestrator to decompose a high-level goal into a VerificationContract.
|
|
58
|
-
* The orchestrator receives the goal and must produce a structured contract in JSON.
|
|
59
|
-
*/
|
|
60
34
|
async runOrchestrator(goal) {
|
|
61
|
-
const
|
|
35
|
+
const runner = this.get("orchestrator");
|
|
62
36
|
const orchestratorGoal = [
|
|
63
37
|
`You are a planning orchestrator. Decompose the following goal into a VerificationContract.`,
|
|
64
38
|
``,
|
|
@@ -75,6 +49,6 @@ export class AgentPool {
|
|
|
75
49
|
``,
|
|
76
50
|
`Output ONLY the JSON object, no prose.`,
|
|
77
51
|
].join("\n");
|
|
78
|
-
return
|
|
52
|
+
return collectText(runner.run({ sessionId: crypto.randomUUID(), goal: orchestratorGoal }));
|
|
79
53
|
}
|
|
80
54
|
}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { RuntimeRunner } from "../runtime/runner.js";
|
|
2
|
+
import { collectText } from "../runtime/runner.js";
|
|
2
3
|
export interface Criterion {
|
|
3
4
|
text: string;
|
|
4
5
|
required: boolean;
|
|
@@ -71,15 +72,15 @@ export interface QualityGate {
|
|
|
71
72
|
evaluate(request: HarnessRequest, outcome: HarnessOutcome): Promise<boolean>;
|
|
72
73
|
}
|
|
73
74
|
export declare class SinglePassHarness {
|
|
74
|
-
private
|
|
75
|
-
constructor(
|
|
75
|
+
private runner;
|
|
76
|
+
constructor(runner: RuntimeRunner);
|
|
76
77
|
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
77
78
|
}
|
|
78
79
|
export declare class EvalLoopHarness {
|
|
79
|
-
private
|
|
80
|
+
private runner;
|
|
80
81
|
private gate;
|
|
81
82
|
private maxAttempts;
|
|
82
|
-
constructor(
|
|
83
|
+
constructor(runner: RuntimeRunner, gate: QualityGate, maxAttempts?: number);
|
|
83
84
|
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
84
85
|
}
|
|
85
86
|
export interface HarnessLoopOptions {
|
|
@@ -87,10 +88,11 @@ export interface HarnessLoopOptions {
|
|
|
87
88
|
skillDir?: string;
|
|
88
89
|
}
|
|
89
90
|
export declare class HarnessLoop {
|
|
90
|
-
private
|
|
91
|
+
private runner;
|
|
91
92
|
private evalProvider;
|
|
92
93
|
private maxAttempts;
|
|
93
94
|
private skillDir?;
|
|
94
|
-
constructor(
|
|
95
|
+
constructor(runner: RuntimeRunner, evalProvider: import("../types.js").LLMProvider, options?: HarnessLoopOptions);
|
|
95
96
|
runStreaming(request: HarnessRequest): AsyncIterable<HarnessEvent>;
|
|
96
97
|
}
|
|
98
|
+
export { collectText };
|
package/dist/harness/harness.js
CHANGED
|
@@ -1,10 +1,12 @@
|
|
|
1
|
+
import { collectText } from "../runtime/runner.js";
|
|
1
2
|
import { writeFile } from "fs/promises";
|
|
2
3
|
import path from "path";
|
|
3
4
|
import { getKernel } from "../kernel.js";
|
|
4
|
-
async function runOnce(
|
|
5
|
+
async function runOnce(runner, req) {
|
|
5
6
|
let text = "";
|
|
6
7
|
let done;
|
|
7
|
-
|
|
8
|
+
const sessionId = crypto.randomUUID();
|
|
9
|
+
for await (const evt of runner.run({ sessionId, goal: req.goal, criteria: req.criteria?.map(c => c.text), extensions: req.extensions })) {
|
|
8
10
|
if (evt.type === "text_delta")
|
|
9
11
|
text += evt.delta;
|
|
10
12
|
else if (evt.type === "done")
|
|
@@ -13,27 +15,27 @@ async function runOnce(agent, req) {
|
|
|
13
15
|
return { result: text, passed: false, iterations: done?.iterations ?? 0, totalTokens: done?.totalTokens ?? 0, status: done?.status ?? "error" };
|
|
14
16
|
}
|
|
15
17
|
export class SinglePassHarness {
|
|
16
|
-
|
|
17
|
-
constructor(
|
|
18
|
-
this.
|
|
18
|
+
runner;
|
|
19
|
+
constructor(runner) {
|
|
20
|
+
this.runner = runner;
|
|
19
21
|
}
|
|
20
22
|
async run(request) {
|
|
21
|
-
return { ...await runOnce(this.
|
|
23
|
+
return { ...await runOnce(this.runner, request), passed: true };
|
|
22
24
|
}
|
|
23
25
|
}
|
|
24
26
|
export class EvalLoopHarness {
|
|
25
|
-
|
|
27
|
+
runner;
|
|
26
28
|
gate;
|
|
27
29
|
maxAttempts;
|
|
28
|
-
constructor(
|
|
29
|
-
this.
|
|
30
|
+
constructor(runner, gate, maxAttempts = 3) {
|
|
31
|
+
this.runner = runner;
|
|
30
32
|
this.gate = gate;
|
|
31
33
|
this.maxAttempts = maxAttempts;
|
|
32
34
|
}
|
|
33
35
|
async run(request) {
|
|
34
36
|
let outcome = { result: "", passed: false, iterations: 0, totalTokens: 0, status: "error" };
|
|
35
37
|
for (let i = 0; i < this.maxAttempts; i++) {
|
|
36
|
-
outcome = await runOnce(this.
|
|
38
|
+
outcome = await runOnce(this.runner, request);
|
|
37
39
|
if (await this.gate.evaluate(request, outcome))
|
|
38
40
|
return { ...outcome, passed: true };
|
|
39
41
|
}
|
|
@@ -41,12 +43,12 @@ export class EvalLoopHarness {
|
|
|
41
43
|
}
|
|
42
44
|
}
|
|
43
45
|
export class HarnessLoop {
|
|
44
|
-
|
|
46
|
+
runner;
|
|
45
47
|
evalProvider;
|
|
46
48
|
maxAttempts;
|
|
47
49
|
skillDir;
|
|
48
|
-
constructor(
|
|
49
|
-
this.
|
|
50
|
+
constructor(runner, evalProvider, options = {}) {
|
|
51
|
+
this.runner = runner;
|
|
50
52
|
this.evalProvider = evalProvider;
|
|
51
53
|
this.maxAttempts = options.maxAttempts ?? 3;
|
|
52
54
|
this.skillDir = options.skillDir;
|
|
@@ -55,22 +57,19 @@ export class HarnessLoop {
|
|
|
55
57
|
const kernel = getKernel();
|
|
56
58
|
const pipeline = new kernel.EvalPipeline({ extractSkillOnPass: true });
|
|
57
59
|
const criteria = request.criteria ?? [];
|
|
58
|
-
// history tracks the full conversation; agent sees only the current goal message
|
|
59
|
-
// but we inject feedback as user messages to guide revision without discarding prior output
|
|
60
60
|
let currentGoal = request.goal;
|
|
61
61
|
let lastIterations = 0;
|
|
62
62
|
let lastTotalTokens = 0;
|
|
63
63
|
let lastStatus = "error";
|
|
64
64
|
let lastResult = "";
|
|
65
65
|
for (let attempt = 1; attempt <= this.maxAttempts; attempt++) {
|
|
66
|
-
|
|
67
|
-
for await (const evt of this.
|
|
66
|
+
const sessionId = crypto.randomUUID();
|
|
67
|
+
for await (const evt of this.runner.run({ sessionId, goal: currentGoal, criteria: criteria.map(c => c.text), extensions: request.extensions })) {
|
|
68
68
|
if (evt.type === "text_delta") {
|
|
69
69
|
lastResult += evt.delta;
|
|
70
70
|
yield { type: "token", text: evt.delta };
|
|
71
71
|
}
|
|
72
72
|
else if (evt.type === "tool_call") {
|
|
73
|
-
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
74
73
|
const tc = evt;
|
|
75
74
|
yield { type: "tool_call", id: tc.id, name: tc.name };
|
|
76
75
|
}
|
|
@@ -83,7 +82,6 @@ export class HarnessLoop {
|
|
|
83
82
|
yield { type: "tool_suspend", callId: ts.callId, suspensionId: ts.suspensionId, ...(ts.payload ? { payload: ts.payload } : {}) };
|
|
84
83
|
}
|
|
85
84
|
else if (evt.type === "tool_result") {
|
|
86
|
-
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
87
85
|
const tr = evt;
|
|
88
86
|
yield { type: "tool_result", callId: tr.callId, content: tr.content, isError: tr.isError };
|
|
89
87
|
}
|
|
@@ -94,13 +92,11 @@ export class HarnessLoop {
|
|
|
94
92
|
lastStatus = d.status;
|
|
95
93
|
}
|
|
96
94
|
}
|
|
97
|
-
// Checkpoint: evaluator judges the completed output
|
|
98
95
|
yield { type: "supervising" };
|
|
99
96
|
const evalAction = pipeline.feedOutcome(request.goal, criteria, lastResult, attempt);
|
|
100
97
|
if (evalAction.kind !== "evaluate")
|
|
101
98
|
break;
|
|
102
99
|
let evalText = "";
|
|
103
|
-
// Wrap harness eval messages in a RenderedContext (system messages → systemText, rest → turns).
|
|
104
100
|
const evalMsgs = evalAction.messages ?? [];
|
|
105
101
|
const evalContext = {
|
|
106
102
|
systemText: evalMsgs.filter((m) => m.role === "system").map((m) => m.content).join("\n\n"),
|
|
@@ -131,7 +127,6 @@ export class HarnessLoop {
|
|
|
131
127
|
return;
|
|
132
128
|
}
|
|
133
129
|
yield { type: "revising", verdict };
|
|
134
|
-
// Inject feedback as next turn's goal — agent sees its prior output + evaluator notes
|
|
135
130
|
currentGoal = `${request.goal}\n\n[Attempt ${attempt} feedback: ${verdict.feedback}]`;
|
|
136
131
|
lastResult = "";
|
|
137
132
|
pipeline.reset();
|
|
@@ -139,3 +134,5 @@ export class HarnessLoop {
|
|
|
139
134
|
yield { type: "max_attempts_reached" };
|
|
140
135
|
}
|
|
141
136
|
}
|
|
137
|
+
// Re-export collectText so harness callers can use it without knowing runner internals.
|
|
138
|
+
export { collectText };
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,17 @@
|
|
|
1
|
-
export {
|
|
2
|
-
export type {
|
|
1
|
+
export { RuntimeRunner, collectText } from "./runtime/runner.js";
|
|
2
|
+
export type { RuntimeOptions } from "./runtime/runner.js";
|
|
3
|
+
export { LocalExecutionPlane } from "./runtime/execution-plane.js";
|
|
4
|
+
export type { ExecutionPlane, RunContext } from "./runtime/execution-plane.js";
|
|
5
|
+
export { InMemorySessionLog, FileSessionLog } from "./runtime/session-log.js";
|
|
6
|
+
export type { SessionLog, SessionEvent } from "./runtime/session-log.js";
|
|
7
|
+
export { EnvCredentialVault, InMemoryCredentialVault, ChainedCredentialVault } from "./runtime/credential-vault.js";
|
|
8
|
+
export type { CredentialVault } from "./runtime/credential-vault.js";
|
|
9
|
+
export { ProcessSandboxPlane } from "./runtime/process-sandbox-plane.js";
|
|
10
|
+
export type { SandboxOptions } from "./runtime/process-sandbox-plane.js";
|
|
11
|
+
export { McpProxyPlane } from "./runtime/mcp-proxy-plane.js";
|
|
12
|
+
export type { McpServerConfig } from "./runtime/mcp-proxy-plane.js";
|
|
13
|
+
export { RemoteVpcPlane } from "./runtime/remote-vpc-plane.js";
|
|
14
|
+
export type { RemoteVpcOptions } from "./runtime/remote-vpc-plane.js";
|
|
3
15
|
export { AnthropicProvider } from "./providers/anthropic.js";
|
|
4
16
|
export { OpenAIChatProvider, OpenAIProvider } from "./providers/openai.js";
|
|
5
17
|
export { DeepSeekProvider } from "./providers/deepseek.js";
|
|
@@ -20,10 +32,8 @@ export type { RegisteredTool } from "./tools/index.js";
|
|
|
20
32
|
export { scanSkillDir, readSkillFile } from "./skills/loader.js";
|
|
21
33
|
export type { SkillMetadata } from "./skills/loader.js";
|
|
22
34
|
export { WorkingMemory } from "./memory/working.js";
|
|
23
|
-
export type { DreamStore, DreamResult,
|
|
35
|
+
export type { DreamStore, DreamResult, SessionData, SessionMessage, MemoryEntry, CurationResult, CurationStats, } from "./memory/protocols.js";
|
|
24
36
|
export type { KnowledgeSource } from "./knowledge/source.js";
|
|
25
|
-
export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
|
|
26
|
-
export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions, QualityGate } from "./harness/harness.js";
|
|
27
37
|
export { ScheduledPrompt } from "./signals/scheduled.js";
|
|
28
38
|
export { SignalGateway } from "./signals/gateway.js";
|
|
29
39
|
export type { RuntimeSignal, SignalSource } from "./signals/types.js";
|
|
@@ -31,6 +41,8 @@ export { PermissionManager, PermissionMode } from "./safety/permissions.js";
|
|
|
31
41
|
export type { PermissionDecision, Permission } from "./safety/permissions.js";
|
|
32
42
|
export { Governance } from "./governance.js";
|
|
33
43
|
export type { GovernanceVerdict } from "./governance.js";
|
|
44
|
+
export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
|
|
45
|
+
export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions, QualityGate } from "./harness/harness.js";
|
|
34
46
|
export type { Message, ToolCall, ToolResult, ToolSchema, ContentPart, TextPart, ImagePart, AudioPart, StreamEvent, TextDelta, ThinkingDelta, ToolCallEvent, ToolChunk, ToolDeltaEvent, ToolSuspendEvent, ToolResultEvent, DoneEvent, ErrorEvent, PermissionRequestEvent, LLMProvider, RetryConfig, TokenUsage, ProviderToolSpec, ProviderRunState, RenderedContext, } from "./types.js";
|
|
35
47
|
export type { AcceptanceCriterion, VerificationContract, ContractCheckResult, } from "./collaboration/contract.js";
|
|
36
48
|
export { ContractBuilder, formatContractForSystemPrompt, contractToCriteriaStrings, } from "./collaboration/contract.js";
|
package/dist/index.js
CHANGED
|
@@ -1,4 +1,12 @@
|
|
|
1
|
-
|
|
1
|
+
// ── Runtime (Layer 1.5) ────────────────────────────────────────────────────
|
|
2
|
+
export { RuntimeRunner, collectText } from "./runtime/runner.js";
|
|
3
|
+
export { LocalExecutionPlane } from "./runtime/execution-plane.js";
|
|
4
|
+
export { InMemorySessionLog, FileSessionLog } from "./runtime/session-log.js";
|
|
5
|
+
export { EnvCredentialVault, InMemoryCredentialVault, ChainedCredentialVault } from "./runtime/credential-vault.js";
|
|
6
|
+
export { ProcessSandboxPlane } from "./runtime/process-sandbox-plane.js";
|
|
7
|
+
export { McpProxyPlane } from "./runtime/mcp-proxy-plane.js";
|
|
8
|
+
export { RemoteVpcPlane } from "./runtime/remote-vpc-plane.js";
|
|
9
|
+
// ── Providers ─────────────────────────────────────────────────────────────
|
|
2
10
|
export { AnthropicProvider } from "./providers/anthropic.js";
|
|
3
11
|
export { OpenAIChatProvider, OpenAIProvider } from "./providers/openai.js";
|
|
4
12
|
export { DeepSeekProvider } from "./providers/deepseek.js";
|
|
@@ -12,14 +20,18 @@ export { OpenAIChatAdapter } from "./providers/openai-chat.js";
|
|
|
12
20
|
export { OpenAIResponsesAdapter, OpenAIResponsesProvider } from "./providers/openai-responses.js";
|
|
13
21
|
export { endpointProfiles, modelProfiles, getModelProfile } from "./providers/profiles.js";
|
|
14
22
|
export { createProvider } from "./providers/catalog.js";
|
|
23
|
+
// ── Tools & Skills ─────────────────────────────────────────────────────────
|
|
15
24
|
export { tool, streamingTool, executeTools, readFile, validateToolArguments } from "./tools/index.js";
|
|
16
25
|
export { scanSkillDir, readSkillFile } from "./skills/loader.js";
|
|
26
|
+
// ── Memory ─────────────────────────────────────────────────────────────────
|
|
17
27
|
export { WorkingMemory } from "./memory/working.js";
|
|
18
|
-
export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
|
|
19
28
|
export { ScheduledPrompt } from "./signals/scheduled.js";
|
|
20
29
|
export { SignalGateway } from "./signals/gateway.js";
|
|
30
|
+
// ── Safety & Governance ────────────────────────────────────────────────────
|
|
21
31
|
export { PermissionManager, PermissionMode } from "./safety/permissions.js";
|
|
22
32
|
export { Governance } from "./governance.js";
|
|
33
|
+
// ── Harness ────────────────────────────────────────────────────────────────
|
|
34
|
+
export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
|
|
23
35
|
export { ContractBuilder, formatContractForSystemPrompt, contractToCriteriaStrings, } from "./collaboration/contract.js";
|
|
24
36
|
export { AgentPool } from "./collaboration/pool.js";
|
|
25
37
|
export { ContractDrivenHarness } from "./collaboration/harness.js";
|