@deepstrike/sdk 0.1.12 → 0.1.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +82 -52
- package/dist/collaboration/harness.js +6 -1
- package/dist/collaboration/modes/creator-verifier.d.ts +2 -2
- package/dist/collaboration/modes/creator-verifier.js +2 -2
- package/dist/collaboration/pool.d.ts +4 -35
- package/dist/collaboration/pool.js +18 -44
- package/dist/harness/harness.d.ts +19 -7
- package/dist/harness/harness.js +27 -22
- package/dist/index.d.ts +19 -7
- package/dist/index.js +15 -3
- package/dist/kernel.d.ts +1 -0
- package/dist/kernel.js +10 -1
- package/dist/memory/protocols.d.ts +0 -4
- package/dist/providers/anthropic.d.ts +7 -2
- package/dist/providers/anthropic.js +48 -9
- package/dist/providers/base.d.ts +1 -0
- package/dist/providers/base.js +6 -0
- package/dist/providers/deepseek.d.ts +4 -2
- package/dist/providers/deepseek.js +49 -1
- package/dist/providers/gemini.d.ts +4 -2
- package/dist/providers/gemini.js +24 -5
- package/dist/providers/kimi.d.ts +3 -1
- package/dist/providers/kimi.js +10 -0
- package/dist/providers/minimax.d.ts +3 -1
- package/dist/providers/minimax.js +8 -0
- package/dist/providers/ollama.d.ts +5 -2
- package/dist/providers/ollama.js +63 -7
- package/dist/providers/openai-responses.d.ts +4 -2
- package/dist/providers/openai-responses.js +23 -3
- package/dist/providers/openai.d.ts +4 -2
- package/dist/providers/openai.js +44 -3
- package/dist/providers/profiles.d.ts +502 -6
- package/dist/providers/profiles.js +225 -82
- package/dist/providers/qwen.d.ts +5 -2
- package/dist/providers/qwen.js +59 -11
- package/dist/runtime/credential-vault.d.ts +18 -0
- package/dist/runtime/credential-vault.js +33 -0
- package/dist/runtime/execution-plane.d.ts +38 -0
- package/dist/runtime/execution-plane.js +146 -0
- package/dist/runtime/mcp-proxy-plane.d.ts +51 -0
- package/dist/runtime/mcp-proxy-plane.js +204 -0
- package/dist/runtime/process-sandbox-plane.d.ts +32 -0
- package/dist/runtime/process-sandbox-plane.js +109 -0
- package/dist/runtime/remote-vpc-plane.d.ts +48 -0
- package/dist/runtime/remote-vpc-plane.js +82 -0
- package/dist/runtime/runner.d.ts +49 -0
- package/dist/runtime/runner.js +364 -0
- package/dist/runtime/session-log.d.ts +67 -0
- package/dist/runtime/session-log.js +84 -0
- package/dist/tools/index.d.ts +7 -2
- package/dist/tools/index.js +83 -0
- package/dist/types.d.ts +52 -1
- package/package.json +3 -2
- package/dist/agent.d.ts +0 -88
- package/dist/agent.js +0 -430
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# DeepStrike Node.js SDK
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Runtime framework built on a Rust kernel. The kernel handles loop control, context compression, skill routing, governance, signal prioritization — the SDK handles all I/O.
|
|
4
4
|
|
|
5
5
|
## Install
|
|
6
6
|
|
|
@@ -33,7 +33,14 @@ The correct platform package is selected and installed automatically via `option
|
|
|
33
33
|
## Quick start
|
|
34
34
|
|
|
35
35
|
```typescript
|
|
36
|
-
import {
|
|
36
|
+
import {
|
|
37
|
+
FileSessionLog,
|
|
38
|
+
LocalExecutionPlane,
|
|
39
|
+
RuntimeRunner,
|
|
40
|
+
OpenAIResponsesProvider,
|
|
41
|
+
collectText,
|
|
42
|
+
tool,
|
|
43
|
+
} from "@deepstrike/sdk"
|
|
37
44
|
|
|
38
45
|
const provider = new OpenAIResponsesProvider(process.env.OPENAI_API_KEY!, "gpt-5-mini")
|
|
39
46
|
|
|
@@ -43,26 +50,34 @@ const add = tool("add", "Add two numbers.", {
|
|
|
43
50
|
required: ["x", "y"],
|
|
44
51
|
}, async ({ x, y }) => String(Number(x) + Number(y)))
|
|
45
52
|
|
|
46
|
-
const
|
|
47
|
-
|
|
53
|
+
const plane = new LocalExecutionPlane().register(add)
|
|
54
|
+
const runner = new RuntimeRunner({
|
|
55
|
+
provider,
|
|
56
|
+
executionPlane: plane,
|
|
57
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
58
|
+
maxTokens: 4096,
|
|
59
|
+
})
|
|
48
60
|
|
|
49
|
-
const result = await
|
|
61
|
+
const result = await collectText(runner.run({
|
|
62
|
+
sessionId: "math-1",
|
|
63
|
+
goal: "What is 17 + 28?",
|
|
64
|
+
}))
|
|
50
65
|
console.log(result)
|
|
51
66
|
```
|
|
52
67
|
|
|
53
68
|
Same-session conversation continuity is explicit via `sessionId`:
|
|
54
69
|
|
|
55
70
|
```typescript
|
|
56
|
-
await
|
|
57
|
-
const reply = await
|
|
71
|
+
await collectText(runner.run({ sessionId: "chat-1", goal: "My name is Ada." }))
|
|
72
|
+
const reply = await collectText(runner.run({ sessionId: "chat-1", goal: "What is my name?" }))
|
|
58
73
|
```
|
|
59
74
|
|
|
60
|
-
|
|
75
|
+
Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when event replay should survive restarts. `wake(sessionId)` resumes from the event log without inserting a duplicate user start event.
|
|
61
76
|
|
|
62
77
|
Streaming:
|
|
63
78
|
|
|
64
79
|
```typescript
|
|
65
|
-
for await (const event of
|
|
80
|
+
for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize README.md" })) {
|
|
66
81
|
if (event.type === "text_delta") process.stdout.write(event.delta)
|
|
67
82
|
else if (event.type === "tool_call") console.log(`\n[→ ${event.name}]`)
|
|
68
83
|
else if (event.type === "tool_result") console.log(` = ${event.content}`)
|
|
@@ -88,6 +103,8 @@ for await (const event of agent.runStreaming("Summarize README.md")) {
|
|
|
88
103
|
|
|
89
104
|
All providers accept `RetryConfig` for exponential backoff and share a `CircuitBreaker`.
|
|
90
105
|
|
|
106
|
+
`extensions` are forwarded by every provider in both `complete()` and `stream()` while SDK-owned structural fields such as `model`, `messages`, `tools`, and streaming flags remain protected. Provider-specific controls still keep their native spellings: for example Anthropic `thinking` / `betas`, OpenAI Responses `reasoning`, Gemini `generationConfig`, Ollama `think` / `options`, DeepSeek `thinking` + `reasoningEffort`, and Qwen `enableThinking` + `thinkingBudget`.
|
|
107
|
+
|
|
91
108
|
OpenAI can also be selected through the provider catalog:
|
|
92
109
|
|
|
93
110
|
```typescript
|
|
@@ -101,14 +118,18 @@ const provider = createProvider({
|
|
|
101
118
|
|
|
102
119
|
---
|
|
103
120
|
|
|
104
|
-
##
|
|
121
|
+
## Runtime options
|
|
105
122
|
|
|
106
123
|
```typescript
|
|
107
|
-
const
|
|
124
|
+
const plane = new LocalExecutionPlane()
|
|
125
|
+
const runner = new RuntimeRunner({
|
|
126
|
+
provider,
|
|
127
|
+
executionPlane: plane,
|
|
128
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
108
129
|
maxTokens: 4096, // context window size
|
|
109
130
|
maxTurns: 25, // max turns (default 25)
|
|
110
131
|
timeoutMs: 60_000, // timeout in ms
|
|
111
|
-
extensions: { temperature: 0.1 }, //
|
|
132
|
+
extensions: { temperature: 0.1 }, // provider-native controls, passed through to the LLM
|
|
112
133
|
skillDir: "./skills", // skill .md files directory
|
|
113
134
|
knowledgeSource: myKS, // KnowledgeSource implementation
|
|
114
135
|
signalSource: rx, // SignalSource for external signals
|
|
@@ -125,19 +146,25 @@ const agent = new Agent(provider, {
|
|
|
125
146
|
```typescript
|
|
126
147
|
import { tool, readFile } from "@deepstrike/sdk"
|
|
127
148
|
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
149
|
+
plane.register(tool("search", "Search.", schema, async (args) => ...))
|
|
150
|
+
plane.register(readFile) // built-in: read files from disk
|
|
151
|
+
plane.unregister("search")
|
|
131
152
|
```
|
|
132
153
|
|
|
133
154
|
---
|
|
134
155
|
|
|
135
156
|
## Skills
|
|
136
157
|
|
|
137
|
-
Skills are `.md` files with YAML frontmatter. Set `skillDir` on the
|
|
158
|
+
Skills are `.md` files with YAML frontmatter. Set `skillDir` on the runner — the kernel auto-injects a `skill` meta-tool, and the LLM loads skills by name on demand.
|
|
138
159
|
|
|
139
160
|
```typescript
|
|
140
|
-
const
|
|
161
|
+
const runner = new RuntimeRunner({
|
|
162
|
+
provider,
|
|
163
|
+
executionPlane: plane,
|
|
164
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
165
|
+
maxTokens: 4096,
|
|
166
|
+
skillDir: "./skills",
|
|
167
|
+
})
|
|
141
168
|
```
|
|
142
169
|
|
|
143
170
|
```markdown
|
|
@@ -158,7 +185,10 @@ effort: 1
|
|
|
158
185
|
Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand.
|
|
159
186
|
|
|
160
187
|
```typescript
|
|
161
|
-
const
|
|
188
|
+
const runner = new RuntimeRunner({
|
|
189
|
+
provider,
|
|
190
|
+
executionPlane: plane,
|
|
191
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
162
192
|
maxTokens: 4096,
|
|
163
193
|
knowledgeSource: {
|
|
164
194
|
async retrieve(query: string, topK: number): Promise<string[]> {
|
|
@@ -194,7 +224,10 @@ class MyStore implements DreamStore {
|
|
|
194
224
|
async search(agentId, query, topK) { ... }
|
|
195
225
|
}
|
|
196
226
|
|
|
197
|
-
const
|
|
227
|
+
const runner = new RuntimeRunner({
|
|
228
|
+
provider,
|
|
229
|
+
executionPlane: plane,
|
|
230
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
198
231
|
maxTokens: 4096,
|
|
199
232
|
dreamStore: new MyStore(),
|
|
200
233
|
agentId: "my-agent", // enables `memory` meta-tool
|
|
@@ -202,23 +235,7 @@ const agent = new Agent(provider, {
|
|
|
202
235
|
|
|
203
236
|
// In-session: LLM calls memory(query) → DreamStore.search()
|
|
204
237
|
// Post-session: trigger memory consolidation
|
|
205
|
-
const result = await
|
|
206
|
-
```
|
|
207
|
-
|
|
208
|
-
### SessionStore (same-session transcript continuity)
|
|
209
|
-
|
|
210
|
-
```typescript
|
|
211
|
-
import type { SessionStore } from "@deepstrike/sdk"
|
|
212
|
-
|
|
213
|
-
class MySessionStore implements SessionStore {
|
|
214
|
-
async loadSession(sessionId) { return db.sessions.get(sessionId) }
|
|
215
|
-
async saveSession(session) { await db.sessions.put(session.sessionId, session) }
|
|
216
|
-
}
|
|
217
|
-
|
|
218
|
-
const agent = new Agent(provider, {
|
|
219
|
-
maxTokens: 4096,
|
|
220
|
-
sessionStore: new MySessionStore(),
|
|
221
|
-
})
|
|
238
|
+
const result = await runner.dream("my-agent", Date.now())
|
|
222
239
|
```
|
|
223
240
|
|
|
224
241
|
---
|
|
@@ -250,7 +267,13 @@ gov.requireParam("write_file", "path")
|
|
|
250
267
|
gov.allowParamValues("set_mode", "mode", ["read", "write"])
|
|
251
268
|
gov.limitParamRange("sleep", "seconds", 0, 10)
|
|
252
269
|
|
|
253
|
-
const
|
|
270
|
+
const runner = new RuntimeRunner({
|
|
271
|
+
provider,
|
|
272
|
+
executionPlane: plane,
|
|
273
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
274
|
+
maxTokens: 4096,
|
|
275
|
+
governance: gov,
|
|
276
|
+
})
|
|
254
277
|
// Every tool call goes through: Permission → Veto → RateLimit → Constraint → Audit
|
|
255
278
|
```
|
|
256
279
|
|
|
@@ -265,10 +288,16 @@ const gw = new SignalGateway()
|
|
|
265
288
|
gw.schedule(new ScheduledPrompt("standup", Date.now() + 3600_000))
|
|
266
289
|
gw.ingest({ kind: "interrupt", urgency: "critical", payload: {} })
|
|
267
290
|
|
|
268
|
-
const
|
|
269
|
-
|
|
291
|
+
const runner = new RuntimeRunner({
|
|
292
|
+
provider,
|
|
293
|
+
executionPlane: plane,
|
|
294
|
+
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
295
|
+
maxTokens: 4096,
|
|
296
|
+
signalSource: gw,
|
|
297
|
+
})
|
|
298
|
+
// kind="interrupt" → immediately stops the running runner
|
|
270
299
|
|
|
271
|
-
|
|
300
|
+
runner.interrupt() // also works directly
|
|
272
301
|
gw.destroy()
|
|
273
302
|
```
|
|
274
303
|
|
|
@@ -280,22 +309,21 @@ gw.destroy()
|
|
|
280
309
|
import { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "@deepstrike/sdk"
|
|
281
310
|
|
|
282
311
|
// 1. SinglePass — run once, always passes
|
|
283
|
-
const outcome = await new SinglePassHarness(
|
|
312
|
+
const outcome = await new SinglePassHarness(runner).run({ goal: "Say hello" })
|
|
284
313
|
|
|
285
314
|
// 2. EvalLoop — retry until QualityGate passes
|
|
286
|
-
const harness = new EvalLoopHarness(
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
})
|
|
315
|
+
const harness = new EvalLoopHarness(runner, {
|
|
316
|
+
async evaluate(_req, out) { return out.result.includes("hello") },
|
|
317
|
+
}, 3)
|
|
290
318
|
|
|
291
319
|
// 3. HarnessLoop — LLM-as-judge with feedback injection + skill extraction
|
|
292
|
-
const loop = new HarnessLoop(
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
})
|
|
297
|
-
|
|
298
|
-
|
|
320
|
+
const loop = new HarnessLoop(runner, evalProvider, { maxAttempts: 3, skillDir: "./skills" })
|
|
321
|
+
for await (const event of loop.runStreaming({
|
|
322
|
+
goal: "Write a haiku",
|
|
323
|
+
criteria: [{ text: "Must be 3 lines", required: true }],
|
|
324
|
+
})) {
|
|
325
|
+
if (event.type === "done") console.log(event.verdict.passed, event.verdict.feedback)
|
|
326
|
+
}
|
|
299
327
|
```
|
|
300
328
|
|
|
301
329
|
---
|
|
@@ -307,6 +335,8 @@ console.log(out.passed, out.feedback)
|
|
|
307
335
|
| `text_delta` | `delta` |
|
|
308
336
|
| `thinking_delta` | `delta` |
|
|
309
337
|
| `tool_call` | `id`, `name`, `arguments` |
|
|
338
|
+
| `tool_delta` | `callId`, `delta?`, `chunk?` |
|
|
339
|
+
| `tool_suspend` | `callId`, `suspensionId`, `payload?` |
|
|
310
340
|
| `tool_result` | `callId`, `content`, `isError` |
|
|
311
341
|
| `permission_request` | `toolName`, `reason` |
|
|
312
342
|
| `done` | `iterations`, `totalTokens`, `status` |
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { collectText } from "../runtime/runner.js";
|
|
1
2
|
import { formatContractForSystemPrompt, contractToCriteriaStrings } from "./contract.js";
|
|
2
3
|
import { HandoffBus } from "./handoff.js";
|
|
3
4
|
/**
|
|
@@ -42,7 +43,11 @@ export class ContractDrivenHarness {
|
|
|
42
43
|
? `\n\n[Previous attempt failed. Violations to fix:\n${this._formatViolationsForFeedback(checkResults)}]`
|
|
43
44
|
: "";
|
|
44
45
|
const executorGoal = `${contractBlock}\n\n---\n\n${currentGoal}${violationNote}`;
|
|
45
|
-
artifact = await this.pool.get("executor").run(
|
|
46
|
+
artifact = await collectText(this.pool.get("executor").run({
|
|
47
|
+
sessionId: crypto.randomUUID(),
|
|
48
|
+
goal: executorGoal,
|
|
49
|
+
criteria: contractToCriteriaStrings(this.contract),
|
|
50
|
+
}));
|
|
46
51
|
// ── Phase 2: Verifier ──────────────────────────────────────────────────
|
|
47
52
|
// Verifier sees: artifact + contract only. No executor history.
|
|
48
53
|
const auditText = await this.pool.runVerifier({ contract: this.contract, artifact });
|
|
@@ -15,8 +15,8 @@ export interface CreatorVerifierMetrics {
|
|
|
15
15
|
* Usage:
|
|
16
16
|
* ```ts
|
|
17
17
|
* const pool = new AgentPool()
|
|
18
|
-
* .add("executor",
|
|
19
|
-
* .add("verifier",
|
|
18
|
+
* .add("executor", executorRunner)
|
|
19
|
+
* .add("verifier", verifierRunner)
|
|
20
20
|
*
|
|
21
21
|
* const mode = new CreatorVerifierMode(pool)
|
|
22
22
|
* const result = await mode.run(contract)
|
|
@@ -8,8 +8,8 @@ import { ContractDrivenHarness } from "../harness.js";
|
|
|
8
8
|
* Usage:
|
|
9
9
|
* ```ts
|
|
10
10
|
* const pool = new AgentPool()
|
|
11
|
-
* .add("executor",
|
|
12
|
-
* .add("verifier",
|
|
11
|
+
* .add("executor", executorRunner)
|
|
12
|
+
* .add("verifier", verifierRunner)
|
|
13
13
|
*
|
|
14
14
|
* const mode = new CreatorVerifierMode(pool)
|
|
15
15
|
* const result = await mode.run(contract)
|
|
@@ -1,46 +1,15 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { RuntimeRunner } from "../runtime/runner.js";
|
|
2
2
|
import type { VerificationContract } from "./contract.js";
|
|
3
|
-
/**
|
|
4
|
-
* Roles in a multi-agent collaboration.
|
|
5
|
-
*
|
|
6
|
-
* - orchestrator: strong-reasoning, produces VerificationContracts; no tools beyond planning
|
|
7
|
-
* - executor: code/task execution, full tool access, sees goal + contract only
|
|
8
|
-
* - verifier: adversarial auditor, no tools, low temperature, sees artifact + contract only
|
|
9
|
-
*/
|
|
10
3
|
export type AgentRole = "orchestrator" | "executor" | "verifier";
|
|
11
|
-
/** Context passed to a verifier run — intentionally minimal. */
|
|
12
4
|
export interface IsolatedVerifierContext {
|
|
13
5
|
contract: VerificationContract;
|
|
14
|
-
/** The artifact produced by the executor. */
|
|
15
6
|
artifact: string;
|
|
16
7
|
}
|
|
17
|
-
/**
|
|
18
|
-
* AgentPool manages a set of role-specific Agent instances.
|
|
19
|
-
*
|
|
20
|
-
* Each role runs in its own Agent instance with an independent history partition,
|
|
21
|
-
* ensuring that the verifier never sees the executor's implementation transcript.
|
|
22
|
-
*
|
|
23
|
-
* Usage:
|
|
24
|
-
* ```ts
|
|
25
|
-
* const pool = new AgentPool()
|
|
26
|
-
* .add("executor", executorAgent)
|
|
27
|
-
* .add("verifier", verifierAgent)
|
|
28
|
-
* ```
|
|
29
|
-
*/
|
|
30
8
|
export declare class AgentPool {
|
|
31
|
-
private
|
|
32
|
-
add(role: AgentRole,
|
|
9
|
+
private runners;
|
|
10
|
+
add(role: AgentRole, runner: RuntimeRunner): this;
|
|
33
11
|
has(role: AgentRole): boolean;
|
|
34
|
-
get(role: AgentRole):
|
|
35
|
-
/**
|
|
36
|
-
* Run the verifier with an isolated context — only the artifact and contract.
|
|
37
|
-
* The verifier does NOT receive the executor's conversation history.
|
|
38
|
-
* Returns a structured audit response as a plain string.
|
|
39
|
-
*/
|
|
12
|
+
get(role: AgentRole): RuntimeRunner;
|
|
40
13
|
runVerifier(ctx: IsolatedVerifierContext): Promise<string>;
|
|
41
|
-
/**
|
|
42
|
-
* Run the orchestrator to decompose a high-level goal into a VerificationContract.
|
|
43
|
-
* The orchestrator receives the goal and must produce a structured contract in JSON.
|
|
44
|
-
*/
|
|
45
14
|
runOrchestrator(goal: string): Promise<string>;
|
|
46
15
|
}
|
|
@@ -1,64 +1,38 @@
|
|
|
1
|
+
import { collectText } from "../runtime/runner.js";
|
|
1
2
|
import { formatContractForSystemPrompt } from "./contract.js";
|
|
2
|
-
/**
|
|
3
|
-
* AgentPool manages a set of role-specific Agent instances.
|
|
4
|
-
*
|
|
5
|
-
* Each role runs in its own Agent instance with an independent history partition,
|
|
6
|
-
* ensuring that the verifier never sees the executor's implementation transcript.
|
|
7
|
-
*
|
|
8
|
-
* Usage:
|
|
9
|
-
* ```ts
|
|
10
|
-
* const pool = new AgentPool()
|
|
11
|
-
* .add("executor", executorAgent)
|
|
12
|
-
* .add("verifier", verifierAgent)
|
|
13
|
-
* ```
|
|
14
|
-
*/
|
|
15
3
|
export class AgentPool {
|
|
16
|
-
|
|
17
|
-
add(role,
|
|
18
|
-
this.
|
|
4
|
+
runners = new Map();
|
|
5
|
+
add(role, runner) {
|
|
6
|
+
this.runners.set(role, runner);
|
|
19
7
|
return this;
|
|
20
8
|
}
|
|
21
9
|
has(role) {
|
|
22
|
-
return this.
|
|
10
|
+
return this.runners.has(role);
|
|
23
11
|
}
|
|
24
12
|
get(role) {
|
|
25
|
-
const
|
|
26
|
-
if (!
|
|
27
|
-
throw new Error(`AgentPool: no
|
|
28
|
-
return
|
|
13
|
+
const runner = this.runners.get(role);
|
|
14
|
+
if (!runner)
|
|
15
|
+
throw new Error(`AgentPool: no runner registered for role "${role}"`);
|
|
16
|
+
return runner;
|
|
29
17
|
}
|
|
30
|
-
/**
|
|
31
|
-
* Run the verifier with an isolated context — only the artifact and contract.
|
|
32
|
-
* The verifier does NOT receive the executor's conversation history.
|
|
33
|
-
* Returns a structured audit response as a plain string.
|
|
34
|
-
*/
|
|
35
18
|
async runVerifier(ctx) {
|
|
36
|
-
const
|
|
19
|
+
const runner = this.get("verifier");
|
|
37
20
|
const contractBlock = formatContractForSystemPrompt(ctx.contract);
|
|
38
21
|
const auditGoal = [
|
|
39
|
-
contractBlock,
|
|
40
|
-
"",
|
|
41
|
-
"
|
|
42
|
-
"",
|
|
43
|
-
"
|
|
44
|
-
"",
|
|
45
|
-
ctx.artifact,
|
|
46
|
-
"",
|
|
47
|
-
"---",
|
|
48
|
-
"",
|
|
22
|
+
contractBlock, "",
|
|
23
|
+
"---", "",
|
|
24
|
+
"## Artifact to Audit", "",
|
|
25
|
+
ctx.artifact, "",
|
|
26
|
+
"---", "",
|
|
49
27
|
"Audit the artifact against every criterion in the contract above.",
|
|
50
28
|
"For each criterion, state whether it PASSED or FAILED and cite specific evidence.",
|
|
51
29
|
"List any anti-patterns you detected.",
|
|
52
30
|
"Conclude with an overall PASS or FAIL verdict.",
|
|
53
31
|
].join("\n");
|
|
54
|
-
return
|
|
32
|
+
return collectText(runner.run({ sessionId: crypto.randomUUID(), goal: auditGoal }));
|
|
55
33
|
}
|
|
56
|
-
/**
|
|
57
|
-
* Run the orchestrator to decompose a high-level goal into a VerificationContract.
|
|
58
|
-
* The orchestrator receives the goal and must produce a structured contract in JSON.
|
|
59
|
-
*/
|
|
60
34
|
async runOrchestrator(goal) {
|
|
61
|
-
const
|
|
35
|
+
const runner = this.get("orchestrator");
|
|
62
36
|
const orchestratorGoal = [
|
|
63
37
|
`You are a planning orchestrator. Decompose the following goal into a VerificationContract.`,
|
|
64
38
|
``,
|
|
@@ -75,6 +49,6 @@ export class AgentPool {
|
|
|
75
49
|
``,
|
|
76
50
|
`Output ONLY the JSON object, no prose.`,
|
|
77
51
|
].join("\n");
|
|
78
|
-
return
|
|
52
|
+
return collectText(runner.run({ sessionId: crypto.randomUUID(), goal: orchestratorGoal }));
|
|
79
53
|
}
|
|
80
54
|
}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { RuntimeRunner } from "../runtime/runner.js";
|
|
2
|
+
import { collectText } from "../runtime/runner.js";
|
|
2
3
|
export interface Criterion {
|
|
3
4
|
text: string;
|
|
4
5
|
required: boolean;
|
|
@@ -38,6 +39,16 @@ export type HarnessEvent = {
|
|
|
38
39
|
type: "tool_call";
|
|
39
40
|
id: string;
|
|
40
41
|
name: string;
|
|
42
|
+
} | {
|
|
43
|
+
type: "tool_delta";
|
|
44
|
+
callId: string;
|
|
45
|
+
delta?: string;
|
|
46
|
+
chunk?: Record<string, unknown>;
|
|
47
|
+
} | {
|
|
48
|
+
type: "tool_suspend";
|
|
49
|
+
callId: string;
|
|
50
|
+
suspensionId: string;
|
|
51
|
+
payload?: Record<string, unknown>;
|
|
41
52
|
} | {
|
|
42
53
|
type: "tool_result";
|
|
43
54
|
callId: string;
|
|
@@ -61,15 +72,15 @@ export interface QualityGate {
|
|
|
61
72
|
evaluate(request: HarnessRequest, outcome: HarnessOutcome): Promise<boolean>;
|
|
62
73
|
}
|
|
63
74
|
export declare class SinglePassHarness {
|
|
64
|
-
private
|
|
65
|
-
constructor(
|
|
75
|
+
private runner;
|
|
76
|
+
constructor(runner: RuntimeRunner);
|
|
66
77
|
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
67
78
|
}
|
|
68
79
|
export declare class EvalLoopHarness {
|
|
69
|
-
private
|
|
80
|
+
private runner;
|
|
70
81
|
private gate;
|
|
71
82
|
private maxAttempts;
|
|
72
|
-
constructor(
|
|
83
|
+
constructor(runner: RuntimeRunner, gate: QualityGate, maxAttempts?: number);
|
|
73
84
|
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
74
85
|
}
|
|
75
86
|
export interface HarnessLoopOptions {
|
|
@@ -77,10 +88,11 @@ export interface HarnessLoopOptions {
|
|
|
77
88
|
skillDir?: string;
|
|
78
89
|
}
|
|
79
90
|
export declare class HarnessLoop {
|
|
80
|
-
private
|
|
91
|
+
private runner;
|
|
81
92
|
private evalProvider;
|
|
82
93
|
private maxAttempts;
|
|
83
94
|
private skillDir?;
|
|
84
|
-
constructor(
|
|
95
|
+
constructor(runner: RuntimeRunner, evalProvider: import("../types.js").LLMProvider, options?: HarnessLoopOptions);
|
|
85
96
|
runStreaming(request: HarnessRequest): AsyncIterable<HarnessEvent>;
|
|
86
97
|
}
|
|
98
|
+
export { collectText };
|
package/dist/harness/harness.js
CHANGED
|
@@ -1,10 +1,12 @@
|
|
|
1
|
+
import { collectText } from "../runtime/runner.js";
|
|
1
2
|
import { writeFile } from "fs/promises";
|
|
2
3
|
import path from "path";
|
|
3
4
|
import { getKernel } from "../kernel.js";
|
|
4
|
-
async function runOnce(
|
|
5
|
+
async function runOnce(runner, req) {
|
|
5
6
|
let text = "";
|
|
6
7
|
let done;
|
|
7
|
-
|
|
8
|
+
const sessionId = crypto.randomUUID();
|
|
9
|
+
for await (const evt of runner.run({ sessionId, goal: req.goal, criteria: req.criteria?.map(c => c.text), extensions: req.extensions })) {
|
|
8
10
|
if (evt.type === "text_delta")
|
|
9
11
|
text += evt.delta;
|
|
10
12
|
else if (evt.type === "done")
|
|
@@ -13,27 +15,27 @@ async function runOnce(agent, req) {
|
|
|
13
15
|
return { result: text, passed: false, iterations: done?.iterations ?? 0, totalTokens: done?.totalTokens ?? 0, status: done?.status ?? "error" };
|
|
14
16
|
}
|
|
15
17
|
export class SinglePassHarness {
|
|
16
|
-
|
|
17
|
-
constructor(
|
|
18
|
-
this.
|
|
18
|
+
runner;
|
|
19
|
+
constructor(runner) {
|
|
20
|
+
this.runner = runner;
|
|
19
21
|
}
|
|
20
22
|
async run(request) {
|
|
21
|
-
return { ...await runOnce(this.
|
|
23
|
+
return { ...await runOnce(this.runner, request), passed: true };
|
|
22
24
|
}
|
|
23
25
|
}
|
|
24
26
|
export class EvalLoopHarness {
|
|
25
|
-
|
|
27
|
+
runner;
|
|
26
28
|
gate;
|
|
27
29
|
maxAttempts;
|
|
28
|
-
constructor(
|
|
29
|
-
this.
|
|
30
|
+
constructor(runner, gate, maxAttempts = 3) {
|
|
31
|
+
this.runner = runner;
|
|
30
32
|
this.gate = gate;
|
|
31
33
|
this.maxAttempts = maxAttempts;
|
|
32
34
|
}
|
|
33
35
|
async run(request) {
|
|
34
36
|
let outcome = { result: "", passed: false, iterations: 0, totalTokens: 0, status: "error" };
|
|
35
37
|
for (let i = 0; i < this.maxAttempts; i++) {
|
|
36
|
-
outcome = await runOnce(this.
|
|
38
|
+
outcome = await runOnce(this.runner, request);
|
|
37
39
|
if (await this.gate.evaluate(request, outcome))
|
|
38
40
|
return { ...outcome, passed: true };
|
|
39
41
|
}
|
|
@@ -41,12 +43,12 @@ export class EvalLoopHarness {
|
|
|
41
43
|
}
|
|
42
44
|
}
|
|
43
45
|
export class HarnessLoop {
|
|
44
|
-
|
|
46
|
+
runner;
|
|
45
47
|
evalProvider;
|
|
46
48
|
maxAttempts;
|
|
47
49
|
skillDir;
|
|
48
|
-
constructor(
|
|
49
|
-
this.
|
|
50
|
+
constructor(runner, evalProvider, options = {}) {
|
|
51
|
+
this.runner = runner;
|
|
50
52
|
this.evalProvider = evalProvider;
|
|
51
53
|
this.maxAttempts = options.maxAttempts ?? 3;
|
|
52
54
|
this.skillDir = options.skillDir;
|
|
@@ -55,27 +57,31 @@ export class HarnessLoop {
|
|
|
55
57
|
const kernel = getKernel();
|
|
56
58
|
const pipeline = new kernel.EvalPipeline({ extractSkillOnPass: true });
|
|
57
59
|
const criteria = request.criteria ?? [];
|
|
58
|
-
// history tracks the full conversation; agent sees only the current goal message
|
|
59
|
-
// but we inject feedback as user messages to guide revision without discarding prior output
|
|
60
60
|
let currentGoal = request.goal;
|
|
61
61
|
let lastIterations = 0;
|
|
62
62
|
let lastTotalTokens = 0;
|
|
63
63
|
let lastStatus = "error";
|
|
64
64
|
let lastResult = "";
|
|
65
65
|
for (let attempt = 1; attempt <= this.maxAttempts; attempt++) {
|
|
66
|
-
|
|
67
|
-
for await (const evt of this.
|
|
66
|
+
const sessionId = crypto.randomUUID();
|
|
67
|
+
for await (const evt of this.runner.run({ sessionId, goal: currentGoal, criteria: criteria.map(c => c.text), extensions: request.extensions })) {
|
|
68
68
|
if (evt.type === "text_delta") {
|
|
69
69
|
lastResult += evt.delta;
|
|
70
70
|
yield { type: "token", text: evt.delta };
|
|
71
71
|
}
|
|
72
72
|
else if (evt.type === "tool_call") {
|
|
73
|
-
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
74
73
|
const tc = evt;
|
|
75
74
|
yield { type: "tool_call", id: tc.id, name: tc.name };
|
|
76
75
|
}
|
|
76
|
+
else if (evt.type === "tool_delta") {
|
|
77
|
+
const td = evt;
|
|
78
|
+
yield { type: "tool_delta", callId: td.callId, ...(td.delta ? { delta: td.delta } : {}), ...(td.chunk ? { chunk: td.chunk } : {}) };
|
|
79
|
+
}
|
|
80
|
+
else if (evt.type === "tool_suspend") {
|
|
81
|
+
const ts = evt;
|
|
82
|
+
yield { type: "tool_suspend", callId: ts.callId, suspensionId: ts.suspensionId, ...(ts.payload ? { payload: ts.payload } : {}) };
|
|
83
|
+
}
|
|
77
84
|
else if (evt.type === "tool_result") {
|
|
78
|
-
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
79
85
|
const tr = evt;
|
|
80
86
|
yield { type: "tool_result", callId: tr.callId, content: tr.content, isError: tr.isError };
|
|
81
87
|
}
|
|
@@ -86,13 +92,11 @@ export class HarnessLoop {
|
|
|
86
92
|
lastStatus = d.status;
|
|
87
93
|
}
|
|
88
94
|
}
|
|
89
|
-
// Checkpoint: evaluator judges the completed output
|
|
90
95
|
yield { type: "supervising" };
|
|
91
96
|
const evalAction = pipeline.feedOutcome(request.goal, criteria, lastResult, attempt);
|
|
92
97
|
if (evalAction.kind !== "evaluate")
|
|
93
98
|
break;
|
|
94
99
|
let evalText = "";
|
|
95
|
-
// Wrap harness eval messages in a RenderedContext (system messages → systemText, rest → turns).
|
|
96
100
|
const evalMsgs = evalAction.messages ?? [];
|
|
97
101
|
const evalContext = {
|
|
98
102
|
systemText: evalMsgs.filter((m) => m.role === "system").map((m) => m.content).join("\n\n"),
|
|
@@ -123,7 +127,6 @@ export class HarnessLoop {
|
|
|
123
127
|
return;
|
|
124
128
|
}
|
|
125
129
|
yield { type: "revising", verdict };
|
|
126
|
-
// Inject feedback as next turn's goal — agent sees its prior output + evaluator notes
|
|
127
130
|
currentGoal = `${request.goal}\n\n[Attempt ${attempt} feedback: ${verdict.feedback}]`;
|
|
128
131
|
lastResult = "";
|
|
129
132
|
pipeline.reset();
|
|
@@ -131,3 +134,5 @@ export class HarnessLoop {
|
|
|
131
134
|
yield { type: "max_attempts_reached" };
|
|
132
135
|
}
|
|
133
136
|
}
|
|
137
|
+
// Re-export collectText so harness callers can use it without knowing runner internals.
|
|
138
|
+
export { collectText };
|