@deepstrike/sdk 0.2.4 → 0.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +287 -70
- package/dist/collaboration/modes/creator-verifier.d.ts +0 -3
- package/dist/collaboration/modes/creator-verifier.js +2 -6
- package/dist/governance.d.ts +36 -0
- package/dist/governance.js +22 -0
- package/dist/index.d.ts +11 -5
- package/dist/index.js +5 -1
- package/dist/memory/agent.d.ts +111 -0
- package/dist/memory/agent.js +151 -0
- package/dist/memory/protocols.d.ts +56 -0
- package/dist/runtime/execution-plane.d.ts +6 -9
- package/dist/runtime/execution-plane.js +53 -25
- package/dist/runtime/kernel-event-log.d.ts +26 -0
- package/dist/runtime/kernel-event-log.js +220 -0
- package/dist/runtime/kernel-primitives-dashboard.d.ts +44 -0
- package/dist/runtime/kernel-primitives-dashboard.js +135 -0
- package/dist/runtime/kernel-step.d.ts +25 -1
- package/dist/runtime/kernel-step.js +12 -3
- package/dist/runtime/large-result-spool.d.ts +84 -0
- package/dist/runtime/large-result-spool.js +167 -0
- package/dist/runtime/os-profile.d.ts +18 -0
- package/dist/runtime/os-profile.js +47 -0
- package/dist/runtime/os-snapshot.d.ts +35 -0
- package/dist/runtime/os-snapshot.js +128 -0
- package/dist/runtime/runner.d.ts +62 -9
- package/dist/runtime/runner.js +425 -135
- package/dist/runtime/session-log.d.ts +118 -4
- package/dist/runtime/session-log.js +15 -4
- package/dist/runtime/sub-agent-orchestrator.d.ts +2 -2
- package/dist/runtime/sub-agent-orchestrator.js +14 -23
- package/dist/types/agent.d.ts +12 -3
- package/dist/types/agent.js +18 -0
- package/dist/types.d.ts +22 -0
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# DeepStrike Node.js SDK
|
|
2
2
|
|
|
3
|
-
Runtime framework built on a Rust kernel. The kernel
|
|
3
|
+
Runtime framework built on a Rust kernel. The kernel owns loop control, context compression, governance, signal routing, and memory paging — the SDK owns all I/O (LLM calls, tool execution, disk, long-term memory).
|
|
4
|
+
|
|
5
|
+
Node.js is the reference SDK for the **Agent OS native profile**: declarative governance and in-kernel signal routing are enabled by default on every run.
|
|
4
6
|
|
|
5
7
|
## Install
|
|
6
8
|
|
|
@@ -24,9 +26,9 @@ Pre-built native addons are available for the following platforms:
|
|
|
24
26
|
| Linux ARM64 (musl / Alpine) | `@deepstrike/core-linux-arm64-musl` |
|
|
25
27
|
| Windows x64 | `@deepstrike/core-win32-x64-msvc` |
|
|
26
28
|
|
|
27
|
-
The correct platform package is selected
|
|
29
|
+
The correct platform package is selected automatically via `optionalDependencies`.
|
|
28
30
|
|
|
29
|
-
> **Note:** `@deepstrike/core` is the low-level
|
|
31
|
+
> **Note:** `@deepstrike/core` is the low-level N-API binding and is managed as an internal dependency of `@deepstrike/sdk`. When developing against a local kernel build, run `npm run test:local-core` from this directory to rebuild the native module from `../crates/deepstrike-node`.
|
|
30
32
|
|
|
31
33
|
---
|
|
32
34
|
|
|
@@ -65,14 +67,14 @@ const result = await collectText(runner.run({
|
|
|
65
67
|
console.log(result)
|
|
66
68
|
```
|
|
67
69
|
|
|
68
|
-
Same-session
|
|
70
|
+
Same-session continuity is explicit via `sessionId`:
|
|
69
71
|
|
|
70
72
|
```typescript
|
|
71
73
|
await collectText(runner.run({ sessionId: "chat-1", goal: "My name is Ada." }))
|
|
72
74
|
const reply = await collectText(runner.run({ sessionId: "chat-1", goal: "What is my name?" }))
|
|
73
75
|
```
|
|
74
76
|
|
|
75
|
-
Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when
|
|
77
|
+
Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when replay should survive restarts. `wake(sessionId)` resumes from the event log without inserting a duplicate `run_started` event.
|
|
76
78
|
|
|
77
79
|
Streaming:
|
|
78
80
|
|
|
@@ -87,6 +89,62 @@ for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize R
|
|
|
87
89
|
|
|
88
90
|
---
|
|
89
91
|
|
|
92
|
+
## Architecture
|
|
93
|
+
|
|
94
|
+
```text
|
|
95
|
+
┌─────────────────────────────────────────────────────────┐
|
|
96
|
+
│ RuntimeRunner (Layer 1.5) │
|
|
97
|
+
│ LLMProvider · ExecutionPlane · SessionLog · DreamStore │
|
|
98
|
+
└───────────────────────────┬─────────────────────────────┘
|
|
99
|
+
│ step(JSON event) ↔ actions / observations
|
|
100
|
+
┌───────────────────────────▼─────────────────────────────┐
|
|
101
|
+
│ @deepstrike/core KernelRuntime │
|
|
102
|
+
│ P1 Syscall · P2 Sched · P3 MM · Proc · IPC │
|
|
103
|
+
└─────────────────────────────────────────────────────────┘
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
The runner drives a single loop:
|
|
107
|
+
|
|
108
|
+
1. Kernel returns an **action** — `call_provider`, `execute_tool`, `evaluate_milestone`, or `done`.
|
|
109
|
+
2. SDK executes the action (stream LLM, run tools, call milestone verifier).
|
|
110
|
+
3. SDK feeds the result back as a kernel **event** (`provider_result`, `tool_results`, …).
|
|
111
|
+
4. Kernel **observations** (compression, page-out, spool, signals, …) are drained into `SessionLog`.
|
|
112
|
+
|
|
113
|
+
Kernel session events carry an optional `category` tag (`syscall` · `sched` · `mm` · `proc` · `ipc`) for diagnostics and OS snapshot rebuilds.
|
|
114
|
+
|
|
115
|
+
### What Agent OS gives you
|
|
116
|
+
|
|
117
|
+
The mechanisms above are not internal refactors — they change what you can build without custom runner code:
|
|
118
|
+
|
|
119
|
+
**Kernel-mediated runtime (M0–M4)**
|
|
120
|
+
Tool calls, spawns, compression, and signals pass through one kernel gate with an explicit lifecycle (Ready / Running / Blocked / Suspended). You implement I/O; the kernel decides *when* and *whether*. Node, Python, and Rust share the same decision path, so `wake(sessionId)` and cross-language tooling see consistent behavior.
|
|
121
|
+
|
|
122
|
+
**Longer, sturdier sessions (Layer-1 spool + semantic page-out)**
|
|
123
|
+
Oversized tool results (> 50 KB) stay in context as a preview plus a `.spool/` reference — the model reads the full payload on demand via ordinary file tools. When pressure triggers semantic eviction, the SDK summarizes archived content into `DreamStore` and satisfies `page_in_requested` on the way back in. Long tasks survive token pressure instead of failing mid-run.
|
|
124
|
+
|
|
125
|
+
**Safety and governance by default (OS native profile)**
|
|
126
|
+
Every run loads declarative `governancePolicy` (deny / ask_user / rate-limit / param rules) and in-kernel signal routing (`attentionPolicy`, default queue 64). Dangerous tools, external interrupts, and approval flows are policy — not ad-hoc `if` checks in your handlers.
|
|
127
|
+
|
|
128
|
+
**Long-term memory as syscalls (Phase-7)**
|
|
129
|
+
`writeMemory` and `queryMemory` run outside the main tool loop: kernel validation before `DreamStore.commit`, search → `selectMemories` → `memory_retrieval_result` on query. Failed writes emit `memory_validation_failed` for audit; good memory is durable without polluting history.
|
|
130
|
+
|
|
131
|
+
**Multi-agent and multi-signal orchestration**
|
|
132
|
+
Sub-agents register in the kernel process table (`agent_process_changed`); parent runs suspend explicitly until `sub_agent_completed`. Signals get disposition (Interrupt / Queue / Observe / Dropped) in-kernel, so gateways, cron, and heartbeats compose with the main loop instead of racing it.
|
|
133
|
+
|
|
134
|
+
**Observable like an OS log**
|
|
135
|
+
Spool, page-out, signals, processes, budgets, and memory events land in `SessionLog` with categories. Rebuild an OS snapshot (`pageOutCount`, `spoolCount`, `processByAgent`, memory counters) from one event stream — replay still strips audit events when reconstructing LLM messages.
|
|
136
|
+
|
|
137
|
+
| You need… | Use… |
|
|
138
|
+
|---|---|
|
|
139
|
+
| Policy before tools run | `governancePolicy` (default: allow-all native profile) |
|
|
140
|
+
| External interrupts | `signalSource` + in-kernel `attentionPolicy` |
|
|
141
|
+
| Huge tool output | Automatic Layer-1 spool; optional custom `resultSpool` |
|
|
142
|
+
| Durable recall across runs | `DreamStore` + semantic `page_out` via `dreamSummarizer` |
|
|
143
|
+
| Programmatic memory I/O | `runner.writeMemory()` / `runner.queryMemory()` |
|
|
144
|
+
| Debug / compliance | `SessionLog` events + OS snapshot helpers |
|
|
145
|
+
|
|
146
|
+
---
|
|
147
|
+
|
|
90
148
|
## Providers
|
|
91
149
|
|
|
92
150
|
| Class | Backend | Notes |
|
|
@@ -103,7 +161,7 @@ for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize R
|
|
|
103
161
|
|
|
104
162
|
All providers accept `RetryConfig` for exponential backoff and share a `CircuitBreaker`.
|
|
105
163
|
|
|
106
|
-
`extensions` are forwarded by every provider in both `complete()` and `stream()` while SDK-owned structural fields such as `model`, `messages`, `tools`, and streaming flags remain protected.
|
|
164
|
+
`extensions` are forwarded by every provider in both `complete()` and `stream()` while SDK-owned structural fields such as `model`, `messages`, `tools`, and streaming flags remain protected.
|
|
107
165
|
|
|
108
166
|
OpenAI can also be selected through the provider catalog:
|
|
109
167
|
|
|
@@ -138,39 +196,101 @@ const runner = new RuntimeRunner({
|
|
|
138
196
|
```
|
|
139
197
|
|
|
140
198
|
- `memory(query)` / `knowledge(query)` meta-tool results → **history** (tool results)
|
|
141
|
-
-
|
|
199
|
+
- Inbound signals are routed by the in-kernel attention policy and rendered into **Slot 3**
|
|
142
200
|
- Anthropic: Slots 1–2 get separate `cache_control` breakpoints
|
|
143
201
|
|
|
144
|
-
Full reference: [docs/context-
|
|
202
|
+
Full reference: [docs/concepts/context-slots-compression.md](../docs/concepts/context-slots-compression.md)
|
|
145
203
|
|
|
146
204
|
---
|
|
147
205
|
|
|
148
206
|
## Runtime options
|
|
149
207
|
|
|
150
208
|
```typescript
|
|
151
|
-
|
|
209
|
+
import {
|
|
210
|
+
DEFAULT_NATIVE_GOVERNANCE_POLICY,
|
|
211
|
+
DEFAULT_NATIVE_ATTENTION_POLICY,
|
|
212
|
+
} from "@deepstrike/sdk"
|
|
213
|
+
|
|
152
214
|
const runner = new RuntimeRunner({
|
|
153
215
|
provider,
|
|
154
216
|
executionPlane: plane,
|
|
155
217
|
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
218
|
+
|
|
219
|
+
// Scheduler budget
|
|
220
|
+
maxTokens: 128_000,
|
|
221
|
+
maxTurns: 25,
|
|
222
|
+
timeoutMs: 60_000,
|
|
223
|
+
schedulerBudget: { maxWallMs: 300_000 },
|
|
224
|
+
|
|
225
|
+
// Agent OS native profile (defaults shown)
|
|
226
|
+
governancePolicy: DEFAULT_NATIVE_GOVERNANCE_POLICY,
|
|
227
|
+
attentionPolicy: DEFAULT_NATIVE_ATTENTION_POLICY, // SignalRouter queue size 64
|
|
228
|
+
|
|
229
|
+
// Host I/O
|
|
230
|
+
extensions: { temperature: 0.1 },
|
|
231
|
+
skillDir: "./skills",
|
|
232
|
+
knowledgeSource: myKS,
|
|
233
|
+
signalSource: gw,
|
|
234
|
+
dreamStore: myStore,
|
|
235
|
+
agentId: "my-agent",
|
|
236
|
+
initialMemory: ["..."],
|
|
237
|
+
|
|
238
|
+
// Memory paging & compression (SDK-side I/O)
|
|
239
|
+
compressionStore: archiveStore, // persist compressed transcript slices
|
|
240
|
+
asyncSummarizer: mySummarizer, // upgrade rule-based compression summaries
|
|
241
|
+
dreamProvider: dreamLlm, // LLM for idle dream() synthesis
|
|
242
|
+
dreamSummarizer: myDreamSummarizer, // LLM for semantic page_out → DreamStore
|
|
243
|
+
|
|
244
|
+
// Sub-agents
|
|
245
|
+
runSpec: { role: "orchestrator", isolation: "process" },
|
|
246
|
+
milestoneContract: myContract,
|
|
247
|
+
milestonePolicy: "require_verifier",
|
|
248
|
+
onMilestoneEvaluate: async ({ phaseId, criteria }) => ({ passed: true, phaseId }),
|
|
249
|
+
subAgentHarness: { evalProvider, maxAttempts: 3 },
|
|
250
|
+
|
|
251
|
+
// Governance UX (AskUser path)
|
|
252
|
+
onPermissionRequest: async (req) => ({ approved: true }),
|
|
253
|
+
|
|
254
|
+
// Diagnostics
|
|
255
|
+
enableDiagnosticsDashboard: true, // CLI view grouped by Syscall / Sched / MM
|
|
171
256
|
})
|
|
172
257
|
```
|
|
173
258
|
|
|
259
|
+
| Option | Purpose |
|
|
260
|
+
|--------|---------|
|
|
261
|
+
| `governancePolicy` | Declarative deny / ask_user / rate-limit / param rules loaded into the kernel before `start_run` |
|
|
262
|
+
| `attentionPolicy` | In-kernel signal router queue size (default 64) |
|
|
263
|
+
| `onPermissionRequest` | Resolves `tool_gated` + `suspended` → kernel `resume` with approved/denied call IDs |
|
|
264
|
+
| `compressionStore` | Writes archived messages on `compressed` observations |
|
|
265
|
+
| `asyncSummarizer` | Background LLM summary after compression; stored as `summary_upgraded` |
|
|
266
|
+
| `dreamSummarizer` | Summarizes `page_out { tier_hint: "semantic" }` into `DreamStore` during a run |
|
|
267
|
+
| `dreamProvider` | Separate LLM for `dream()` idle consolidation (falls back to `provider`) |
|
|
268
|
+
|
|
269
|
+
Rebuild an OS diagnostics snapshot from session events:
|
|
270
|
+
|
|
271
|
+
```typescript
|
|
272
|
+
import { rebuildOsSnapshotFromSessionEvents } from "@deepstrike/sdk"
|
|
273
|
+
|
|
274
|
+
const events = (await sessionLog.read(sessionId)).map(e => e.event)
|
|
275
|
+
const snap = rebuildOsSnapshotFromSessionEvents(events)
|
|
276
|
+
// snap.pageOutCount, snap.spoolCount, snap.signals, snap.processByAgent, …
|
|
277
|
+
```
|
|
278
|
+
|
|
279
|
+
---
|
|
280
|
+
|
|
281
|
+
## Large result spool (Layer 1)
|
|
282
|
+
|
|
283
|
+
When a single tool result exceeds **50 KB**, the kernel keeps a short preview in context and emits `large_result_spooled`. The SDK writes the full payload to `.spool/` under the process cwd (SHA-256 keyed files) and logs `spool_ref` in the session.
|
|
284
|
+
|
|
285
|
+
The model can retrieve full content via ordinary read tools — `LocalExecutionPlane` transparently resolves paths under `.spool/`:
|
|
286
|
+
|
|
287
|
+
```typescript
|
|
288
|
+
// Kernel context shows a preview + spool reference.
|
|
289
|
+
// LLM calls read_file({ path: ".spool/abc123…" }) → full content returned.
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
No configuration is required; customize the directory by passing a `resultSpool` instance when constructing `RuntimeRunner` (see tests under `tests/runtime/large-result-spool.test.ts`).
|
|
293
|
+
|
|
174
294
|
---
|
|
175
295
|
|
|
176
296
|
## Tools
|
|
@@ -179,10 +299,28 @@ const runner = new RuntimeRunner({
|
|
|
179
299
|
import { tool, readFile } from "@deepstrike/sdk"
|
|
180
300
|
|
|
181
301
|
plane.register(tool("search", "Search.", schema, async (args) => ...))
|
|
182
|
-
plane.register(readFile) // built-in: read files from disk
|
|
302
|
+
plane.register(readFile) // built-in: read files from disk (also resolves .spool/ refs)
|
|
183
303
|
plane.unregister("search")
|
|
184
304
|
```
|
|
185
305
|
|
|
306
|
+
Execution planes:
|
|
307
|
+
|
|
308
|
+
| Plane | Use case |
|
|
309
|
+
|-------|----------|
|
|
310
|
+
| `LocalExecutionPlane` | In-process tools (default) |
|
|
311
|
+
| `FilteredExecutionPlane` | Capability-filtered sub-agent tools |
|
|
312
|
+
| `ProcessSandboxPlane` | OS subprocess isolation |
|
|
313
|
+
| `McpProxyPlane` | MCP server tools |
|
|
314
|
+
| `RemoteVpcPlane` | Remote execution |
|
|
315
|
+
|
|
316
|
+
Mount capabilities on an active run:
|
|
317
|
+
|
|
318
|
+
```typescript
|
|
319
|
+
runner.mountTool(schema)
|
|
320
|
+
runner.mountSkill("summarize", "Summarize text")
|
|
321
|
+
runner.unmountCapability("tool", "search")
|
|
322
|
+
```
|
|
323
|
+
|
|
186
324
|
---
|
|
187
325
|
|
|
188
326
|
## Skills
|
|
@@ -214,9 +352,11 @@ effort: 1
|
|
|
214
352
|
|
|
215
353
|
## Knowledge
|
|
216
354
|
|
|
217
|
-
Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand.
|
|
355
|
+
Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand. Runtime retrieval results land in **history** as tool results.
|
|
218
356
|
|
|
219
|
-
To inject durable knowledge at startup (Slot 2, cacheable on Anthropic), use `initialMemory` or
|
|
357
|
+
To inject durable knowledge at startup (Slot 2, cacheable on Anthropic), use `initialMemory` or `runner.pushKnowledge()`.
|
|
358
|
+
|
|
359
|
+
Before tool execution the kernel may emit `page_in_requested`; the SDK satisfies it from `DreamStore`, `KnowledgeSource`, and a local semantic page-out cache, then feeds `page_in` back to the kernel.
|
|
220
360
|
|
|
221
361
|
```typescript
|
|
222
362
|
const runner = new RuntimeRunner({
|
|
@@ -238,7 +378,7 @@ const runner = new RuntimeRunner({
|
|
|
238
378
|
|
|
239
379
|
### WorkingMemory (SDK-side scratch pad)
|
|
240
380
|
|
|
241
|
-
`WorkingMemory` is an SDK helper — not the kernel
|
|
381
|
+
`WorkingMemory` is an SDK helper — not the kernel working partition. Kernel task state lives in `task_state` and renders into Slot 3 (`turns[0]`).
|
|
242
382
|
|
|
243
383
|
```typescript
|
|
244
384
|
import { WorkingMemory } from "@deepstrike/sdk"
|
|
@@ -248,7 +388,7 @@ mem.get("step") // 1
|
|
|
248
388
|
mem.clear()
|
|
249
389
|
```
|
|
250
390
|
|
|
251
|
-
### DreamStore (long-term memory
|
|
391
|
+
### DreamStore (long-term memory)
|
|
252
392
|
|
|
253
393
|
```typescript
|
|
254
394
|
import type { DreamStore } from "@deepstrike/sdk"
|
|
@@ -266,32 +406,91 @@ const runner = new RuntimeRunner({
|
|
|
266
406
|
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
267
407
|
maxTokens: 4096,
|
|
268
408
|
dreamStore: new MyStore(),
|
|
269
|
-
agentId: "my-agent", // enables `memory` meta-tool
|
|
409
|
+
agentId: "my-agent", // enables `memory` meta-tool + semantic page-out archival
|
|
270
410
|
})
|
|
411
|
+
```
|
|
412
|
+
|
|
413
|
+
Three memory paths:
|
|
271
414
|
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
415
|
+
| Path | When | What happens |
|
|
416
|
+
|------|------|--------------|
|
|
417
|
+
| In-session `memory(query)` | LLM calls meta-tool | `DreamStore.search()` → history tool result |
|
|
418
|
+
| `initialMemory` | Run start | Injected into Slot 2 (`systemKnowledge`) |
|
|
419
|
+
| Semantic `page_out` | Kernel evicts with `tier_hint: "semantic"` | SDK summarizes via `dreamSummarizer` / `dreamProvider` → `DreamStore.commit()` |
|
|
420
|
+
| `dream(agentId)` | Explicit idle call | `IdlePipeline` batch-consolidates past sessions |
|
|
421
|
+
|
|
422
|
+
```typescript
|
|
423
|
+
// Post-session batch consolidation
|
|
275
424
|
const result = await runner.dream("my-agent", Date.now())
|
|
276
425
|
```
|
|
277
426
|
|
|
427
|
+
### Phase-7 memory syscalls (`writeMemory` / `queryMemory`)
|
|
428
|
+
|
|
429
|
+
Kernel-validated long-term memory I/O outside the main tool loop:
|
|
430
|
+
|
|
431
|
+
```typescript
|
|
432
|
+
await runner.writeMemory({
|
|
433
|
+
metadata: {
|
|
434
|
+
name: "prefers-small-tests",
|
|
435
|
+
description: "User prefers focused unit tests",
|
|
436
|
+
kind: "feedback",
|
|
437
|
+
created_at: Date.now(),
|
|
438
|
+
updated_at: Date.now(),
|
|
439
|
+
},
|
|
440
|
+
content: "User prefers focused unit tests for SDK behavior.",
|
|
441
|
+
}, { sessionId: "my-session" })
|
|
442
|
+
|
|
443
|
+
const hits = await runner.queryMemory({
|
|
444
|
+
current_context: "Need testing preferences",
|
|
445
|
+
active_tools: [],
|
|
446
|
+
already_surfaced: [],
|
|
447
|
+
top_k: 5,
|
|
448
|
+
}, { sessionId: "my-session" })
|
|
449
|
+
```
|
|
450
|
+
|
|
451
|
+
Session events: `memory_written`, `memory_queried`, `memory_validation_failed`, `memory_retrieval_result`.
|
|
452
|
+
|
|
278
453
|
---
|
|
279
454
|
|
|
280
455
|
## Governance
|
|
281
456
|
|
|
282
|
-
###
|
|
457
|
+
### In-kernel declarative policy (preferred)
|
|
458
|
+
|
|
459
|
+
Every run loads `governancePolicy` into the kernel via `load_governance_policy`. The kernel enforces rules **before** tools execute:
|
|
283
460
|
|
|
284
461
|
```typescript
|
|
285
|
-
import {
|
|
462
|
+
import type { GovernancePolicy } from "@deepstrike/sdk"
|
|
463
|
+
|
|
464
|
+
const policy: GovernancePolicy = {
|
|
465
|
+
rules: [
|
|
466
|
+
{ pattern: "read_file", action: "allow" },
|
|
467
|
+
{ pattern: "write_file", action: "ask_user" },
|
|
468
|
+
{ pattern: "run_command", action: "ask_user" },
|
|
469
|
+
{ pattern: "*", action: "deny" },
|
|
470
|
+
],
|
|
471
|
+
rateLimits: [{ tool: "api_call", maxCalls: 10, windowMs: 60_000 }],
|
|
472
|
+
}
|
|
286
473
|
|
|
287
|
-
const
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
474
|
+
const runner = new RuntimeRunner({
|
|
475
|
+
provider,
|
|
476
|
+
executionPlane: plane,
|
|
477
|
+
sessionLog,
|
|
478
|
+
governancePolicy: policy,
|
|
479
|
+
onPermissionRequest: async (req) => {
|
|
480
|
+
console.log(`Approve ${req.toolName}?`, req.arguments)
|
|
481
|
+
return { approved: true }
|
|
482
|
+
},
|
|
483
|
+
})
|
|
292
484
|
```
|
|
293
485
|
|
|
294
|
-
|
|
486
|
+
- `deny` → tool rejected with `tool_denied`
|
|
487
|
+
- `ask_user` → `tool_gated` + `suspended`; resolve via `onPermissionRequest`, then kernel `resume`
|
|
488
|
+
|
|
489
|
+
Default when omitted: allow-all (`DEFAULT_NATIVE_GOVERNANCE_POLICY`).
|
|
490
|
+
|
|
491
|
+
### Standalone Governance class
|
|
492
|
+
|
|
493
|
+
`Governance` wraps the native governance evaluator for SDK-side use (tests, custom gates). It is **not** wired automatically into `RuntimeRunner` — use `governancePolicy` for run-time enforcement.
|
|
295
494
|
|
|
296
495
|
```typescript
|
|
297
496
|
import { Governance } from "@deepstrike/sdk"
|
|
@@ -299,45 +498,65 @@ import { Governance } from "@deepstrike/sdk"
|
|
|
299
498
|
const gov = new Governance("allow")
|
|
300
499
|
gov.addPermissionRule("danger.*", "deny")
|
|
301
500
|
gov.blockTool("rm_rf")
|
|
302
|
-
gov.
|
|
303
|
-
gov.requireParam("write_file", "path")
|
|
304
|
-
gov.allowParamValues("set_mode", "mode", ["read", "write"])
|
|
305
|
-
gov.limitParamRange("sleep", "seconds", 0, 10)
|
|
306
|
-
|
|
307
|
-
const runner = new RuntimeRunner({
|
|
308
|
-
provider,
|
|
309
|
-
executionPlane: plane,
|
|
310
|
-
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
311
|
-
maxTokens: 4096,
|
|
312
|
-
governance: gov,
|
|
313
|
-
})
|
|
314
|
-
// Every tool call goes through: Permission → Veto → RateLimit → Constraint → Audit
|
|
501
|
+
gov.evaluate("read_file", '{"path":"x"}')
|
|
315
502
|
```
|
|
316
503
|
|
|
504
|
+
### SDK PermissionManager
|
|
505
|
+
|
|
506
|
+
`PermissionManager` is a separate SDK-side permission layer for apps that manage their own approval UX outside the kernel loop.
|
|
507
|
+
|
|
317
508
|
---
|
|
318
509
|
|
|
319
510
|
## Signals
|
|
320
511
|
|
|
512
|
+
Inbound signals are routed by the in-kernel attention policy (default queue size 64):
|
|
513
|
+
|
|
514
|
+
| Urgency | Typical disposition |
|
|
515
|
+
|---------|-------------------|
|
|
516
|
+
| `critical` / `high` | `interrupt_now` — may yield a new `call_provider` action |
|
|
517
|
+
| `normal` / `low` | `queue` — buffered; no action until dequeued |
|
|
518
|
+
| queue full | `dropped` |
|
|
519
|
+
|
|
321
520
|
```typescript
|
|
322
521
|
import { SignalGateway, ScheduledPrompt } from "@deepstrike/sdk"
|
|
323
522
|
|
|
324
523
|
const gw = new SignalGateway()
|
|
325
524
|
gw.schedule(new ScheduledPrompt("standup", Date.now() + 3600_000))
|
|
326
|
-
gw.ingest({ kind: "
|
|
525
|
+
gw.ingest({ kind: "alert", urgency: "normal", payload: { goal: "Check deploy" } })
|
|
327
526
|
|
|
328
527
|
const runner = new RuntimeRunner({
|
|
329
528
|
provider,
|
|
330
529
|
executionPlane: plane,
|
|
331
|
-
sessionLog
|
|
332
|
-
maxTokens: 4096,
|
|
530
|
+
sessionLog,
|
|
333
531
|
signalSource: gw,
|
|
532
|
+
attentionPolicy: { maxQueueSize: 64 },
|
|
334
533
|
})
|
|
335
|
-
// kind="interrupt" → immediately stops the running runner
|
|
336
534
|
|
|
337
|
-
runner.interrupt() //
|
|
535
|
+
runner.interrupt() // cooperative abort → kernel timeout path
|
|
338
536
|
gw.destroy()
|
|
339
537
|
```
|
|
340
538
|
|
|
539
|
+
Each routed signal produces a `signal_disposed` session event (`category: "ipc"`).
|
|
540
|
+
|
|
541
|
+
---
|
|
542
|
+
|
|
543
|
+
## Sub-agents
|
|
544
|
+
|
|
545
|
+
Spawn isolated child agents through the kernel process table:
|
|
546
|
+
|
|
547
|
+
```typescript
|
|
548
|
+
for await (const evt of runner.spawnSubAgent({
|
|
549
|
+
role: "researcher",
|
|
550
|
+
isolation: "process",
|
|
551
|
+
goal: "Find three sources on topic X",
|
|
552
|
+
criteria: ["At least 3 URLs"],
|
|
553
|
+
})) {
|
|
554
|
+
if (evt.type === "done") console.log(evt.status)
|
|
555
|
+
}
|
|
556
|
+
```
|
|
557
|
+
|
|
558
|
+
Requires an active parent run (`run()` / `wake()` in progress). The kernel emits `agent_process_changed`; the default `SubAgentOrchestrator` runs the child with a filtered execution plane and feeds `sub_agent_completed` back.
|
|
559
|
+
|
|
341
560
|
---
|
|
342
561
|
|
|
343
562
|
## Harness (evaluation framework)
|
|
@@ -345,30 +564,20 @@ gw.destroy()
|
|
|
345
564
|
```typescript
|
|
346
565
|
import { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "@deepstrike/sdk"
|
|
347
566
|
|
|
348
|
-
// 1. SinglePass — run once, always passes
|
|
349
567
|
const outcome = await new SinglePassHarness(runner).run({ goal: "Say hello" })
|
|
350
568
|
|
|
351
|
-
// 2. EvalLoop — retry until QualityGate passes
|
|
352
569
|
const harness = new EvalLoopHarness(runner, {
|
|
353
570
|
async evaluate(_req, out) { return out.result.includes("hello") },
|
|
354
571
|
}, 3)
|
|
355
572
|
|
|
356
|
-
// 3. HarnessLoop — LLM-as-judge with feedback injection + skill extraction
|
|
357
573
|
const loop = new HarnessLoop(runner, evalProvider, { maxAttempts: 3, skillDir: "./skills" })
|
|
358
574
|
|
|
359
|
-
// Sub-agents: pass subAgentHarness on RuntimeRunner to auto-evaluate spawned children
|
|
360
575
|
const runnerWithHarness = new RuntimeRunner({
|
|
361
576
|
provider,
|
|
362
577
|
executionPlane: plane,
|
|
363
578
|
sessionLog,
|
|
364
579
|
subAgentHarness: { evalProvider, maxAttempts: 3 },
|
|
365
580
|
})
|
|
366
|
-
for await (const event of loop.runStreaming({
|
|
367
|
-
goal: "Write a haiku",
|
|
368
|
-
criteria: [{ text: "Must be 3 lines", required: true }],
|
|
369
|
-
})) {
|
|
370
|
-
if (event.type === "done") console.log(event.verdict.passed, event.verdict.feedback)
|
|
371
|
-
}
|
|
372
581
|
```
|
|
373
582
|
|
|
374
583
|
---
|
|
@@ -387,4 +596,12 @@ for await (const event of loop.runStreaming({
|
|
|
387
596
|
| `done` | `iterations`, `totalTokens`, `status` |
|
|
388
597
|
| `error` | `message` |
|
|
389
598
|
|
|
390
|
-
`status`: `completed` · `max_turns` · `token_budget` · `timeout` · `user_abort` · `error`
|
|
599
|
+
`status`: `completed` · `max_turns` · `token_budget` · `timeout` · `user_abort` · `error` · `milestone_pending`
|
|
600
|
+
|
|
601
|
+
---
|
|
602
|
+
|
|
603
|
+
## Further reading
|
|
604
|
+
|
|
605
|
+
- [SDK OS parity matrix](../docs/sdk-os-parity.md)
|
|
606
|
+
- [Kernel ABI reference](../docs/reference/kernel-abi.md)
|
|
607
|
+
- [Context slots & compression](../docs/concepts/context-slots-compression.md)
|
|
@@ -35,8 +35,6 @@ export declare class CreatorVerifierMode {
|
|
|
35
35
|
maxAttempts?: number;
|
|
36
36
|
/** Stable orchestration session for kernel lineage audit. */
|
|
37
37
|
coordinatorSessionId?: string;
|
|
38
|
-
/** Opt out of kernel spawn path and use legacy independent runner sessions. */
|
|
39
|
-
useLegacyRunners?: boolean;
|
|
40
38
|
});
|
|
41
39
|
run(contract: VerificationContract): Promise<ContractOutcome>;
|
|
42
40
|
/** Aggregate drift metrics across all runs through this mode instance. */
|
|
@@ -64,7 +62,6 @@ export declare class OrchestrationMode {
|
|
|
64
62
|
constructor(pool: AgentPool, options?: {
|
|
65
63
|
maxAttempts?: number;
|
|
66
64
|
coordinatorSessionId?: string;
|
|
67
|
-
useLegacyRunners?: boolean;
|
|
68
65
|
});
|
|
69
66
|
run(goal: string): Promise<ContractOutcome & {
|
|
70
67
|
contract: VerificationContract;
|
|
@@ -30,9 +30,7 @@ export class CreatorVerifierMode {
|
|
|
30
30
|
}
|
|
31
31
|
async run(contract) {
|
|
32
32
|
this._total++;
|
|
33
|
-
|
|
34
|
-
this.pool.ensureCoordinator(this.options.coordinatorSessionId);
|
|
35
|
-
}
|
|
33
|
+
this.pool.ensureCoordinator(this.options.coordinatorSessionId);
|
|
36
34
|
const harness = new ContractDrivenHarness(this.pool, contract, {
|
|
37
35
|
maxAttempts: this.options.maxAttempts ?? 3,
|
|
38
36
|
});
|
|
@@ -81,9 +79,7 @@ export class OrchestrationMode {
|
|
|
81
79
|
this.inner = new CreatorVerifierMode(pool, options);
|
|
82
80
|
}
|
|
83
81
|
async run(goal) {
|
|
84
|
-
|
|
85
|
-
this.pool.ensureCoordinator(this.options.coordinatorSessionId);
|
|
86
|
-
}
|
|
82
|
+
this.pool.ensureCoordinator(this.options.coordinatorSessionId);
|
|
87
83
|
// Step 1: orchestrator produces a VerificationContract
|
|
88
84
|
const contractJson = await this.pool.orchestrate(goal);
|
|
89
85
|
const contract = this._parseContract(contractJson, goal);
|
package/dist/governance.d.ts
CHANGED
|
@@ -15,3 +15,39 @@ export declare class Governance {
|
|
|
15
15
|
evaluate(toolName: string, argsJson: string): GovernanceVerdict;
|
|
16
16
|
}
|
|
17
17
|
export type { GovernanceVerdict };
|
|
18
|
+
type GovernancePolicyAction = "allow" | "deny" | "ask_user";
|
|
19
|
+
export interface GovernancePolicy {
|
|
20
|
+
defaultAction?: GovernancePolicyAction;
|
|
21
|
+
rules?: {
|
|
22
|
+
pattern: string;
|
|
23
|
+
action: GovernancePolicyAction;
|
|
24
|
+
}[];
|
|
25
|
+
vetoes?: string[];
|
|
26
|
+
rateLimits?: {
|
|
27
|
+
tool: string;
|
|
28
|
+
maxCalls: number;
|
|
29
|
+
windowMs: number;
|
|
30
|
+
}[];
|
|
31
|
+
constraints?: GovernanceConstraint[];
|
|
32
|
+
}
|
|
33
|
+
export type GovernanceConstraint = {
|
|
34
|
+
kind: "required";
|
|
35
|
+
tool: string;
|
|
36
|
+
path: string;
|
|
37
|
+
} | {
|
|
38
|
+
kind: "enum";
|
|
39
|
+
tool: string;
|
|
40
|
+
path: string;
|
|
41
|
+
values: string[];
|
|
42
|
+
} | {
|
|
43
|
+
kind: "range";
|
|
44
|
+
tool: string;
|
|
45
|
+
path: string;
|
|
46
|
+
min?: number;
|
|
47
|
+
max?: number;
|
|
48
|
+
};
|
|
49
|
+
/**
|
|
50
|
+
* Convert a declarative {@link GovernancePolicy} into the `load_governance_policy`
|
|
51
|
+
* kernel event payload (snake_case wire fields). Pure — no side effects.
|
|
52
|
+
*/
|
|
53
|
+
export declare function governancePolicyToKernelEvent(policy: GovernancePolicy): Record<string, unknown>;
|
package/dist/governance.js
CHANGED
|
@@ -32,3 +32,25 @@ export class Governance {
|
|
|
32
32
|
return this.inner.evaluate(toolName, argsJson);
|
|
33
33
|
}
|
|
34
34
|
}
|
|
35
|
+
/**
|
|
36
|
+
* Convert a declarative {@link GovernancePolicy} into the `load_governance_policy`
|
|
37
|
+
* kernel event payload (snake_case wire fields). Pure — no side effects.
|
|
38
|
+
*/
|
|
39
|
+
export function governancePolicyToKernelEvent(policy) {
|
|
40
|
+
return {
|
|
41
|
+
kind: "load_governance_policy",
|
|
42
|
+
...(policy.defaultAction ? { default_action: policy.defaultAction } : {}),
|
|
43
|
+
rules: (policy.rules ?? []).map(r => ({ tool_pattern: r.pattern, action: r.action })),
|
|
44
|
+
vetoed_tools: policy.vetoes ?? [],
|
|
45
|
+
rate_limits: (policy.rateLimits ?? []).map(rl => ({
|
|
46
|
+
tool: rl.tool,
|
|
47
|
+
max_calls: rl.maxCalls,
|
|
48
|
+
window_ms: rl.windowMs,
|
|
49
|
+
})),
|
|
50
|
+
constraints: (policy.constraints ?? []).map(c => c.kind === "enum"
|
|
51
|
+
? { kind: "enum", tool: c.tool, path: c.path, values: c.values }
|
|
52
|
+
: c.kind === "range"
|
|
53
|
+
? { kind: "range", tool: c.tool, path: c.path, ...(c.min !== undefined ? { min: c.min } : {}), ...(c.max !== undefined ? { max: c.max } : {}) }
|
|
54
|
+
: { kind: "required", tool: c.tool, path: c.path }),
|
|
55
|
+
};
|
|
56
|
+
}
|