@deepstrike/sdk 0.2.4 → 0.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +305 -69
- package/dist/collaboration/modes/creator-verifier.d.ts +0 -3
- package/dist/collaboration/modes/creator-verifier.js +2 -6
- package/dist/governance.d.ts +36 -0
- package/dist/governance.js +22 -0
- package/dist/index.d.ts +14 -6
- package/dist/index.js +5 -1
- package/dist/kernel.d.ts +43 -0
- package/dist/memory/agent.d.ts +66 -0
- package/dist/memory/agent.js +151 -0
- package/dist/memory/protocols.d.ts +56 -0
- package/dist/runtime/execution-plane.d.ts +6 -9
- package/dist/runtime/execution-plane.js +53 -25
- package/dist/runtime/kernel-event-log.d.ts +26 -0
- package/dist/runtime/kernel-event-log.js +220 -0
- package/dist/runtime/kernel-primitives-dashboard.d.ts +44 -0
- package/dist/runtime/kernel-primitives-dashboard.js +135 -0
- package/dist/runtime/kernel-step.d.ts +25 -1
- package/dist/runtime/kernel-step.js +12 -3
- package/dist/runtime/large-result-spool.d.ts +84 -0
- package/dist/runtime/large-result-spool.js +167 -0
- package/dist/runtime/os-profile.d.ts +30 -0
- package/dist/runtime/os-profile.js +71 -0
- package/dist/runtime/os-snapshot.d.ts +35 -0
- package/dist/runtime/os-snapshot.js +128 -0
- package/dist/runtime/runner.d.ts +79 -9
- package/dist/runtime/runner.js +463 -135
- package/dist/runtime/session-log.d.ts +118 -4
- package/dist/runtime/session-log.js +15 -4
- package/dist/runtime/sub-agent-orchestrator.d.ts +2 -2
- package/dist/runtime/sub-agent-orchestrator.js +14 -23
- package/dist/types/agent.d.ts +12 -3
- package/dist/types/agent.js +18 -0
- package/dist/types.d.ts +22 -0
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# DeepStrike Node.js SDK
|
|
2
2
|
|
|
3
|
-
Runtime framework built on a Rust kernel. The kernel
|
|
3
|
+
Runtime framework built on a Rust kernel. The kernel owns loop control, context compression, governance, signal routing, and memory paging — the SDK owns all I/O (LLM calls, tool execution, disk, long-term memory).
|
|
4
|
+
|
|
5
|
+
Node.js is the reference SDK for the **Agent OS native profile**: declarative governance and in-kernel signal routing are enabled by default on every run.
|
|
4
6
|
|
|
5
7
|
## Install
|
|
6
8
|
|
|
@@ -24,9 +26,9 @@ Pre-built native addons are available for the following platforms:
|
|
|
24
26
|
| Linux ARM64 (musl / Alpine) | `@deepstrike/core-linux-arm64-musl` |
|
|
25
27
|
| Windows x64 | `@deepstrike/core-win32-x64-msvc` |
|
|
26
28
|
|
|
27
|
-
The correct platform package is selected
|
|
29
|
+
The correct platform package is selected automatically via `optionalDependencies`.
|
|
28
30
|
|
|
29
|
-
> **Note:** `@deepstrike/core` is the low-level
|
|
31
|
+
> **Note:** `@deepstrike/core` is the low-level N-API binding and is managed as an internal dependency of `@deepstrike/sdk`. When developing against a local kernel build, run `npm run test:local-core` from this directory to rebuild the native module from `../crates/deepstrike-node`.
|
|
30
32
|
|
|
31
33
|
---
|
|
32
34
|
|
|
@@ -65,14 +67,14 @@ const result = await collectText(runner.run({
|
|
|
65
67
|
console.log(result)
|
|
66
68
|
```
|
|
67
69
|
|
|
68
|
-
Same-session
|
|
70
|
+
Same-session continuity is explicit via `sessionId`:
|
|
69
71
|
|
|
70
72
|
```typescript
|
|
71
73
|
await collectText(runner.run({ sessionId: "chat-1", goal: "My name is Ada." }))
|
|
72
74
|
const reply = await collectText(runner.run({ sessionId: "chat-1", goal: "What is my name?" }))
|
|
73
75
|
```
|
|
74
76
|
|
|
75
|
-
Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when
|
|
77
|
+
Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when replay should survive restarts. `wake(sessionId)` resumes from the event log without inserting a duplicate `run_started` event.
|
|
76
78
|
|
|
77
79
|
Streaming:
|
|
78
80
|
|
|
@@ -87,6 +89,62 @@ for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize R
|
|
|
87
89
|
|
|
88
90
|
---
|
|
89
91
|
|
|
92
|
+
## Architecture
|
|
93
|
+
|
|
94
|
+
```text
|
|
95
|
+
┌─────────────────────────────────────────────────────────┐
|
|
96
|
+
│ RuntimeRunner (Layer 1.5) │
|
|
97
|
+
│ LLMProvider · ExecutionPlane · SessionLog · DreamStore │
|
|
98
|
+
└───────────────────────────┬─────────────────────────────┘
|
|
99
|
+
│ step(JSON event) ↔ actions / observations
|
|
100
|
+
┌───────────────────────────▼─────────────────────────────┐
|
|
101
|
+
│ @deepstrike/core KernelRuntime │
|
|
102
|
+
│ P1 Syscall · P2 Sched · P3 MM · Proc · IPC │
|
|
103
|
+
└─────────────────────────────────────────────────────────┘
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
The runner drives a single loop:
|
|
107
|
+
|
|
108
|
+
1. Kernel returns an **action** — `call_provider`, `execute_tool`, `evaluate_milestone`, or `done`.
|
|
109
|
+
2. SDK executes the action (stream LLM, run tools, call milestone verifier).
|
|
110
|
+
3. SDK feeds the result back as a kernel **event** (`provider_result`, `tool_results`, …).
|
|
111
|
+
4. Kernel **observations** (compression, page-out, spool, signals, …) are drained into `SessionLog`.
|
|
112
|
+
|
|
113
|
+
Kernel session events carry an optional `category` tag (`syscall` · `sched` · `mm` · `proc` · `ipc`) for diagnostics and OS snapshot rebuilds.
|
|
114
|
+
|
|
115
|
+
### What Agent OS gives you
|
|
116
|
+
|
|
117
|
+
The mechanisms above are not internal refactors — they change what you can build without custom runner code:
|
|
118
|
+
|
|
119
|
+
**Kernel-mediated runtime (M0–M4)**
|
|
120
|
+
Tool calls, spawns, compression, and signals pass through one kernel gate with an explicit lifecycle (Ready / Running / Blocked / Suspended). You implement I/O; the kernel decides *when* and *whether*. Node, Python, and Rust share the same decision path, so `wake(sessionId)` and cross-language tooling see consistent behavior.
|
|
121
|
+
|
|
122
|
+
**Longer, sturdier sessions (Layer-1 spool + semantic page-out)**
|
|
123
|
+
Oversized tool results (> 50 KB) stay in context as a preview plus a `.spool/` reference — the model reads the full payload on demand via ordinary file tools. When pressure triggers semantic eviction, the SDK summarizes archived content into `DreamStore` and satisfies `page_in_requested` on the way back in. Long tasks survive token pressure instead of failing mid-run.
|
|
124
|
+
|
|
125
|
+
**Safety and governance by default (OS native profile)**
|
|
126
|
+
Every run loads declarative `governancePolicy` (deny / ask_user / rate-limit / param rules) and in-kernel signal routing (`attentionPolicy`, default queue 64). Dangerous tools, external interrupts, and approval flows are policy — not ad-hoc `if` checks in your handlers.
|
|
127
|
+
|
|
128
|
+
**Long-term memory as syscalls (Phase-7)**
|
|
129
|
+
`writeMemory` and `queryMemory` run outside the main tool loop: kernel validation before `DreamStore.commit`, search → `selectMemories` → `memory_retrieval_result` on query. Failed writes emit `memory_validation_failed` for audit; good memory is durable without polluting history.
|
|
130
|
+
|
|
131
|
+
**Multi-agent and multi-signal orchestration**
|
|
132
|
+
Sub-agents register in the kernel process table (`agent_process_changed`); parent runs suspend explicitly until `sub_agent_completed`. Signals get disposition (Interrupt / Queue / Observe / Dropped) in-kernel, so gateways, cron, and heartbeats compose with the main loop instead of racing it.
|
|
133
|
+
|
|
134
|
+
**Observable like an OS log**
|
|
135
|
+
Spool, page-out, signals, processes, budgets, and memory events land in `SessionLog` with categories. Rebuild an OS snapshot (`pageOutCount`, `spoolCount`, `processByAgent`, memory counters) from one event stream — replay still strips audit events when reconstructing LLM messages.
|
|
136
|
+
|
|
137
|
+
| You need… | Use… |
|
|
138
|
+
|---|---|
|
|
139
|
+
| Policy before tools run | `governancePolicy` (default: allow-all native profile) |
|
|
140
|
+
| External interrupts | `signalSource` + in-kernel `attentionPolicy` |
|
|
141
|
+
| Huge tool output | Automatic Layer-1 spool; optional custom `resultSpool` |
|
|
142
|
+
| Durable recall across runs | `DreamStore` + semantic `page_out` via `dreamSummarizer` |
|
|
143
|
+
| Programmatic memory I/O | `runner.writeMemory()` / `runner.queryMemory()` |
|
|
144
|
+
| Debug / compliance | `SessionLog` events + OS snapshot helpers |
|
|
145
|
+
|
|
146
|
+
---
|
|
147
|
+
|
|
90
148
|
## Providers
|
|
91
149
|
|
|
92
150
|
| Class | Backend | Notes |
|
|
@@ -103,7 +161,7 @@ for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize R
|
|
|
103
161
|
|
|
104
162
|
All providers accept `RetryConfig` for exponential backoff and share a `CircuitBreaker`.
|
|
105
163
|
|
|
106
|
-
`extensions` are forwarded by every provider in both `complete()` and `stream()` while SDK-owned structural fields such as `model`, `messages`, `tools`, and streaming flags remain protected.
|
|
164
|
+
`extensions` are forwarded by every provider in both `complete()` and `stream()` while SDK-owned structural fields such as `model`, `messages`, `tools`, and streaming flags remain protected.
|
|
107
165
|
|
|
108
166
|
OpenAI can also be selected through the provider catalog:
|
|
109
167
|
|
|
@@ -138,39 +196,120 @@ const runner = new RuntimeRunner({
|
|
|
138
196
|
```
|
|
139
197
|
|
|
140
198
|
- `memory(query)` / `knowledge(query)` meta-tool results → **history** (tool results)
|
|
141
|
-
-
|
|
199
|
+
- Inbound signals are routed by the in-kernel attention policy and rendered into **Slot 3**
|
|
142
200
|
- Anthropic: Slots 1–2 get separate `cache_control` breakpoints
|
|
143
201
|
|
|
144
|
-
Full reference: [docs/context-
|
|
202
|
+
Full reference: [docs/concepts/context-slots-compression.md](../docs/concepts/context-slots-compression.md)
|
|
145
203
|
|
|
146
204
|
---
|
|
147
205
|
|
|
148
206
|
## Runtime options
|
|
149
207
|
|
|
150
208
|
```typescript
|
|
151
|
-
|
|
209
|
+
import {
|
|
210
|
+
DEFAULT_NATIVE_GOVERNANCE_POLICY,
|
|
211
|
+
DEFAULT_NATIVE_ATTENTION_POLICY,
|
|
212
|
+
} from "@deepstrike/sdk"
|
|
213
|
+
|
|
152
214
|
const runner = new RuntimeRunner({
|
|
153
215
|
provider,
|
|
154
216
|
executionPlane: plane,
|
|
155
217
|
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
maxAttempts: 3,
|
|
218
|
+
|
|
219
|
+
// Scheduler budget
|
|
220
|
+
maxTokens: 128_000,
|
|
221
|
+
maxTurns: 25,
|
|
222
|
+
timeoutMs: 60_000,
|
|
223
|
+
schedulerBudget: { maxWallMs: 300_000 },
|
|
224
|
+
|
|
225
|
+
// Resource quotas (M2) — enforced at the kernel syscall trap. Opt-in; omit for unbounded.
|
|
226
|
+
resourceQuota: {
|
|
227
|
+
maxConcurrentSubagents: 4, // deny spawn while at cap
|
|
228
|
+
maxSpawnDepth: 2, // deny spawn past nesting depth
|
|
229
|
+
memoryWritesPerWindow: { maxWrites: 20, windowMs: 60_000 }, // rate-limit writeMemory
|
|
169
230
|
},
|
|
170
|
-
|
|
231
|
+
|
|
232
|
+
// Long-term memory policy (set_memory_policy) — opt-in, kernel-enforced; omit for defaults.
|
|
233
|
+
memoryPolicy: {
|
|
234
|
+
memoryPath: "./.memory", // where the SDK persists/scans memories (SDK-consumed)
|
|
235
|
+
staleWarningDays: 30, // flag recalled memories older than this (SDK-consumed)
|
|
236
|
+
retrievalTopK: 5, // kernel caps query_memory requested_k to this
|
|
237
|
+
validationEnabled: true, // false → admit writes without validation
|
|
238
|
+
maxContentBytes: 10_000, // override write_memory content-size limit
|
|
239
|
+
maxNameLength: 100, // override write_memory name-length limit
|
|
240
|
+
},
|
|
241
|
+
|
|
242
|
+
// Agent OS native profile (defaults shown)
|
|
243
|
+
governancePolicy: DEFAULT_NATIVE_GOVERNANCE_POLICY,
|
|
244
|
+
attentionPolicy: DEFAULT_NATIVE_ATTENTION_POLICY, // SignalRouter queue size 64
|
|
245
|
+
|
|
246
|
+
// Host I/O
|
|
247
|
+
extensions: { temperature: 0.1 },
|
|
248
|
+
skillDir: "./skills",
|
|
249
|
+
knowledgeSource: myKS,
|
|
250
|
+
signalSource: gw,
|
|
251
|
+
dreamStore: myStore,
|
|
252
|
+
agentId: "my-agent",
|
|
253
|
+
initialMemory: ["..."],
|
|
254
|
+
|
|
255
|
+
// Memory paging & compression (SDK-side I/O)
|
|
256
|
+
compressionStore: archiveStore, // persist compressed transcript slices
|
|
257
|
+
asyncSummarizer: mySummarizer, // upgrade rule-based compression summaries
|
|
258
|
+
dreamProvider: dreamLlm, // LLM for idle dream() synthesis
|
|
259
|
+
dreamSummarizer: myDreamSummarizer, // LLM for semantic page_out → DreamStore
|
|
260
|
+
|
|
261
|
+
// Sub-agents
|
|
262
|
+
runSpec: { role: "orchestrator", isolation: "process" },
|
|
263
|
+
milestoneContract: myContract,
|
|
264
|
+
milestonePolicy: "require_verifier",
|
|
265
|
+
onMilestoneEvaluate: async ({ phaseId, criteria }) => ({ passed: true, phaseId }),
|
|
266
|
+
subAgentHarness: { evalProvider, maxAttempts: 3 },
|
|
267
|
+
|
|
268
|
+
// Governance UX (AskUser path)
|
|
269
|
+
onPermissionRequest: async (req) => ({ approved: true }),
|
|
270
|
+
|
|
271
|
+
// Diagnostics
|
|
272
|
+
enableDiagnosticsDashboard: true, // CLI view grouped by Syscall / Sched / MM
|
|
171
273
|
})
|
|
172
274
|
```
|
|
173
275
|
|
|
276
|
+
| Option | Purpose |
|
|
277
|
+
|--------|---------|
|
|
278
|
+
| `governancePolicy` | Declarative deny / ask_user / rate-limit / param rules loaded into the kernel before `start_run` |
|
|
279
|
+
| `attentionPolicy` | In-kernel signal router queue size (default 64) |
|
|
280
|
+
| `resourceQuota` | M2 declarative limits — `maxConcurrentSubagents` / `maxSpawnDepth` / `memoryWritesPerWindow` — enforced at the kernel syscall trap (`set_resource_quota`); over-quota spawns roll back, over-rate writes surface as `memory_validation_failed` |
|
|
281
|
+
| `memoryPolicy` | Long-term memory config sent as `set_memory_policy` and **kernel-enforced**: `validationEnabled: false` admits writes without validation, `maxContentBytes` / `maxNameLength` override validation limits, `retrievalTopK` caps `query_memory` breadth; `memoryPath` / `staleWarningDays` are SDK-consumed (requires `dreamStore` + `agentId` to enable memory) |
|
|
282
|
+
| `onPermissionRequest` | Resolves `tool_gated` + `suspended` → kernel `resume` with approved/denied call IDs |
|
|
283
|
+
| `compressionStore` | Writes archived messages on `compressed` observations |
|
|
284
|
+
| `asyncSummarizer` | Background LLM summary after compression; stored as `summary_upgraded` |
|
|
285
|
+
| `dreamSummarizer` | Summarizes `page_out { tier_hint: "semantic" }` into `DreamStore` during a run |
|
|
286
|
+
| `dreamProvider` | Separate LLM for `dream()` idle consolidation (falls back to `provider`) |
|
|
287
|
+
|
|
288
|
+
Rebuild an OS diagnostics snapshot from session events:
|
|
289
|
+
|
|
290
|
+
```typescript
|
|
291
|
+
import { rebuildOsSnapshotFromSessionEvents } from "@deepstrike/sdk"
|
|
292
|
+
|
|
293
|
+
const events = (await sessionLog.read(sessionId)).map(e => e.event)
|
|
294
|
+
const snap = rebuildOsSnapshotFromSessionEvents(events)
|
|
295
|
+
// snap.pageOutCount, snap.spoolCount, snap.signals, snap.processByAgent, …
|
|
296
|
+
```
|
|
297
|
+
|
|
298
|
+
---
|
|
299
|
+
|
|
300
|
+
## Large result spool (Layer 1)
|
|
301
|
+
|
|
302
|
+
When a single tool result exceeds **50 KB**, the kernel keeps a short preview in context and emits `large_result_spooled`. The SDK writes the full payload to `.spool/` under the process cwd (SHA-256 keyed files) and logs `spool_ref` in the session.
|
|
303
|
+
|
|
304
|
+
The model can retrieve full content via ordinary read tools — `LocalExecutionPlane` transparently resolves paths under `.spool/`:
|
|
305
|
+
|
|
306
|
+
```typescript
|
|
307
|
+
// Kernel context shows a preview + spool reference.
|
|
308
|
+
// LLM calls read_file({ path: ".spool/abc123…" }) → full content returned.
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
No configuration is required; customize the directory by passing a `resultSpool` instance when constructing `RuntimeRunner` (see tests under `tests/runtime/large-result-spool.test.ts`).
|
|
312
|
+
|
|
174
313
|
---
|
|
175
314
|
|
|
176
315
|
## Tools
|
|
@@ -179,10 +318,28 @@ const runner = new RuntimeRunner({
|
|
|
179
318
|
import { tool, readFile } from "@deepstrike/sdk"
|
|
180
319
|
|
|
181
320
|
plane.register(tool("search", "Search.", schema, async (args) => ...))
|
|
182
|
-
plane.register(readFile) // built-in: read files from disk
|
|
321
|
+
plane.register(readFile) // built-in: read files from disk (also resolves .spool/ refs)
|
|
183
322
|
plane.unregister("search")
|
|
184
323
|
```
|
|
185
324
|
|
|
325
|
+
Execution planes:
|
|
326
|
+
|
|
327
|
+
| Plane | Use case |
|
|
328
|
+
|-------|----------|
|
|
329
|
+
| `LocalExecutionPlane` | In-process tools (default) |
|
|
330
|
+
| `FilteredExecutionPlane` | Capability-filtered sub-agent tools |
|
|
331
|
+
| `ProcessSandboxPlane` | OS subprocess isolation |
|
|
332
|
+
| `McpProxyPlane` | MCP server tools |
|
|
333
|
+
| `RemoteVpcPlane` | Remote execution |
|
|
334
|
+
|
|
335
|
+
Mount capabilities on an active run:
|
|
336
|
+
|
|
337
|
+
```typescript
|
|
338
|
+
runner.mountTool(schema)
|
|
339
|
+
runner.mountSkill("summarize", "Summarize text")
|
|
340
|
+
runner.unmountCapability("tool", "search")
|
|
341
|
+
```
|
|
342
|
+
|
|
186
343
|
---
|
|
187
344
|
|
|
188
345
|
## Skills
|
|
@@ -214,9 +371,11 @@ effort: 1
|
|
|
214
371
|
|
|
215
372
|
## Knowledge
|
|
216
373
|
|
|
217
|
-
Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand.
|
|
374
|
+
Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand. Runtime retrieval results land in **history** as tool results.
|
|
218
375
|
|
|
219
|
-
To inject durable knowledge at startup (Slot 2, cacheable on Anthropic), use `initialMemory` or
|
|
376
|
+
To inject durable knowledge at startup (Slot 2, cacheable on Anthropic), use `initialMemory` or `runner.pushKnowledge()`.
|
|
377
|
+
|
|
378
|
+
Before tool execution the kernel may emit `page_in_requested`; the SDK satisfies it from `DreamStore`, `KnowledgeSource`, and a local semantic page-out cache, then feeds `page_in` back to the kernel.
|
|
220
379
|
|
|
221
380
|
```typescript
|
|
222
381
|
const runner = new RuntimeRunner({
|
|
@@ -238,7 +397,7 @@ const runner = new RuntimeRunner({
|
|
|
238
397
|
|
|
239
398
|
### WorkingMemory (SDK-side scratch pad)
|
|
240
399
|
|
|
241
|
-
`WorkingMemory` is an SDK helper — not the kernel
|
|
400
|
+
`WorkingMemory` is an SDK helper — not the kernel working partition. Kernel task state lives in `task_state` and renders into Slot 3 (`turns[0]`).
|
|
242
401
|
|
|
243
402
|
```typescript
|
|
244
403
|
import { WorkingMemory } from "@deepstrike/sdk"
|
|
@@ -248,7 +407,7 @@ mem.get("step") // 1
|
|
|
248
407
|
mem.clear()
|
|
249
408
|
```
|
|
250
409
|
|
|
251
|
-
### DreamStore (long-term memory
|
|
410
|
+
### DreamStore (long-term memory)
|
|
252
411
|
|
|
253
412
|
```typescript
|
|
254
413
|
import type { DreamStore } from "@deepstrike/sdk"
|
|
@@ -266,32 +425,91 @@ const runner = new RuntimeRunner({
|
|
|
266
425
|
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
267
426
|
maxTokens: 4096,
|
|
268
427
|
dreamStore: new MyStore(),
|
|
269
|
-
agentId: "my-agent", // enables `memory` meta-tool
|
|
428
|
+
agentId: "my-agent", // enables `memory` meta-tool + semantic page-out archival
|
|
270
429
|
})
|
|
430
|
+
```
|
|
431
|
+
|
|
432
|
+
Three memory paths:
|
|
271
433
|
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
434
|
+
| Path | When | What happens |
|
|
435
|
+
|------|------|--------------|
|
|
436
|
+
| In-session `memory(query)` | LLM calls meta-tool | `DreamStore.search()` → history tool result |
|
|
437
|
+
| `initialMemory` | Run start | Injected into Slot 2 (`systemKnowledge`) |
|
|
438
|
+
| Semantic `page_out` | Kernel evicts with `tier_hint: "semantic"` | SDK summarizes via `dreamSummarizer` / `dreamProvider` → `DreamStore.commit()` |
|
|
439
|
+
| `dream(agentId)` | Explicit idle call | `IdlePipeline` batch-consolidates past sessions |
|
|
440
|
+
|
|
441
|
+
```typescript
|
|
442
|
+
// Post-session batch consolidation
|
|
275
443
|
const result = await runner.dream("my-agent", Date.now())
|
|
276
444
|
```
|
|
277
445
|
|
|
446
|
+
### Phase-7 memory syscalls (`writeMemory` / `queryMemory`)
|
|
447
|
+
|
|
448
|
+
Kernel-validated long-term memory I/O outside the main tool loop:
|
|
449
|
+
|
|
450
|
+
```typescript
|
|
451
|
+
await runner.writeMemory({
|
|
452
|
+
metadata: {
|
|
453
|
+
name: "prefers-small-tests",
|
|
454
|
+
description: "User prefers focused unit tests",
|
|
455
|
+
kind: "feedback",
|
|
456
|
+
created_at: Date.now(),
|
|
457
|
+
updated_at: Date.now(),
|
|
458
|
+
},
|
|
459
|
+
content: "User prefers focused unit tests for SDK behavior.",
|
|
460
|
+
}, { sessionId: "my-session" })
|
|
461
|
+
|
|
462
|
+
const hits = await runner.queryMemory({
|
|
463
|
+
current_context: "Need testing preferences",
|
|
464
|
+
active_tools: [],
|
|
465
|
+
already_surfaced: [],
|
|
466
|
+
top_k: 5,
|
|
467
|
+
}, { sessionId: "my-session" })
|
|
468
|
+
```
|
|
469
|
+
|
|
470
|
+
Session events: `memory_written`, `memory_queried`, `memory_validation_failed`, `memory_retrieval_result`.
|
|
471
|
+
|
|
278
472
|
---
|
|
279
473
|
|
|
280
474
|
## Governance
|
|
281
475
|
|
|
282
|
-
###
|
|
476
|
+
### In-kernel declarative policy (preferred)
|
|
477
|
+
|
|
478
|
+
Every run loads `governancePolicy` into the kernel via `load_governance_policy`. The kernel enforces rules **before** tools execute:
|
|
283
479
|
|
|
284
480
|
```typescript
|
|
285
|
-
import {
|
|
481
|
+
import type { GovernancePolicy } from "@deepstrike/sdk"
|
|
482
|
+
|
|
483
|
+
const policy: GovernancePolicy = {
|
|
484
|
+
rules: [
|
|
485
|
+
{ pattern: "read_file", action: "allow" },
|
|
486
|
+
{ pattern: "write_file", action: "ask_user" },
|
|
487
|
+
{ pattern: "run_command", action: "ask_user" },
|
|
488
|
+
{ pattern: "*", action: "deny" },
|
|
489
|
+
],
|
|
490
|
+
rateLimits: [{ tool: "api_call", maxCalls: 10, windowMs: 60_000 }],
|
|
491
|
+
}
|
|
286
492
|
|
|
287
|
-
const
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
493
|
+
const runner = new RuntimeRunner({
|
|
494
|
+
provider,
|
|
495
|
+
executionPlane: plane,
|
|
496
|
+
sessionLog,
|
|
497
|
+
governancePolicy: policy,
|
|
498
|
+
onPermissionRequest: async (req) => {
|
|
499
|
+
console.log(`Approve ${req.toolName}?`, req.arguments)
|
|
500
|
+
return { approved: true }
|
|
501
|
+
},
|
|
502
|
+
})
|
|
292
503
|
```
|
|
293
504
|
|
|
294
|
-
|
|
505
|
+
- `deny` → tool rejected with `tool_denied`
|
|
506
|
+
- `ask_user` → `tool_gated` + `suspended`; resolve via `onPermissionRequest`, then kernel `resume`
|
|
507
|
+
|
|
508
|
+
Default when omitted: allow-all (`DEFAULT_NATIVE_GOVERNANCE_POLICY`).
|
|
509
|
+
|
|
510
|
+
### Standalone Governance class
|
|
511
|
+
|
|
512
|
+
`Governance` wraps the native governance evaluator for SDK-side use (tests, custom gates). It is **not** wired automatically into `RuntimeRunner` — use `governancePolicy` for run-time enforcement.
|
|
295
513
|
|
|
296
514
|
```typescript
|
|
297
515
|
import { Governance } from "@deepstrike/sdk"
|
|
@@ -299,45 +517,65 @@ import { Governance } from "@deepstrike/sdk"
|
|
|
299
517
|
const gov = new Governance("allow")
|
|
300
518
|
gov.addPermissionRule("danger.*", "deny")
|
|
301
519
|
gov.blockTool("rm_rf")
|
|
302
|
-
gov.
|
|
303
|
-
gov.requireParam("write_file", "path")
|
|
304
|
-
gov.allowParamValues("set_mode", "mode", ["read", "write"])
|
|
305
|
-
gov.limitParamRange("sleep", "seconds", 0, 10)
|
|
306
|
-
|
|
307
|
-
const runner = new RuntimeRunner({
|
|
308
|
-
provider,
|
|
309
|
-
executionPlane: plane,
|
|
310
|
-
sessionLog: new FileSessionLog(".deepstrike/sessions"),
|
|
311
|
-
maxTokens: 4096,
|
|
312
|
-
governance: gov,
|
|
313
|
-
})
|
|
314
|
-
// Every tool call goes through: Permission → Veto → RateLimit → Constraint → Audit
|
|
520
|
+
gov.evaluate("read_file", '{"path":"x"}')
|
|
315
521
|
```
|
|
316
522
|
|
|
523
|
+
### SDK PermissionManager
|
|
524
|
+
|
|
525
|
+
`PermissionManager` is a separate SDK-side permission layer for apps that manage their own approval UX outside the kernel loop.
|
|
526
|
+
|
|
317
527
|
---
|
|
318
528
|
|
|
319
529
|
## Signals
|
|
320
530
|
|
|
531
|
+
Inbound signals are routed by the in-kernel attention policy (default queue size 64):
|
|
532
|
+
|
|
533
|
+
| Urgency | Typical disposition |
|
|
534
|
+
|---------|-------------------|
|
|
535
|
+
| `critical` / `high` | `interrupt_now` — may yield a new `call_provider` action |
|
|
536
|
+
| `normal` / `low` | `queue` — buffered; no action until dequeued |
|
|
537
|
+
| queue full | `dropped` |
|
|
538
|
+
|
|
321
539
|
```typescript
|
|
322
540
|
import { SignalGateway, ScheduledPrompt } from "@deepstrike/sdk"
|
|
323
541
|
|
|
324
542
|
const gw = new SignalGateway()
|
|
325
543
|
gw.schedule(new ScheduledPrompt("standup", Date.now() + 3600_000))
|
|
326
|
-
gw.ingest({ kind: "
|
|
544
|
+
gw.ingest({ kind: "alert", urgency: "normal", payload: { goal: "Check deploy" } })
|
|
327
545
|
|
|
328
546
|
const runner = new RuntimeRunner({
|
|
329
547
|
provider,
|
|
330
548
|
executionPlane: plane,
|
|
331
|
-
sessionLog
|
|
332
|
-
maxTokens: 4096,
|
|
549
|
+
sessionLog,
|
|
333
550
|
signalSource: gw,
|
|
551
|
+
attentionPolicy: { maxQueueSize: 64 },
|
|
334
552
|
})
|
|
335
|
-
// kind="interrupt" → immediately stops the running runner
|
|
336
553
|
|
|
337
|
-
runner.interrupt() //
|
|
554
|
+
runner.interrupt() // cooperative abort → kernel timeout path
|
|
338
555
|
gw.destroy()
|
|
339
556
|
```
|
|
340
557
|
|
|
558
|
+
Each routed signal produces a `signal_disposed` session event (`category: "ipc"`).
|
|
559
|
+
|
|
560
|
+
---
|
|
561
|
+
|
|
562
|
+
## Sub-agents
|
|
563
|
+
|
|
564
|
+
Spawn isolated child agents through the kernel process table:
|
|
565
|
+
|
|
566
|
+
```typescript
|
|
567
|
+
for await (const evt of runner.spawnSubAgent({
|
|
568
|
+
role: "researcher",
|
|
569
|
+
isolation: "process",
|
|
570
|
+
goal: "Find three sources on topic X",
|
|
571
|
+
criteria: ["At least 3 URLs"],
|
|
572
|
+
})) {
|
|
573
|
+
if (evt.type === "done") console.log(evt.status)
|
|
574
|
+
}
|
|
575
|
+
```
|
|
576
|
+
|
|
577
|
+
Requires an active parent run (`run()` / `wake()` in progress). The kernel emits `agent_process_changed`; the default `SubAgentOrchestrator` runs the child with a filtered execution plane and feeds `sub_agent_completed` back.
|
|
578
|
+
|
|
341
579
|
---
|
|
342
580
|
|
|
343
581
|
## Harness (evaluation framework)
|
|
@@ -345,30 +583,20 @@ gw.destroy()
|
|
|
345
583
|
```typescript
|
|
346
584
|
import { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "@deepstrike/sdk"
|
|
347
585
|
|
|
348
|
-
// 1. SinglePass — run once, always passes
|
|
349
586
|
const outcome = await new SinglePassHarness(runner).run({ goal: "Say hello" })
|
|
350
587
|
|
|
351
|
-
// 2. EvalLoop — retry until QualityGate passes
|
|
352
588
|
const harness = new EvalLoopHarness(runner, {
|
|
353
589
|
async evaluate(_req, out) { return out.result.includes("hello") },
|
|
354
590
|
}, 3)
|
|
355
591
|
|
|
356
|
-
// 3. HarnessLoop — LLM-as-judge with feedback injection + skill extraction
|
|
357
592
|
const loop = new HarnessLoop(runner, evalProvider, { maxAttempts: 3, skillDir: "./skills" })
|
|
358
593
|
|
|
359
|
-
// Sub-agents: pass subAgentHarness on RuntimeRunner to auto-evaluate spawned children
|
|
360
594
|
const runnerWithHarness = new RuntimeRunner({
|
|
361
595
|
provider,
|
|
362
596
|
executionPlane: plane,
|
|
363
597
|
sessionLog,
|
|
364
598
|
subAgentHarness: { evalProvider, maxAttempts: 3 },
|
|
365
599
|
})
|
|
366
|
-
for await (const event of loop.runStreaming({
|
|
367
|
-
goal: "Write a haiku",
|
|
368
|
-
criteria: [{ text: "Must be 3 lines", required: true }],
|
|
369
|
-
})) {
|
|
370
|
-
if (event.type === "done") console.log(event.verdict.passed, event.verdict.feedback)
|
|
371
|
-
}
|
|
372
600
|
```
|
|
373
601
|
|
|
374
602
|
---
|
|
@@ -387,4 +615,12 @@ for await (const event of loop.runStreaming({
|
|
|
387
615
|
| `done` | `iterations`, `totalTokens`, `status` |
|
|
388
616
|
| `error` | `message` |
|
|
389
617
|
|
|
390
|
-
`status`: `completed` · `max_turns` · `token_budget` · `timeout` · `user_abort` · `error`
|
|
618
|
+
`status`: `completed` · `max_turns` · `token_budget` · `timeout` · `user_abort` · `error` · `milestone_pending`
|
|
619
|
+
|
|
620
|
+
---
|
|
621
|
+
|
|
622
|
+
## Further reading
|
|
623
|
+
|
|
624
|
+
- [SDK OS parity matrix](../docs/sdk-os-parity.md)
|
|
625
|
+
- [Kernel ABI reference](../docs/reference/kernel-abi.md)
|
|
626
|
+
- [Context slots & compression](../docs/concepts/context-slots-compression.md)
|
|
@@ -35,8 +35,6 @@ export declare class CreatorVerifierMode {
|
|
|
35
35
|
maxAttempts?: number;
|
|
36
36
|
/** Stable orchestration session for kernel lineage audit. */
|
|
37
37
|
coordinatorSessionId?: string;
|
|
38
|
-
/** Opt out of kernel spawn path and use legacy independent runner sessions. */
|
|
39
|
-
useLegacyRunners?: boolean;
|
|
40
38
|
});
|
|
41
39
|
run(contract: VerificationContract): Promise<ContractOutcome>;
|
|
42
40
|
/** Aggregate drift metrics across all runs through this mode instance. */
|
|
@@ -64,7 +62,6 @@ export declare class OrchestrationMode {
|
|
|
64
62
|
constructor(pool: AgentPool, options?: {
|
|
65
63
|
maxAttempts?: number;
|
|
66
64
|
coordinatorSessionId?: string;
|
|
67
|
-
useLegacyRunners?: boolean;
|
|
68
65
|
});
|
|
69
66
|
run(goal: string): Promise<ContractOutcome & {
|
|
70
67
|
contract: VerificationContract;
|
|
@@ -30,9 +30,7 @@ export class CreatorVerifierMode {
|
|
|
30
30
|
}
|
|
31
31
|
async run(contract) {
|
|
32
32
|
this._total++;
|
|
33
|
-
|
|
34
|
-
this.pool.ensureCoordinator(this.options.coordinatorSessionId);
|
|
35
|
-
}
|
|
33
|
+
this.pool.ensureCoordinator(this.options.coordinatorSessionId);
|
|
36
34
|
const harness = new ContractDrivenHarness(this.pool, contract, {
|
|
37
35
|
maxAttempts: this.options.maxAttempts ?? 3,
|
|
38
36
|
});
|
|
@@ -81,9 +79,7 @@ export class OrchestrationMode {
|
|
|
81
79
|
this.inner = new CreatorVerifierMode(pool, options);
|
|
82
80
|
}
|
|
83
81
|
async run(goal) {
|
|
84
|
-
|
|
85
|
-
this.pool.ensureCoordinator(this.options.coordinatorSessionId);
|
|
86
|
-
}
|
|
82
|
+
this.pool.ensureCoordinator(this.options.coordinatorSessionId);
|
|
87
83
|
// Step 1: orchestrator produces a VerificationContract
|
|
88
84
|
const contractJson = await this.pool.orchestrate(goal);
|
|
89
85
|
const contract = this._parseContract(contractJson, goal);
|
package/dist/governance.d.ts
CHANGED
|
@@ -15,3 +15,39 @@ export declare class Governance {
|
|
|
15
15
|
evaluate(toolName: string, argsJson: string): GovernanceVerdict;
|
|
16
16
|
}
|
|
17
17
|
export type { GovernanceVerdict };
|
|
18
|
+
type GovernancePolicyAction = "allow" | "deny" | "ask_user";
|
|
19
|
+
export interface GovernancePolicy {
|
|
20
|
+
defaultAction?: GovernancePolicyAction;
|
|
21
|
+
rules?: {
|
|
22
|
+
pattern: string;
|
|
23
|
+
action: GovernancePolicyAction;
|
|
24
|
+
}[];
|
|
25
|
+
vetoes?: string[];
|
|
26
|
+
rateLimits?: {
|
|
27
|
+
tool: string;
|
|
28
|
+
maxCalls: number;
|
|
29
|
+
windowMs: number;
|
|
30
|
+
}[];
|
|
31
|
+
constraints?: GovernanceConstraint[];
|
|
32
|
+
}
|
|
33
|
+
export type GovernanceConstraint = {
|
|
34
|
+
kind: "required";
|
|
35
|
+
tool: string;
|
|
36
|
+
path: string;
|
|
37
|
+
} | {
|
|
38
|
+
kind: "enum";
|
|
39
|
+
tool: string;
|
|
40
|
+
path: string;
|
|
41
|
+
values: string[];
|
|
42
|
+
} | {
|
|
43
|
+
kind: "range";
|
|
44
|
+
tool: string;
|
|
45
|
+
path: string;
|
|
46
|
+
min?: number;
|
|
47
|
+
max?: number;
|
|
48
|
+
};
|
|
49
|
+
/**
|
|
50
|
+
* Convert a declarative {@link GovernancePolicy} into the `load_governance_policy`
|
|
51
|
+
* kernel event payload (snake_case wire fields). Pure — no side effects.
|
|
52
|
+
*/
|
|
53
|
+
export declare function governancePolicyToKernelEvent(policy: GovernancePolicy): Record<string, unknown>;
|
package/dist/governance.js
CHANGED
|
@@ -32,3 +32,25 @@ export class Governance {
|
|
|
32
32
|
return this.inner.evaluate(toolName, argsJson);
|
|
33
33
|
}
|
|
34
34
|
}
|
|
35
|
+
/**
|
|
36
|
+
* Convert a declarative {@link GovernancePolicy} into the `load_governance_policy`
|
|
37
|
+
* kernel event payload (snake_case wire fields). Pure — no side effects.
|
|
38
|
+
*/
|
|
39
|
+
export function governancePolicyToKernelEvent(policy) {
|
|
40
|
+
return {
|
|
41
|
+
kind: "load_governance_policy",
|
|
42
|
+
...(policy.defaultAction ? { default_action: policy.defaultAction } : {}),
|
|
43
|
+
rules: (policy.rules ?? []).map(r => ({ tool_pattern: r.pattern, action: r.action })),
|
|
44
|
+
vetoed_tools: policy.vetoes ?? [],
|
|
45
|
+
rate_limits: (policy.rateLimits ?? []).map(rl => ({
|
|
46
|
+
tool: rl.tool,
|
|
47
|
+
max_calls: rl.maxCalls,
|
|
48
|
+
window_ms: rl.windowMs,
|
|
49
|
+
})),
|
|
50
|
+
constraints: (policy.constraints ?? []).map(c => c.kind === "enum"
|
|
51
|
+
? { kind: "enum", tool: c.tool, path: c.path, values: c.values }
|
|
52
|
+
: c.kind === "range"
|
|
53
|
+
? { kind: "range", tool: c.tool, path: c.path, ...(c.min !== undefined ? { min: c.min } : {}), ...(c.max !== undefined ? { max: c.max } : {}) }
|
|
54
|
+
: { kind: "required", tool: c.tool, path: c.path }),
|
|
55
|
+
};
|
|
56
|
+
}
|