@deepstrike/sdk 0.2.4 → 0.2.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/README.md +305 -69
  2. package/dist/collaboration/modes/creator-verifier.d.ts +0 -3
  3. package/dist/collaboration/modes/creator-verifier.js +2 -6
  4. package/dist/governance.d.ts +36 -0
  5. package/dist/governance.js +22 -0
  6. package/dist/index.d.ts +14 -6
  7. package/dist/index.js +5 -1
  8. package/dist/kernel.d.ts +43 -0
  9. package/dist/memory/agent.d.ts +66 -0
  10. package/dist/memory/agent.js +151 -0
  11. package/dist/memory/protocols.d.ts +56 -0
  12. package/dist/runtime/execution-plane.d.ts +6 -9
  13. package/dist/runtime/execution-plane.js +53 -25
  14. package/dist/runtime/kernel-event-log.d.ts +26 -0
  15. package/dist/runtime/kernel-event-log.js +220 -0
  16. package/dist/runtime/kernel-primitives-dashboard.d.ts +44 -0
  17. package/dist/runtime/kernel-primitives-dashboard.js +135 -0
  18. package/dist/runtime/kernel-step.d.ts +25 -1
  19. package/dist/runtime/kernel-step.js +12 -3
  20. package/dist/runtime/large-result-spool.d.ts +84 -0
  21. package/dist/runtime/large-result-spool.js +167 -0
  22. package/dist/runtime/os-profile.d.ts +30 -0
  23. package/dist/runtime/os-profile.js +71 -0
  24. package/dist/runtime/os-snapshot.d.ts +35 -0
  25. package/dist/runtime/os-snapshot.js +128 -0
  26. package/dist/runtime/runner.d.ts +79 -9
  27. package/dist/runtime/runner.js +463 -135
  28. package/dist/runtime/session-log.d.ts +118 -4
  29. package/dist/runtime/session-log.js +15 -4
  30. package/dist/runtime/sub-agent-orchestrator.d.ts +2 -2
  31. package/dist/runtime/sub-agent-orchestrator.js +14 -23
  32. package/dist/types/agent.d.ts +12 -3
  33. package/dist/types/agent.js +18 -0
  34. package/dist/types.d.ts +22 -0
  35. package/package.json +2 -2
package/README.md CHANGED
@@ -1,6 +1,8 @@
1
1
  # DeepStrike Node.js SDK
2
2
 
3
- Runtime framework built on a Rust kernel. The kernel handles loop control, context compression, skill routing, governance, signal prioritization — the SDK handles all I/O.
3
+ Runtime framework built on a Rust kernel. The kernel owns loop control, context compression, governance, signal routing, and memory paging — the SDK owns all I/O (LLM calls, tool execution, disk, long-term memory).
4
+
5
+ Node.js is the reference SDK for the **Agent OS native profile**: declarative governance and in-kernel signal routing are enabled by default on every run.
4
6
 
5
7
  ## Install
6
8
 
@@ -24,9 +26,9 @@ Pre-built native addons are available for the following platforms:
24
26
  | Linux ARM64 (musl / Alpine) | `@deepstrike/core-linux-arm64-musl` |
25
27
  | Windows x64 | `@deepstrike/core-win32-x64-msvc` |
26
28
 
27
- The correct platform package is selected and installed automatically via `optionalDependencies`. No postinstall download is required.
29
+ The correct platform package is selected automatically via `optionalDependencies`.
28
30
 
29
- > **Note:** `@deepstrike/core` is the low-level native addon package and is not intended for direct use. It is an internal dependency automatically managed by `@deepstrike/sdk`. Direct installation is only relevant when building from Rust source.
31
+ > **Note:** `@deepstrike/core` is the low-level N-API binding and is managed as an internal dependency of `@deepstrike/sdk`. When developing against a local kernel build, run `npm run test:local-core` from this directory to rebuild the native module from `../crates/deepstrike-node`.
30
32
 
31
33
  ---
32
34
 
@@ -65,14 +67,14 @@ const result = await collectText(runner.run({
65
67
  console.log(result)
66
68
  ```
67
69
 
68
- Same-session conversation continuity is explicit via `sessionId`:
70
+ Same-session continuity is explicit via `sessionId`:
69
71
 
70
72
  ```typescript
71
73
  await collectText(runner.run({ sessionId: "chat-1", goal: "My name is Ada." }))
72
74
  const reply = await collectText(runner.run({ sessionId: "chat-1", goal: "What is my name?" }))
73
75
  ```
74
76
 
75
- Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when event replay should survive restarts. `wake(sessionId)` resumes from the event log without inserting a duplicate user start event.
77
+ Use `InMemorySessionLog` for process-local sessions or `FileSessionLog` when replay should survive restarts. `wake(sessionId)` resumes from the event log without inserting a duplicate `run_started` event.
76
78
 
77
79
  Streaming:
78
80
 
@@ -87,6 +89,62 @@ for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize R
87
89
 
88
90
  ---
89
91
 
92
+ ## Architecture
93
+
94
+ ```text
95
+ ┌─────────────────────────────────────────────────────────┐
96
+ │ RuntimeRunner (Layer 1.5) │
97
+ │ LLMProvider · ExecutionPlane · SessionLog · DreamStore │
98
+ └───────────────────────────┬─────────────────────────────┘
99
+ │ step(JSON event) ↔ actions / observations
100
+ ┌───────────────────────────▼─────────────────────────────┐
101
+ │ @deepstrike/core KernelRuntime │
102
+ │ P1 Syscall · P2 Sched · P3 MM · Proc · IPC │
103
+ └─────────────────────────────────────────────────────────┘
104
+ ```
105
+
106
+ The runner drives a single loop:
107
+
108
+ 1. Kernel returns an **action** — `call_provider`, `execute_tool`, `evaluate_milestone`, or `done`.
109
+ 2. SDK executes the action (stream LLM, run tools, call milestone verifier).
110
+ 3. SDK feeds the result back as a kernel **event** (`provider_result`, `tool_results`, …).
111
+ 4. Kernel **observations** (compression, page-out, spool, signals, …) are drained into `SessionLog`.
112
+
113
+ Kernel session events carry an optional `category` tag (`syscall` · `sched` · `mm` · `proc` · `ipc`) for diagnostics and OS snapshot rebuilds.
114
+
115
+ ### What Agent OS gives you
116
+
117
+ The mechanisms above are not internal refactors — they change what you can build without custom runner code:
118
+
119
+ **Kernel-mediated runtime (M0–M4)**
120
+ Tool calls, spawns, compression, and signals pass through one kernel gate with an explicit lifecycle (Ready / Running / Blocked / Suspended). You implement I/O; the kernel decides *when* and *whether*. Node, Python, and Rust share the same decision path, so `wake(sessionId)` and cross-language tooling see consistent behavior.
121
+
122
+ **Longer, sturdier sessions (Layer-1 spool + semantic page-out)**
123
+ Oversized tool results (> 50 KB) stay in context as a preview plus a `.spool/` reference — the model reads the full payload on demand via ordinary file tools. When pressure triggers semantic eviction, the SDK summarizes archived content into `DreamStore` and satisfies `page_in_requested` on the way back in. Long tasks survive token pressure instead of failing mid-run.
124
+
125
+ **Safety and governance by default (OS native profile)**
126
+ Every run loads declarative `governancePolicy` (deny / ask_user / rate-limit / param rules) and in-kernel signal routing (`attentionPolicy`, default queue 64). Dangerous tools, external interrupts, and approval flows are policy — not ad-hoc `if` checks in your handlers.
127
+
128
+ **Long-term memory as syscalls (Phase-7)**
129
+ `writeMemory` and `queryMemory` run outside the main tool loop: kernel validation before `DreamStore.commit`, search → `selectMemories` → `memory_retrieval_result` on query. Failed writes emit `memory_validation_failed` for audit; good memory is durable without polluting history.
130
+
131
+ **Multi-agent and multi-signal orchestration**
132
+ Sub-agents register in the kernel process table (`agent_process_changed`); parent runs suspend explicitly until `sub_agent_completed`. Signals get disposition (Interrupt / Queue / Observe / Dropped) in-kernel, so gateways, cron, and heartbeats compose with the main loop instead of racing it.
133
+
134
+ **Observable like an OS log**
135
+ Spool, page-out, signals, processes, budgets, and memory events land in `SessionLog` with categories. Rebuild an OS snapshot (`pageOutCount`, `spoolCount`, `processByAgent`, memory counters) from one event stream — replay still strips audit events when reconstructing LLM messages.
136
+
137
+ | You need… | Use… |
138
+ |---|---|
139
+ | Policy before tools run | `governancePolicy` (default: allow-all native profile) |
140
+ | External interrupts | `signalSource` + in-kernel `attentionPolicy` |
141
+ | Huge tool output | Automatic Layer-1 spool; optional custom `resultSpool` |
142
+ | Durable recall across runs | `DreamStore` + semantic `page_out` via `dreamSummarizer` |
143
+ | Programmatic memory I/O | `runner.writeMemory()` / `runner.queryMemory()` |
144
+ | Debug / compliance | `SessionLog` events + OS snapshot helpers |
145
+
146
+ ---
147
+
90
148
  ## Providers
91
149
 
92
150
  | Class | Backend | Notes |
@@ -103,7 +161,7 @@ for await (const event of runner.run({ sessionId: "readme-1", goal: "Summarize R
103
161
 
104
162
  All providers accept `RetryConfig` for exponential backoff and share a `CircuitBreaker`.
105
163
 
106
- `extensions` are forwarded by every provider in both `complete()` and `stream()` while SDK-owned structural fields such as `model`, `messages`, `tools`, and streaming flags remain protected. Provider-specific controls still keep their native spellings: for example Anthropic `thinking` / `betas`, OpenAI Responses `reasoning`, Gemini `generationConfig`, Ollama `think` / `options`, DeepSeek `thinking` + `reasoningEffort`, and Qwen `enableThinking` + `thinkingBudget`.
164
+ `extensions` are forwarded by every provider in both `complete()` and `stream()` while SDK-owned structural fields such as `model`, `messages`, `tools`, and streaming flags remain protected.
107
165
 
108
166
  OpenAI can also be selected through the provider catalog:
109
167
 
@@ -138,39 +196,120 @@ const runner = new RuntimeRunner({
138
196
  ```
139
197
 
140
198
  - `memory(query)` / `knowledge(query)` meta-tool results → **history** (tool results)
141
- - External signals → **Slot 3** via `push_signal()`, cleared after each render
199
+ - Inbound signals are routed by the in-kernel attention policy and rendered into **Slot 3**
142
200
  - Anthropic: Slots 1–2 get separate `cache_control` breakpoints
143
201
 
144
- Full reference: [docs/context-partition-compression.md](../docs/context-partition-compression.md)
202
+ Full reference: [docs/concepts/context-slots-compression.md](../docs/concepts/context-slots-compression.md)
145
203
 
146
204
  ---
147
205
 
148
206
  ## Runtime options
149
207
 
150
208
  ```typescript
151
- const plane = new LocalExecutionPlane()
209
+ import {
210
+ DEFAULT_NATIVE_GOVERNANCE_POLICY,
211
+ DEFAULT_NATIVE_ATTENTION_POLICY,
212
+ } from "@deepstrike/sdk"
213
+
152
214
  const runner = new RuntimeRunner({
153
215
  provider,
154
216
  executionPlane: plane,
155
217
  sessionLog: new FileSessionLog(".deepstrike/sessions"),
156
- maxTokens: 4096, // context window size
157
- maxTurns: 25, // max turns (default 25)
158
- timeoutMs: 60_000, // timeout in ms
159
- extensions: { temperature: 0.1 }, // provider-native controls, passed through to the LLM
160
- skillDir: "./skills", // skill .md files directory
161
- knowledgeSource: myKS, // KnowledgeSource implementation
162
- signalSource: rx, // SignalSource for external signals
163
- dreamStore: myStore, // DreamStore for long-term memory
164
- agentId: "my-agent", // required with dreamStore for memory meta-tool
165
- initialMemory: ["..."], // preloaded blocks → Slot 2 (systemKnowledge)
166
- subAgentHarness: { // optional: sub-agents run through HarnessLoop
167
- evalProvider,
168
- maxAttempts: 3,
218
+
219
+ // Scheduler budget
220
+ maxTokens: 128_000,
221
+ maxTurns: 25,
222
+ timeoutMs: 60_000,
223
+ schedulerBudget: { maxWallMs: 300_000 },
224
+
225
+ // Resource quotas (M2) — enforced at the kernel syscall trap. Opt-in; omit for unbounded.
226
+ resourceQuota: {
227
+ maxConcurrentSubagents: 4, // deny spawn while at cap
228
+ maxSpawnDepth: 2, // deny spawn past nesting depth
229
+ memoryWritesPerWindow: { maxWrites: 20, windowMs: 60_000 }, // rate-limit writeMemory
169
230
  },
170
- governance: gov, // Governance pipeline instance
231
+
232
+ // Long-term memory policy (set_memory_policy) — opt-in, kernel-enforced; omit for defaults.
233
+ memoryPolicy: {
234
+ memoryPath: "./.memory", // where the SDK persists/scans memories (SDK-consumed)
235
+ staleWarningDays: 30, // flag recalled memories older than this (SDK-consumed)
236
+ retrievalTopK: 5, // kernel caps query_memory requested_k to this
237
+ validationEnabled: true, // false → admit writes without validation
238
+ maxContentBytes: 10_000, // override write_memory content-size limit
239
+ maxNameLength: 100, // override write_memory name-length limit
240
+ },
241
+
242
+ // Agent OS native profile (defaults shown)
243
+ governancePolicy: DEFAULT_NATIVE_GOVERNANCE_POLICY,
244
+ attentionPolicy: DEFAULT_NATIVE_ATTENTION_POLICY, // SignalRouter queue size 64
245
+
246
+ // Host I/O
247
+ extensions: { temperature: 0.1 },
248
+ skillDir: "./skills",
249
+ knowledgeSource: myKS,
250
+ signalSource: gw,
251
+ dreamStore: myStore,
252
+ agentId: "my-agent",
253
+ initialMemory: ["..."],
254
+
255
+ // Memory paging & compression (SDK-side I/O)
256
+ compressionStore: archiveStore, // persist compressed transcript slices
257
+ asyncSummarizer: mySummarizer, // upgrade rule-based compression summaries
258
+ dreamProvider: dreamLlm, // LLM for idle dream() synthesis
259
+ dreamSummarizer: myDreamSummarizer, // LLM for semantic page_out → DreamStore
260
+
261
+ // Sub-agents
262
+ runSpec: { role: "orchestrator", isolation: "process" },
263
+ milestoneContract: myContract,
264
+ milestonePolicy: "require_verifier",
265
+ onMilestoneEvaluate: async ({ phaseId, criteria }) => ({ passed: true, phaseId }),
266
+ subAgentHarness: { evalProvider, maxAttempts: 3 },
267
+
268
+ // Governance UX (AskUser path)
269
+ onPermissionRequest: async (req) => ({ approved: true }),
270
+
271
+ // Diagnostics
272
+ enableDiagnosticsDashboard: true, // CLI view grouped by Syscall / Sched / MM
171
273
  })
172
274
  ```
173
275
 
276
+ | Option | Purpose |
277
+ |--------|---------|
278
+ | `governancePolicy` | Declarative deny / ask_user / rate-limit / param rules loaded into the kernel before `start_run` |
279
+ | `attentionPolicy` | In-kernel signal router queue size (default 64) |
280
+ | `resourceQuota` | M2 declarative limits — `maxConcurrentSubagents` / `maxSpawnDepth` / `memoryWritesPerWindow` — enforced at the kernel syscall trap (`set_resource_quota`); over-quota spawns roll back, over-rate writes surface as `memory_validation_failed` |
281
+ | `memoryPolicy` | Long-term memory config sent as `set_memory_policy` and **kernel-enforced**: `validationEnabled: false` admits writes without validation, `maxContentBytes` / `maxNameLength` override validation limits, `retrievalTopK` caps `query_memory` breadth; `memoryPath` / `staleWarningDays` are SDK-consumed (requires `dreamStore` + `agentId` to enable memory) |
282
+ | `onPermissionRequest` | Resolves `tool_gated` + `suspended` → kernel `resume` with approved/denied call IDs |
283
+ | `compressionStore` | Writes archived messages on `compressed` observations |
284
+ | `asyncSummarizer` | Background LLM summary after compression; stored as `summary_upgraded` |
285
+ | `dreamSummarizer` | Summarizes `page_out { tier_hint: "semantic" }` into `DreamStore` during a run |
286
+ | `dreamProvider` | Separate LLM for `dream()` idle consolidation (falls back to `provider`) |
287
+
288
+ Rebuild an OS diagnostics snapshot from session events:
289
+
290
+ ```typescript
291
+ import { rebuildOsSnapshotFromSessionEvents } from "@deepstrike/sdk"
292
+
293
+ const events = (await sessionLog.read(sessionId)).map(e => e.event)
294
+ const snap = rebuildOsSnapshotFromSessionEvents(events)
295
+ // snap.pageOutCount, snap.spoolCount, snap.signals, snap.processByAgent, …
296
+ ```
297
+
298
+ ---
299
+
300
+ ## Large result spool (Layer 1)
301
+
302
+ When a single tool result exceeds **50 KB**, the kernel keeps a short preview in context and emits `large_result_spooled`. The SDK writes the full payload to `.spool/` under the process cwd (SHA-256 keyed files) and logs `spool_ref` in the session.
303
+
304
+ The model can retrieve full content via ordinary read tools — `LocalExecutionPlane` transparently resolves paths under `.spool/`:
305
+
306
+ ```typescript
307
+ // Kernel context shows a preview + spool reference.
308
+ // LLM calls read_file({ path: ".spool/abc123…" }) → full content returned.
309
+ ```
310
+
311
+ No configuration is required; customize the directory by passing a `resultSpool` instance when constructing `RuntimeRunner` (see tests under `tests/runtime/large-result-spool.test.ts`).
312
+
174
313
  ---
175
314
 
176
315
  ## Tools
@@ -179,10 +318,28 @@ const runner = new RuntimeRunner({
179
318
  import { tool, readFile } from "@deepstrike/sdk"
180
319
 
181
320
  plane.register(tool("search", "Search.", schema, async (args) => ...))
182
- plane.register(readFile) // built-in: read files from disk
321
+ plane.register(readFile) // built-in: read files from disk (also resolves .spool/ refs)
183
322
  plane.unregister("search")
184
323
  ```
185
324
 
325
+ Execution planes:
326
+
327
+ | Plane | Use case |
328
+ |-------|----------|
329
+ | `LocalExecutionPlane` | In-process tools (default) |
330
+ | `FilteredExecutionPlane` | Capability-filtered sub-agent tools |
331
+ | `ProcessSandboxPlane` | OS subprocess isolation |
332
+ | `McpProxyPlane` | MCP server tools |
333
+ | `RemoteVpcPlane` | Remote execution |
334
+
335
+ Mount capabilities on an active run:
336
+
337
+ ```typescript
338
+ runner.mountTool(schema)
339
+ runner.mountSkill("summarize", "Summarize text")
340
+ runner.unmountCapability("tool", "search")
341
+ ```
342
+
186
343
  ---
187
344
 
188
345
  ## Skills
@@ -214,9 +371,11 @@ effort: 1
214
371
 
215
372
  ## Knowledge
216
373
 
217
- Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand. **Runtime retrieval results land in history** as tool results.
374
+ Implement `KnowledgeSource` to connect any RAG system. The kernel injects a `knowledge` meta-tool that the LLM calls on demand. Runtime retrieval results land in **history** as tool results.
218
375
 
219
- To inject durable knowledge at startup (Slot 2, cacheable on Anthropic), use `initialMemory` or kernel `add_knowledge_message`.
376
+ To inject durable knowledge at startup (Slot 2, cacheable on Anthropic), use `initialMemory` or `runner.pushKnowledge()`.
377
+
378
+ Before tool execution the kernel may emit `page_in_requested`; the SDK satisfies it from `DreamStore`, `KnowledgeSource`, and a local semantic page-out cache, then feeds `page_in` back to the kernel.
220
379
 
221
380
  ```typescript
222
381
  const runner = new RuntimeRunner({
@@ -238,7 +397,7 @@ const runner = new RuntimeRunner({
238
397
 
239
398
  ### WorkingMemory (SDK-side scratch pad)
240
399
 
241
- `WorkingMemory` is an SDK helper — not the kernel `working` partition (removed). Kernel task state lives in `task_state` and renders into Slot 3 (`turns[0]`).
400
+ `WorkingMemory` is an SDK helper — not the kernel working partition. Kernel task state lives in `task_state` and renders into Slot 3 (`turns[0]`).
242
401
 
243
402
  ```typescript
244
403
  import { WorkingMemory } from "@deepstrike/sdk"
@@ -248,7 +407,7 @@ mem.get("step") // 1
248
407
  mem.clear()
249
408
  ```
250
409
 
251
- ### DreamStore (long-term memory + dreaming pipeline)
410
+ ### DreamStore (long-term memory)
252
411
 
253
412
  ```typescript
254
413
  import type { DreamStore } from "@deepstrike/sdk"
@@ -266,32 +425,91 @@ const runner = new RuntimeRunner({
266
425
  sessionLog: new FileSessionLog(".deepstrike/sessions"),
267
426
  maxTokens: 4096,
268
427
  dreamStore: new MyStore(),
269
- agentId: "my-agent", // enables `memory` meta-tool
428
+ agentId: "my-agent", // enables `memory` meta-tool + semantic page-out archival
270
429
  })
430
+ ```
431
+
432
+ Three memory paths:
271
433
 
272
- // In-session: LLM calls memory(query) → DreamStore.search() → history tool result
273
- // Preload: initialMemory → Slot 2 (systemKnowledge)
274
- // Post-session: trigger memory consolidation
434
+ | Path | When | What happens |
435
+ |------|------|--------------|
436
+ | In-session `memory(query)` | LLM calls meta-tool | `DreamStore.search()` → history tool result |
437
+ | `initialMemory` | Run start | Injected into Slot 2 (`systemKnowledge`) |
438
+ | Semantic `page_out` | Kernel evicts with `tier_hint: "semantic"` | SDK summarizes via `dreamSummarizer` / `dreamProvider` → `DreamStore.commit()` |
439
+ | `dream(agentId)` | Explicit idle call | `IdlePipeline` batch-consolidates past sessions |
440
+
441
+ ```typescript
442
+ // Post-session batch consolidation
275
443
  const result = await runner.dream("my-agent", Date.now())
276
444
  ```
277
445
 
446
+ ### Phase-7 memory syscalls (`writeMemory` / `queryMemory`)
447
+
448
+ Kernel-validated long-term memory I/O outside the main tool loop:
449
+
450
+ ```typescript
451
+ await runner.writeMemory({
452
+ metadata: {
453
+ name: "prefers-small-tests",
454
+ description: "User prefers focused unit tests",
455
+ kind: "feedback",
456
+ created_at: Date.now(),
457
+ updated_at: Date.now(),
458
+ },
459
+ content: "User prefers focused unit tests for SDK behavior.",
460
+ }, { sessionId: "my-session" })
461
+
462
+ const hits = await runner.queryMemory({
463
+ current_context: "Need testing preferences",
464
+ active_tools: [],
465
+ already_surfaced: [],
466
+ top_k: 5,
467
+ }, { sessionId: "my-session" })
468
+ ```
469
+
470
+ Session events: `memory_written`, `memory_queried`, `memory_validation_failed`, `memory_retrieval_result`.
471
+
278
472
  ---
279
473
 
280
474
  ## Governance
281
475
 
282
- ### SDK PermissionManager
476
+ ### In-kernel declarative policy (preferred)
477
+
478
+ Every run loads `governancePolicy` into the kernel via `load_governance_policy`. The kernel enforces rules **before** tools execute:
283
479
 
284
480
  ```typescript
285
- import { PermissionManager, PermissionMode } from "@deepstrike/sdk"
481
+ import type { GovernancePolicy } from "@deepstrike/sdk"
482
+
483
+ const policy: GovernancePolicy = {
484
+ rules: [
485
+ { pattern: "read_file", action: "allow" },
486
+ { pattern: "write_file", action: "ask_user" },
487
+ { pattern: "run_command", action: "ask_user" },
488
+ { pattern: "*", action: "deny" },
489
+ ],
490
+ rateLimits: [{ tool: "api_call", maxCalls: 10, windowMs: 60_000 }],
491
+ }
286
492
 
287
- const pm = new PermissionManager(PermissionMode.DEFAULT)
288
- pm.grant("fs", "read")
289
- pm.grantWithApproval("db", "write", "Needs DBA approval")
290
- pm.revoke("db", "drop")
291
- pm.evaluate("fs", "read") // { allowed: true, ... }
493
+ const runner = new RuntimeRunner({
494
+ provider,
495
+ executionPlane: plane,
496
+ sessionLog,
497
+ governancePolicy: policy,
498
+ onPermissionRequest: async (req) => {
499
+ console.log(`Approve ${req.toolName}?`, req.arguments)
500
+ return { approved: true }
501
+ },
502
+ })
292
503
  ```
293
504
 
294
- ### Kernel Governance (full pipeline)
505
+ - `deny` → tool rejected with `tool_denied`
506
+ - `ask_user` → `tool_gated` + `suspended`; resolve via `onPermissionRequest`, then kernel `resume`
507
+
508
+ Default when omitted: allow-all (`DEFAULT_NATIVE_GOVERNANCE_POLICY`).
509
+
510
+ ### Standalone Governance class
511
+
512
+ `Governance` wraps the native governance evaluator for SDK-side use (tests, custom gates). It is **not** wired automatically into `RuntimeRunner` — use `governancePolicy` for run-time enforcement.
295
513
 
296
514
  ```typescript
297
515
  import { Governance } from "@deepstrike/sdk"
@@ -299,45 +517,65 @@ import { Governance } from "@deepstrike/sdk"
299
517
  const gov = new Governance("allow")
300
518
  gov.addPermissionRule("danger.*", "deny")
301
519
  gov.blockTool("rm_rf")
302
- gov.setRateLimit("api_call", 10, 60_000)
303
- gov.requireParam("write_file", "path")
304
- gov.allowParamValues("set_mode", "mode", ["read", "write"])
305
- gov.limitParamRange("sleep", "seconds", 0, 10)
306
-
307
- const runner = new RuntimeRunner({
308
- provider,
309
- executionPlane: plane,
310
- sessionLog: new FileSessionLog(".deepstrike/sessions"),
311
- maxTokens: 4096,
312
- governance: gov,
313
- })
314
- // Every tool call goes through: Permission → Veto → RateLimit → Constraint → Audit
520
+ gov.evaluate("read_file", '{"path":"x"}')
315
521
  ```
316
522
 
523
+ ### SDK PermissionManager
524
+
525
+ `PermissionManager` is a separate SDK-side permission layer for apps that manage their own approval UX outside the kernel loop.
526
+
317
527
  ---
318
528
 
319
529
  ## Signals
320
530
 
531
+ Inbound signals are routed by the in-kernel attention policy (default queue size 64):
532
+
533
+ | Urgency | Typical disposition |
534
+ |---------|-------------------|
535
+ | `critical` / `high` | `interrupt_now` — may yield a new `call_provider` action |
536
+ | `normal` / `low` | `queue` — buffered; no action until dequeued |
537
+ | queue full | `dropped` |
538
+
321
539
  ```typescript
322
540
  import { SignalGateway, ScheduledPrompt } from "@deepstrike/sdk"
323
541
 
324
542
  const gw = new SignalGateway()
325
543
  gw.schedule(new ScheduledPrompt("standup", Date.now() + 3600_000))
326
- gw.ingest({ kind: "interrupt", urgency: "critical", payload: {} })
544
+ gw.ingest({ kind: "alert", urgency: "normal", payload: { goal: "Check deploy" } })
327
545
 
328
546
  const runner = new RuntimeRunner({
329
547
  provider,
330
548
  executionPlane: plane,
331
- sessionLog: new FileSessionLog(".deepstrike/sessions"),
332
- maxTokens: 4096,
549
+ sessionLog,
333
550
  signalSource: gw,
551
+ attentionPolicy: { maxQueueSize: 64 },
334
552
  })
335
- // kind="interrupt" → immediately stops the running runner
336
553
 
337
- runner.interrupt() // also works directly
554
+ runner.interrupt() // cooperative abort → kernel timeout path
338
555
  gw.destroy()
339
556
  ```
340
557
 
558
+ Each routed signal produces a `signal_disposed` session event (`category: "ipc"`).
559
+
560
+ ---
561
+
562
+ ## Sub-agents
563
+
564
+ Spawn isolated child agents through the kernel process table:
565
+
566
+ ```typescript
567
+ for await (const evt of runner.spawnSubAgent({
568
+ role: "researcher",
569
+ isolation: "process",
570
+ goal: "Find three sources on topic X",
571
+ criteria: ["At least 3 URLs"],
572
+ })) {
573
+ if (evt.type === "done") console.log(evt.status)
574
+ }
575
+ ```
576
+
577
+ Requires an active parent run (`run()` / `wake()` in progress). The kernel emits `agent_process_changed`; the default `SubAgentOrchestrator` runs the child with a filtered execution plane and feeds `sub_agent_completed` back.
578
+
341
579
  ---
342
580
 
343
581
  ## Harness (evaluation framework)
@@ -345,30 +583,20 @@ gw.destroy()
345
583
  ```typescript
346
584
  import { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "@deepstrike/sdk"
347
585
 
348
- // 1. SinglePass — run once, always passes
349
586
  const outcome = await new SinglePassHarness(runner).run({ goal: "Say hello" })
350
587
 
351
- // 2. EvalLoop — retry until QualityGate passes
352
588
  const harness = new EvalLoopHarness(runner, {
353
589
  async evaluate(_req, out) { return out.result.includes("hello") },
354
590
  }, 3)
355
591
 
356
- // 3. HarnessLoop — LLM-as-judge with feedback injection + skill extraction
357
592
  const loop = new HarnessLoop(runner, evalProvider, { maxAttempts: 3, skillDir: "./skills" })
358
593
 
359
- // Sub-agents: pass subAgentHarness on RuntimeRunner to auto-evaluate spawned children
360
594
  const runnerWithHarness = new RuntimeRunner({
361
595
  provider,
362
596
  executionPlane: plane,
363
597
  sessionLog,
364
598
  subAgentHarness: { evalProvider, maxAttempts: 3 },
365
599
  })
366
- for await (const event of loop.runStreaming({
367
- goal: "Write a haiku",
368
- criteria: [{ text: "Must be 3 lines", required: true }],
369
- })) {
370
- if (event.type === "done") console.log(event.verdict.passed, event.verdict.feedback)
371
- }
372
600
  ```
373
601
 
374
602
  ---
@@ -387,4 +615,12 @@ for await (const event of loop.runStreaming({
387
615
  | `done` | `iterations`, `totalTokens`, `status` |
388
616
  | `error` | `message` |
389
617
 
390
- `status`: `completed` · `max_turns` · `token_budget` · `timeout` · `user_abort` · `error`
618
+ `status`: `completed` · `max_turns` · `token_budget` · `timeout` · `user_abort` · `error` · `milestone_pending`
619
+
620
+ ---
621
+
622
+ ## Further reading
623
+
624
+ - [SDK OS parity matrix](../docs/sdk-os-parity.md)
625
+ - [Kernel ABI reference](../docs/reference/kernel-abi.md)
626
+ - [Context slots & compression](../docs/concepts/context-slots-compression.md)
@@ -35,8 +35,6 @@ export declare class CreatorVerifierMode {
35
35
  maxAttempts?: number;
36
36
  /** Stable orchestration session for kernel lineage audit. */
37
37
  coordinatorSessionId?: string;
38
- /** Opt out of kernel spawn path and use legacy independent runner sessions. */
39
- useLegacyRunners?: boolean;
40
38
  });
41
39
  run(contract: VerificationContract): Promise<ContractOutcome>;
42
40
  /** Aggregate drift metrics across all runs through this mode instance. */
@@ -64,7 +62,6 @@ export declare class OrchestrationMode {
64
62
  constructor(pool: AgentPool, options?: {
65
63
  maxAttempts?: number;
66
64
  coordinatorSessionId?: string;
67
- useLegacyRunners?: boolean;
68
65
  });
69
66
  run(goal: string): Promise<ContractOutcome & {
70
67
  contract: VerificationContract;
@@ -30,9 +30,7 @@ export class CreatorVerifierMode {
30
30
  }
31
31
  async run(contract) {
32
32
  this._total++;
33
- if (!this.options.useLegacyRunners) {
34
- this.pool.ensureCoordinator(this.options.coordinatorSessionId);
35
- }
33
+ this.pool.ensureCoordinator(this.options.coordinatorSessionId);
36
34
  const harness = new ContractDrivenHarness(this.pool, contract, {
37
35
  maxAttempts: this.options.maxAttempts ?? 3,
38
36
  });
@@ -81,9 +79,7 @@ export class OrchestrationMode {
81
79
  this.inner = new CreatorVerifierMode(pool, options);
82
80
  }
83
81
  async run(goal) {
84
- if (!this.options.useLegacyRunners) {
85
- this.pool.ensureCoordinator(this.options.coordinatorSessionId);
86
- }
82
+ this.pool.ensureCoordinator(this.options.coordinatorSessionId);
87
83
  // Step 1: orchestrator produces a VerificationContract
88
84
  const contractJson = await this.pool.orchestrate(goal);
89
85
  const contract = this._parseContract(contractJson, goal);
@@ -15,3 +15,39 @@ export declare class Governance {
15
15
  evaluate(toolName: string, argsJson: string): GovernanceVerdict;
16
16
  }
17
17
  export type { GovernanceVerdict };
18
+ type GovernancePolicyAction = "allow" | "deny" | "ask_user";
19
+ export interface GovernancePolicy {
20
+ defaultAction?: GovernancePolicyAction;
21
+ rules?: {
22
+ pattern: string;
23
+ action: GovernancePolicyAction;
24
+ }[];
25
+ vetoes?: string[];
26
+ rateLimits?: {
27
+ tool: string;
28
+ maxCalls: number;
29
+ windowMs: number;
30
+ }[];
31
+ constraints?: GovernanceConstraint[];
32
+ }
33
+ export type GovernanceConstraint = {
34
+ kind: "required";
35
+ tool: string;
36
+ path: string;
37
+ } | {
38
+ kind: "enum";
39
+ tool: string;
40
+ path: string;
41
+ values: string[];
42
+ } | {
43
+ kind: "range";
44
+ tool: string;
45
+ path: string;
46
+ min?: number;
47
+ max?: number;
48
+ };
49
+ /**
50
+ * Convert a declarative {@link GovernancePolicy} into the `load_governance_policy`
51
+ * kernel event payload (snake_case wire fields). Pure — no side effects.
52
+ */
53
+ export declare function governancePolicyToKernelEvent(policy: GovernancePolicy): Record<string, unknown>;
@@ -32,3 +32,25 @@ export class Governance {
32
32
  return this.inner.evaluate(toolName, argsJson);
33
33
  }
34
34
  }
35
+ /**
36
+ * Convert a declarative {@link GovernancePolicy} into the `load_governance_policy`
37
+ * kernel event payload (snake_case wire fields). Pure — no side effects.
38
+ */
39
+ export function governancePolicyToKernelEvent(policy) {
40
+ return {
41
+ kind: "load_governance_policy",
42
+ ...(policy.defaultAction ? { default_action: policy.defaultAction } : {}),
43
+ rules: (policy.rules ?? []).map(r => ({ tool_pattern: r.pattern, action: r.action })),
44
+ vetoed_tools: policy.vetoes ?? [],
45
+ rate_limits: (policy.rateLimits ?? []).map(rl => ({
46
+ tool: rl.tool,
47
+ max_calls: rl.maxCalls,
48
+ window_ms: rl.windowMs,
49
+ })),
50
+ constraints: (policy.constraints ?? []).map(c => c.kind === "enum"
51
+ ? { kind: "enum", tool: c.tool, path: c.path, values: c.values }
52
+ : c.kind === "range"
53
+ ? { kind: "range", tool: c.tool, path: c.path, ...(c.min !== undefined ? { min: c.min } : {}), ...(c.max !== undefined ? { max: c.max } : {}) }
54
+ : { kind: "required", tool: c.tool, path: c.path }),
55
+ };
56
+ }