@deepstrike/sdk 0.2.18 → 0.2.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -15,6 +15,11 @@ export { LocalExecutionPlane } from "./runtime/execution-plane.js";
15
15
  export type { ExecutionPlane, RunContext } from "./runtime/execution-plane.js";
16
16
  export { InMemorySessionLog, FileSessionLog } from "./runtime/session-log.js";
17
17
  export type { SessionLog, SessionEvent } from "./runtime/session-log.js";
18
+ export { ReplayProvider } from "./runtime/replay-provider.js";
19
+ export type { ReplayProviderOpts } from "./runtime/replay-provider.js";
20
+ export { extractRecordedMessages } from "./runtime/replay-fixture.js";
21
+ export { judge, buildEvalMessages, parseVerdict, verdictOutputSchema } from "./runtime/eval.js";
22
+ export type { Criterion, Verdict, VerdictDetail, JudgeArgs } from "./runtime/eval.js";
18
23
  export { DEFAULT_NATIVE_ATTENTION_POLICY, DEFAULT_NATIVE_GOVERNANCE_POLICY, assertNativeProfile, osProfile, } from "./runtime/os-profile.js";
19
24
  export type { NativeOsProfile, OsProfileId } from "./runtime/os-profile.js";
20
25
  export { rebuildOsSnapshotFromSessionEvents, sessionLogHasRequiredCategories, } from "./runtime/os-snapshot.js";
package/dist/index.js CHANGED
@@ -9,6 +9,9 @@ export { FilteredExecutionPlane } from "./runtime/filtered-plane.js";
9
9
  export { SubAgentOrchestrator, defaultSubAgentOrchestrator, spawnStandalone } from "./runtime/sub-agent-orchestrator.js";
10
10
  export { LocalExecutionPlane } from "./runtime/execution-plane.js";
11
11
  export { InMemorySessionLog, FileSessionLog } from "./runtime/session-log.js";
12
+ export { ReplayProvider } from "./runtime/replay-provider.js";
13
+ export { extractRecordedMessages } from "./runtime/replay-fixture.js";
14
+ export { judge, buildEvalMessages, parseVerdict, verdictOutputSchema } from "./runtime/eval.js";
12
15
  export { DEFAULT_NATIVE_ATTENTION_POLICY, DEFAULT_NATIVE_GOVERNANCE_POLICY, assertNativeProfile, osProfile, } from "./runtime/os-profile.js";
13
16
  export { rebuildOsSnapshotFromSessionEvents, sessionLogHasRequiredCategories, } from "./runtime/os-snapshot.js";
14
17
  export { categoryForKind, kernelObservationToSessionEvent } from "./runtime/kernel-event-log.js";
@@ -0,0 +1,60 @@
1
+ /**
2
+ * `judge()` — one-shot quality scoring against a goal + criteria using the kernel's `gen_eval`.
3
+ *
4
+ * Wraps the three kernel free functions `buildEvalMessages` / `parseVerdict` / `verdictOutputSchema`
5
+ * (folded out of the old EvalPipeline class in 0.5.0) into a small typed surface that's safe to
6
+ * call from a benchmark harness, a CI gate, or any caller that just wants "does this result meet
7
+ * the criteria?" without setting up `HarnessLoop`.
8
+ *
9
+ * The judge is a single LLM call: build the eval prompt → stream → parse verdict. No retry loop,
10
+ * no skill extraction, no harness state. Use `HarnessLoop` if you want the retry/refine flow.
11
+ */
12
+ import type { LLMProvider, Message } from "../types.js";
13
+ export interface Criterion {
14
+ /** The criterion text the judge evaluates against. */
15
+ text: string;
16
+ /** When true (default), failing this criterion fails the overall verdict. */
17
+ required?: boolean;
18
+ /** Optional weight for weighted scoring (kernel-defined semantics). */
19
+ weight?: number;
20
+ }
21
+ export interface VerdictDetail {
22
+ criterion: string;
23
+ passed: boolean;
24
+ score: number;
25
+ feedback: string;
26
+ }
27
+ export interface Verdict {
28
+ passed: boolean;
29
+ /** 0..1 — kernel-defined aggregate score. */
30
+ overallScore: number;
31
+ feedback: string;
32
+ details: VerdictDetail[];
33
+ }
34
+ export interface JudgeArgs {
35
+ /** Provider used for the eval LLM call. Often a cheaper model than the main run. */
36
+ provider: LLMProvider;
37
+ /** The task goal the result is being evaluated against. */
38
+ goal: string;
39
+ /** The criteria the judge scores against. */
40
+ criteria: Criterion[];
41
+ /** The agent's result text (final reply, or a structured summary when the run was incomplete). */
42
+ result: string;
43
+ /** Optional abort signal forwarded to provider.stream. */
44
+ signal?: AbortSignal;
45
+ }
46
+ /**
47
+ * Build the kernel's eval prompt for (goal, criteria, result).
48
+ * Exposed in case a caller wants to render the prompt without calling the LLM (e.g., dry-run cost
49
+ * estimation, fixture generation). For the common case, use `judge()`.
50
+ */
51
+ export declare function buildEvalMessages(goal: string, criteria: Criterion[], result: string): Message[];
52
+ /** Parse a Verdict from raw judge-LLM text. Throws on schema mismatch. */
53
+ export declare function parseVerdict(text: string): Verdict;
54
+ /** The JSON Schema the kernel expects judge output to conform to. */
55
+ export declare function verdictOutputSchema(): Record<string, unknown>;
56
+ /**
57
+ * Run one judge pass: render the eval prompt, stream the provider, parse the verdict.
58
+ * Throws when the provider returns no text or returns content that fails verdict parsing.
59
+ */
60
+ export declare function judge(args: JudgeArgs): Promise<Verdict>;
@@ -0,0 +1,55 @@
1
+ /**
2
+ * `judge()` — one-shot quality scoring against a goal + criteria using the kernel's `gen_eval`.
3
+ *
4
+ * Wraps the three kernel free functions `buildEvalMessages` / `parseVerdict` / `verdictOutputSchema`
5
+ * (folded out of the old EvalPipeline class in 0.5.0) into a small typed surface that's safe to
6
+ * call from a benchmark harness, a CI gate, or any caller that just wants "does this result meet
7
+ * the criteria?" without setting up `HarnessLoop`.
8
+ *
9
+ * The judge is a single LLM call: build the eval prompt → stream → parse verdict. No retry loop,
10
+ * no skill extraction, no harness state. Use `HarnessLoop` if you want the retry/refine flow.
11
+ */
12
+ import { getKernel } from "../kernel.js";
13
+ /**
14
+ * Build the kernel's eval prompt for (goal, criteria, result).
15
+ * Exposed in case a caller wants to render the prompt without calling the LLM (e.g., dry-run cost
16
+ * estimation, fixture generation). For the common case, use `judge()`.
17
+ */
18
+ export function buildEvalMessages(goal, criteria, result) {
19
+ return getKernel().buildEvalMessages(goal, criteria.map(c => ({ text: c.text, required: c.required ?? true, weight: c.weight })), result, 1, // attempt
20
+ false);
21
+ }
22
+ /** Parse a Verdict from raw judge-LLM text. Throws on schema mismatch. */
23
+ export function parseVerdict(text) {
24
+ const v = getKernel().parseVerdict(text);
25
+ return {
26
+ passed: v.passed,
27
+ overallScore: v.overallScore,
28
+ feedback: v.feedback,
29
+ details: v.details ?? [],
30
+ };
31
+ }
32
+ /** The JSON Schema the kernel expects judge output to conform to. */
33
+ export function verdictOutputSchema() {
34
+ return JSON.parse(getKernel().verdictOutputSchema(false));
35
+ }
36
+ /**
37
+ * Run one judge pass: render the eval prompt, stream the provider, parse the verdict.
38
+ * Throws when the provider returns no text or returns content that fails verdict parsing.
39
+ */
40
+ export async function judge(args) {
41
+ const msgs = buildEvalMessages(args.goal, args.criteria, args.result);
42
+ const ctx = {
43
+ systemText: msgs.filter(m => m.role === "system").map(m => m.content).join("\n\n"),
44
+ turns: msgs.filter(m => m.role !== "system"),
45
+ };
46
+ let text = "";
47
+ for await (const evt of args.provider.stream(ctx, [], undefined, undefined, args.signal)) {
48
+ if (evt.type === "text_delta")
49
+ text += evt.delta;
50
+ }
51
+ if (!text) {
52
+ throw new Error("judge: provider produced no text");
53
+ }
54
+ return parseVerdict(text);
55
+ }
@@ -24,6 +24,9 @@ export function skillMetadataToKernel(skill) {
24
24
  out.when_to_use = skill.whenToUse;
25
25
  if (skill.effort !== undefined)
26
26
  out.effort = skill.effort;
27
+ // P1-B: forward declared tool ids (additive; omitted when empty so existing skills' wire is unchanged).
28
+ if (skill.allowedTools?.length)
29
+ out.allowed_tools = skill.allowedTools;
27
30
  return out;
28
31
  }
29
32
  export function messageToKernelMessage(message) {
@@ -0,0 +1,26 @@
1
+ /**
2
+ * Fixture helpers for `ReplayProvider`.
3
+ *
4
+ * The canonical persistence shape for a recorded run is a `SessionLog` of `llm_completed` events
5
+ * (already written by the runner on every live run). `extractRecordedMessages` walks such a log and
6
+ * pulls the assistant turns in order, so the fixture is just "a prior session log + the messages
7
+ * the LLM produced". No new on-disk format.
8
+ */
9
+ import type { Message } from "../types.js";
10
+ import type { SessionEvent } from "./session-log.js";
11
+ /**
12
+ * Extract the ordered list of assistant Messages from a recorded session log.
13
+ *
14
+ * Walks `llm_completed` events (which is what the runner appends for every LLM call) and produces
15
+ * one Message per event. Pass the result directly to `new ReplayProvider(messages)`.
16
+ *
17
+ * Accepts both wire shapes the SDK uses interchangeably:
18
+ * - in-memory: `{ toolCalls, tokenCount, providerReplay }` (camelCase)
19
+ * - serialised session-log: `{ tool_calls, token_count, provider_replay }` (snake_case)
20
+ *
21
+ * @param events Session events, in original order. Accepts both `{ event, seq }` (the shape
22
+ * `SessionLog.read()` returns) and a bare `SessionEvent[]`.
23
+ */
24
+ export declare function extractRecordedMessages(events: Array<{
25
+ event: SessionEvent;
26
+ } | SessionEvent>): Message[];
@@ -0,0 +1,51 @@
1
+ /**
2
+ * Fixture helpers for `ReplayProvider`.
3
+ *
4
+ * The canonical persistence shape for a recorded run is a `SessionLog` of `llm_completed` events
5
+ * (already written by the runner on every live run). `extractRecordedMessages` walks such a log and
6
+ * pulls the assistant turns in order, so the fixture is just "a prior session log + the messages
7
+ * the LLM produced". No new on-disk format.
8
+ */
9
+ /**
10
+ * Extract the ordered list of assistant Messages from a recorded session log.
11
+ *
12
+ * Walks `llm_completed` events (which is what the runner appends for every LLM call) and produces
13
+ * one Message per event. Pass the result directly to `new ReplayProvider(messages)`.
14
+ *
15
+ * Accepts both wire shapes the SDK uses interchangeably:
16
+ * - in-memory: `{ toolCalls, tokenCount, providerReplay }` (camelCase)
17
+ * - serialised session-log: `{ tool_calls, token_count, provider_replay }` (snake_case)
18
+ *
19
+ * @param events Session events, in original order. Accepts both `{ event, seq }` (the shape
20
+ * `SessionLog.read()` returns) and a bare `SessionEvent[]`.
21
+ */
22
+ export function extractRecordedMessages(events) {
23
+ const out = [];
24
+ for (const entry of events) {
25
+ const event = isWrapped(entry) ? entry.event : entry;
26
+ if (event.kind !== "llm_completed")
27
+ continue;
28
+ const e = event;
29
+ const tcRaw = (e.toolCalls ?? e.tool_calls);
30
+ const tokenCount = (e.tokenCount ?? e.token_count);
31
+ out.push({
32
+ role: "assistant",
33
+ content: typeof e.content === "string" ? e.content : "",
34
+ ...(Array.isArray(tcRaw) && tcRaw.length > 0
35
+ ? { toolCalls: normalizeToolCalls(tcRaw) }
36
+ : {}),
37
+ ...(tokenCount !== undefined ? { tokenCount } : {}),
38
+ });
39
+ }
40
+ return out;
41
+ }
42
+ function isWrapped(x) {
43
+ return !!x && typeof x === "object" && "event" in x;
44
+ }
45
+ function normalizeToolCalls(tcs) {
46
+ return tcs.map(tc => ({
47
+ id: tc.id,
48
+ name: tc.name,
49
+ arguments: typeof tc.arguments === "string" ? tc.arguments : JSON.stringify(tc.arguments),
50
+ }));
51
+ }
@@ -0,0 +1,69 @@
1
+ /**
2
+ * ReplayProvider — an LLMProvider that emits previously-recorded assistant messages
3
+ * instead of calling a real LLM API.
4
+ *
5
+ * Purpose: deterministic re-runs for benchmarking, CI, and golden regression. Useful when you want
6
+ * to hold the model's behavior constant and measure something else (prompt-size cost Δ across
7
+ * RuntimeOptions variants, codegen of follow-on kernel work, etc.).
8
+ *
9
+ * Distinct from `provider-replay.ts`: that file's `seedProviderReplay` / `peekProviderReplay` is a
10
+ * session-repair *reasoning-content cache* — it preserves `reasoning_content` / `native_blocks` so
11
+ * the model sees its own thinking when context is re-rendered. It does NOT skip LLM calls.
12
+ * `ReplayProvider` is the orthogonal, request-skipping mechanism: it returns recorded responses
13
+ * directly, never hitting an API.
14
+ *
15
+ * Cost-accounting under replay:
16
+ * - `inputTokens` is ESTIMATED from the rendered context this call carries (NOT a recorded value
17
+ * from the original run). That's the point of replay-for-benchmarking: prompt may differ across
18
+ * variants, response is pinned, so a cost Δ purely reflects the prompt change.
19
+ * - `outputTokens` is taken from `message.tokenCount` when present; otherwise estimated from
20
+ * `message.content.length / 4`.
21
+ * - `cacheReadInputTokens` / `cacheCreationInputTokens` are emitted as 0 — replay has no real
22
+ * cache state. Mechanisms whose Δ depends on cache behavior must validate with a live A/B too.
23
+ *
24
+ * Tokenizer: by default a `chars/4` estimator (±20% for English; worse for code/JSON). For tighter
25
+ * numbers plug `opts.tokenizer = tiktokenEncoder` or similar.
26
+ */
27
+ import type { LLMProvider, Message, ProviderDescriptor, ProviderRunState, RenderedContext, StreamEvent, ToolSchema } from "../types.js";
28
+ export interface ReplayProviderOpts {
29
+ /**
30
+ * Maps a rendered-context text payload to a token count. Defaults to `chars / 4`.
31
+ * Pass a real encoder (tiktoken etc.) for accurate cost accounting under replay.
32
+ */
33
+ tokenizer?: (text: string) => number;
34
+ /**
35
+ * Provider descriptor advertised via `descriptor()`. Defaults to a generic
36
+ * `{ provider: "replay", protocol: "replay", ... }` shape. Override when a downstream consumer
37
+ * needs to detect the original provider (e.g., for protocol-specific decoding paths).
38
+ */
39
+ descriptor?: ProviderDescriptor;
40
+ /**
41
+ * When true, `stream()` and `complete()` wrap to the start once the fixture is exhausted,
42
+ * instead of throwing. Useful for loop tests that need to keep going past the recorded length.
43
+ * Defaults to false.
44
+ */
45
+ wrap?: boolean;
46
+ }
47
+ export declare class ReplayProvider implements LLMProvider {
48
+ private cursor;
49
+ private readonly messages;
50
+ private readonly tokenizer;
51
+ private readonly _descriptor;
52
+ private readonly wrap;
53
+ /**
54
+ * @param messages Ordered list of assistant messages to replay (one per LLM call).
55
+ * @param opts Optional tokenizer / descriptor / wrap-around behavior.
56
+ */
57
+ constructor(messages: ReadonlyArray<Message>, opts?: ReplayProviderOpts);
58
+ descriptor(): ProviderDescriptor;
59
+ /** Number of messages consumed so far. */
60
+ consumed(): number;
61
+ /** Number of messages remaining in the fixture (returns 0 in wrap mode once cursor passes end). */
62
+ remaining(): number;
63
+ /** Reset the cursor — useful for re-running the same fixture in a fresh session. */
64
+ reset(): void;
65
+ complete(_context: RenderedContext, _tools: ToolSchema[]): Promise<Message>;
66
+ stream(context: RenderedContext, tools: ToolSchema[], _extensions?: Record<string, unknown>, _state?: ProviderRunState, _signal?: AbortSignal): AsyncIterable<StreamEvent>;
67
+ private pull;
68
+ private estimateInputTokens;
69
+ }
@@ -0,0 +1,152 @@
1
+ /**
2
+ * ReplayProvider — an LLMProvider that emits previously-recorded assistant messages
3
+ * instead of calling a real LLM API.
4
+ *
5
+ * Purpose: deterministic re-runs for benchmarking, CI, and golden regression. Useful when you want
6
+ * to hold the model's behavior constant and measure something else (prompt-size cost Δ across
7
+ * RuntimeOptions variants, codegen of follow-on kernel work, etc.).
8
+ *
9
+ * Distinct from `provider-replay.ts`: that file's `seedProviderReplay` / `peekProviderReplay` is a
10
+ * session-repair *reasoning-content cache* — it preserves `reasoning_content` / `native_blocks` so
11
+ * the model sees its own thinking when context is re-rendered. It does NOT skip LLM calls.
12
+ * `ReplayProvider` is the orthogonal, request-skipping mechanism: it returns recorded responses
13
+ * directly, never hitting an API.
14
+ *
15
+ * Cost-accounting under replay:
16
+ * - `inputTokens` is ESTIMATED from the rendered context this call carries (NOT a recorded value
17
+ * from the original run). That's the point of replay-for-benchmarking: prompt may differ across
18
+ * variants, response is pinned, so a cost Δ purely reflects the prompt change.
19
+ * - `outputTokens` is taken from `message.tokenCount` when present; otherwise estimated from
20
+ * `message.content.length / 4`.
21
+ * - `cacheReadInputTokens` / `cacheCreationInputTokens` are emitted as 0 — replay has no real
22
+ * cache state. Mechanisms whose Δ depends on cache behavior must validate with a live A/B too.
23
+ *
24
+ * Tokenizer: by default a `chars/4` estimator (±20% for English; worse for code/JSON). For tighter
25
+ * numbers plug `opts.tokenizer = tiktokenEncoder` or similar.
26
+ */
27
+ const DEFAULT_DESCRIPTOR = {
28
+ provider: "replay",
29
+ protocol: "openai-chat",
30
+ model: "replay",
31
+ reasoning: { supported: false, preserveAcrossToolTurns: false },
32
+ toolCalls: { supported: true, requiresStrictPairing: false },
33
+ };
34
+ export class ReplayProvider {
35
+ cursor = 0;
36
+ messages;
37
+ tokenizer;
38
+ _descriptor;
39
+ wrap;
40
+ /**
41
+ * @param messages Ordered list of assistant messages to replay (one per LLM call).
42
+ * @param opts Optional tokenizer / descriptor / wrap-around behavior.
43
+ */
44
+ constructor(messages, opts = {}) {
45
+ this.messages = messages;
46
+ this.tokenizer = opts.tokenizer ?? defaultTokenizer;
47
+ this._descriptor = opts.descriptor ?? DEFAULT_DESCRIPTOR;
48
+ this.wrap = !!opts.wrap;
49
+ }
50
+ descriptor() {
51
+ return this._descriptor;
52
+ }
53
+ /** Number of messages consumed so far. */
54
+ consumed() {
55
+ return this.cursor;
56
+ }
57
+ /** Number of messages remaining in the fixture (returns 0 in wrap mode once cursor passes end). */
58
+ remaining() {
59
+ return Math.max(0, this.messages.length - this.cursor);
60
+ }
61
+ /** Reset the cursor — useful for re-running the same fixture in a fresh session. */
62
+ reset() {
63
+ this.cursor = 0;
64
+ }
65
+ async complete(_context, _tools) {
66
+ const msg = this.pull();
67
+ return {
68
+ role: "assistant",
69
+ content: msg.content,
70
+ ...(msg.toolCalls ? { toolCalls: msg.toolCalls } : {}),
71
+ ...(msg.tokenCount !== undefined ? { tokenCount: msg.tokenCount } : {}),
72
+ };
73
+ }
74
+ async *stream(context, tools, _extensions, _state, _signal) {
75
+ const msg = this.pull();
76
+ const inputTokens = this.estimateInputTokens(context, tools);
77
+ const outputTokens = msg.tokenCount !== undefined ? msg.tokenCount : this.tokenizer(msg.content || "");
78
+ const usage = {
79
+ type: "usage",
80
+ totalTokens: inputTokens + outputTokens,
81
+ inputTokens,
82
+ outputTokens,
83
+ cacheReadInputTokens: 0,
84
+ cacheCreationInputTokens: 0,
85
+ };
86
+ yield usage;
87
+ if (msg.content) {
88
+ const delta = { type: "text_delta", delta: msg.content };
89
+ yield delta;
90
+ }
91
+ for (const tc of msg.toolCalls ?? []) {
92
+ let args = {};
93
+ try {
94
+ args = JSON.parse(tc.arguments || "{}");
95
+ }
96
+ catch {
97
+ // Malformed recorded arguments — pass an empty object. The runner's downstream tool
98
+ // execution will surface the error if the tool needs them.
99
+ }
100
+ const call = { type: "tool_call", id: tc.id, name: tc.name, arguments: args };
101
+ yield call;
102
+ }
103
+ }
104
+ pull() {
105
+ if (this.cursor >= this.messages.length) {
106
+ if (this.wrap && this.messages.length > 0) {
107
+ this.cursor = 0;
108
+ }
109
+ else {
110
+ throw new Error(`ReplayProvider: fixture exhausted (consumed=${this.cursor}, total=${this.messages.length})`);
111
+ }
112
+ }
113
+ return this.messages[this.cursor++];
114
+ }
115
+ estimateInputTokens(context, tools) {
116
+ const text = renderContextToText(context, tools);
117
+ return this.tokenizer(text);
118
+ }
119
+ }
120
+ // ── helpers ───────────────────────────────────────────────────────────────────
121
+ function defaultTokenizer(text) {
122
+ return Math.ceil(text.length / 4);
123
+ }
124
+ function renderContextToText(context, tools) {
125
+ const parts = [];
126
+ if (context.systemText)
127
+ parts.push(context.systemText);
128
+ if (context.systemStable)
129
+ parts.push(context.systemStable);
130
+ if (context.systemKnowledge)
131
+ parts.push(context.systemKnowledge);
132
+ if (context.stateTurn?.content)
133
+ parts.push(context.stateTurn.content);
134
+ for (const turn of context.turns ?? []) {
135
+ if (turn.content)
136
+ parts.push(turn.content);
137
+ for (const part of turn.contentParts ?? []) {
138
+ const p = part;
139
+ if (typeof p.output === "string")
140
+ parts.push(p.output);
141
+ else if (typeof p.text === "string")
142
+ parts.push(p.text);
143
+ }
144
+ for (const tc of turn.toolCalls ?? []) {
145
+ parts.push(`${tc.name} ${tc.arguments}`);
146
+ }
147
+ }
148
+ for (const tool of tools) {
149
+ parts.push(`${tool.name} ${tool.description} ${tool.parameters}`);
150
+ }
151
+ return parts.join("\n");
152
+ }
@@ -15,6 +15,30 @@ import { LargeResultSpool } from "./large-result-spool.js";
15
15
  export interface SchedulerBudget {
16
16
  maxWallMs?: number;
17
17
  }
18
+ /** P0-C tool-gating telemetry: per-LLM-turn metrics, emitted via `RuntimeOptions.onTurnMetrics`.
19
+ * Pure observation — no behavior change. Feeds the go/no-go analysis for epoch skill gating (P1-B):
20
+ * - `toolsExposed` vs `toolsCalled` quantifies over-exposure.
21
+ * - `activeSkill` across consecutive turns yields the skill *dwell* `D` (how long a skill stays
22
+ * loaded) — the break-even input that decides whether dynamic gating beats the cache-bust cost.
23
+ * - `cacheReadTokens` / `cacheCreationTokens` give the prompt-cache hit baseline to compare against
24
+ * after B/D ship. */
25
+ export interface TurnMetrics {
26
+ /** 1-based kernel turn this LLM call belongs to. */
27
+ turn: number;
28
+ /** Number of tool schemas exposed to the model this turn (base + meta, after run-profile gating). */
29
+ toolsExposed: number;
30
+ /** Number of tool calls the model emitted this turn. */
31
+ toolsCalled: number;
32
+ /** The skill loaded and in effect going into this turn (the most recent `skill` tool call's name),
33
+ * or undefined if none is active. Consecutive equal values measure dwell. */
34
+ activeSkill?: string;
35
+ /** Full prompt size the provider reported (uncached + cache read + cache creation). */
36
+ inputTokens: number;
37
+ /** Tokens served from the prompt cache this turn (Anthropic `cache_read_input_tokens`). */
38
+ cacheReadTokens: number;
39
+ /** Tokens written to the prompt cache this turn (Anthropic `cache_creation_input_tokens`). */
40
+ cacheCreationTokens: number;
41
+ }
18
42
  export interface RuntimeOptions {
19
43
  provider: LLMProvider;
20
44
  /** M4/G5: cumulative token cap for this run (the kernel's `max_total_tokens`). A workflow node's
@@ -98,6 +122,20 @@ export interface RuntimeOptions {
98
122
  }) => Promise<MilestoneCheckResult> | MilestoneCheckResult;
99
123
  /** Passed to kernel start_run for role/isolation metadata. */
100
124
  runSpec?: AgentRunSpec;
125
+ /** P0-A tool gating: a static per-run tool profile — only these tool ids (plus the
126
+ * skill/memory/knowledge/update_plan meta-tools) are exposed to the model each turn.
127
+ * Sugar that lowers to the same `capability_filter` sub-agents use; byte-stable across
128
+ * the run, so it never busts the prompt-cache prefix. Augments `runSpec`'s filter when
129
+ * both are set; synthesizes a minimal run spec when `runSpec` is absent. Omitted/empty
130
+ * ⇒ all registered tools exposed (no gating). */
131
+ allowedToolIds?: string[];
132
+ /** P0-C: optional per-turn metrics sink for tool-gating telemetry (see `TurnMetrics`). Pure
133
+ * observation; invoked once per LLM turn. Never throws into the run loop (errors are swallowed). */
134
+ onTurnMetrics?: (metrics: TurnMetrics) => void;
135
+ /** P1-B/D stable-core: tool ids that stay exposed even when an active skill narrows the toolset
136
+ * (read/search/bash etc.). Empty/absent ⇒ skills narrow to exactly their declared `allowed_tools`
137
+ * + meta-tools. Opt-in: with no skill declaring `allowed_tools`, gating never engages. */
138
+ stableCoreToolIds?: string[];
101
139
  /** Loaded via load_milestone_contract before run start. */
102
140
  milestoneContract?: MilestoneContract;
103
141
  /** Custom sub-agent host driver; defaults to SubAgentOrchestrator. */
@@ -819,15 +819,19 @@ export class RuntimeRunner {
819
819
  if (this.opts.skillDir) {
820
820
  const { scanSkillDir } = await import("../skills/loader.js");
821
821
  const metas = await scanSkillDir(this.opts.skillDir);
822
+ // P1-B: pass the full SkillMetadata (incl. `allowedTools`) straight through — re-mapping it
823
+ // field-by-field previously dropped `allowedTools`.
822
824
  kernelApply(runtime, this.pendingObservations, {
823
825
  kind: "set_available_skills",
824
- skills: metas.map((m) => skillMetadataToKernel({
825
- name: m.name,
826
- description: m.description,
827
- whenToUse: m.whenToUse,
828
- effort: m.effort,
829
- estimatedTokens: m.estimatedTokens ?? 0,
830
- })),
826
+ skills: metas.map(m => skillMetadataToKernel(m)),
827
+ });
828
+ }
829
+ // P1-B/D: configure the stable-core tool ids (always exposed under skill gating). Empty/absent
830
+ // ⇒ skills narrow to exactly their declared tools + meta-tools.
831
+ if (this.opts.stableCoreToolIds?.length) {
832
+ kernelApply(runtime, this.pendingObservations, {
833
+ kind: "set_stable_core_tools",
834
+ tool_ids: this.opts.stableCoreToolIds,
831
835
  });
832
836
  }
833
837
  if (this.opts.dreamStore && this.opts.agentId) {
@@ -876,14 +880,43 @@ export class RuntimeRunner {
876
880
  kind: "preload_history",
877
881
  messages: replayed.map(messageToKernelMessage),
878
882
  });
883
+ // P1-B B3: rebuild active-skill gating after a wake by re-emitting SkillActivated for each
884
+ // `skill` tool call in the replayed history (active_skills is not snapshotted — graceful).
885
+ // The catalog (set_available_skills) was already fed above, so allowed_tools resolves.
886
+ for (const m of replayed) {
887
+ for (const tc of m.toolCalls ?? []) {
888
+ if (tc.name !== "skill")
889
+ continue;
890
+ try {
891
+ const name = JSON.parse(tc.arguments || "{}").name;
892
+ if (name)
893
+ kernelApply(runtime, this.pendingObservations, { kind: "skill_activated", name });
894
+ }
895
+ catch { /* malformed skill args — skip */ }
896
+ }
897
+ }
879
898
  }
880
899
  const sessionStart = Date.now();
881
900
  const startPayload = {
882
901
  kind: "start_run",
883
902
  task: { goal, criteria },
884
903
  };
885
- if (this.opts.runSpec) {
886
- startPayload.run_spec = agentRunSpecToKernel(this.opts.runSpec);
904
+ // P0-A: lower an explicit `runSpec` and/or the `allowedToolIds` profile to the kernel's
905
+ // `capability_filter`. `allowedToolIds` augments an explicit spec's filter, else synthesizes
906
+ // a minimal top-level spec carrying just the filter (reuses the existing run_spec wire — no
907
+ // new ABI). Unset on both ⇒ no run_spec ⇒ no gating (铁律: no config = old behavior).
908
+ const allowedToolIds = this.opts.allowedToolIds;
909
+ const hasProfile = allowedToolIds !== undefined && allowedToolIds.length > 0;
910
+ if (this.opts.runSpec || hasProfile) {
911
+ const baseSpec = this.opts.runSpec ?? {
912
+ identity: { agentId: this.opts.agentId ?? "root", sessionId, isSubAgent: false },
913
+ role: "custom",
914
+ goal,
915
+ };
916
+ const spec = hasProfile
917
+ ? { ...baseSpec, capabilityFilter: { ...baseSpec.capabilityFilter, allowedIds: allowedToolIds } }
918
+ : baseSpec;
919
+ startPayload.run_spec = agentRunSpecToKernel(spec);
887
920
  }
888
921
  const osProfile = assertNativeProfile(this.opts.osProfile ?? "native");
889
922
  const attentionPolicy = this.opts.attentionPolicy ?? osProfile.attentionPolicy;
@@ -944,6 +977,9 @@ export class RuntimeRunner {
944
977
  ? kernelAction(runtime, this.pendingObservations, { kind: "resume" })
945
978
  : kernelAction(runtime, this.pendingObservations, startPayload);
946
979
  let hasAttemptedReactiveCompact = false;
980
+ // P0-C: the skill loaded and in effect going into the current turn (updated when the model's
981
+ // `skill` tool call resolves). Drives the per-turn `activeSkill` metric → dwell measurement.
982
+ let activeSkill;
947
983
  while (!runtime.isTerminal()) {
948
984
  // Page-in must run before appendObservations drains pending kernel observations.
949
985
  if (action.kind === "execute_tool") {
@@ -985,6 +1021,8 @@ export class RuntimeRunner {
985
1021
  let turnTokens = 0;
986
1022
  let turnInputTokens = 0;
987
1023
  let turnOutputTokens = 0;
1024
+ let turnCacheReadTokens = 0;
1025
+ let turnCacheCreationTokens = 0;
988
1026
  let shouldRetry = false;
989
1027
  const abortSignal = this.abortController?.signal;
990
1028
  try {
@@ -999,6 +1037,9 @@ export class RuntimeRunner {
999
1037
  turnTokens = usageEvt.totalTokens;
1000
1038
  turnInputTokens = usageEvt.inputTokens ?? 0;
1001
1039
  turnOutputTokens = usageEvt.outputTokens ?? 0;
1040
+ // P0-C: capture the prompt-cache split for the tool-gating hit-rate baseline.
1041
+ turnCacheReadTokens = usageEvt.cacheReadInputTokens ?? 0;
1042
+ turnCacheCreationTokens = usageEvt.cacheCreationInputTokens ?? 0;
1002
1043
  continue;
1003
1044
  }
1004
1045
  yield evt;
@@ -1079,6 +1120,32 @@ export class RuntimeRunner {
1079
1120
  toolCalls: finalToolCalls,
1080
1121
  providerReplay,
1081
1122
  }));
1123
+ // P0-C: emit per-turn tool-gating telemetry. `activeSkill` reflects the skill in effect
1124
+ // GOING INTO this turn; a `skill` call here only takes effect next turn, so emit first, then
1125
+ // advance. Wrapped so a faulty sink can never break the run (pure observation).
1126
+ if (this.opts.onTurnMetrics) {
1127
+ try {
1128
+ this.opts.onTurnMetrics({
1129
+ turn: runtime.turn(),
1130
+ toolsExposed: tools.length,
1131
+ toolsCalled: finalToolCalls.length,
1132
+ activeSkill,
1133
+ inputTokens: turnInputTokens,
1134
+ cacheReadTokens: turnCacheReadTokens,
1135
+ cacheCreationTokens: turnCacheCreationTokens,
1136
+ });
1137
+ }
1138
+ catch { /* metrics must never break the run */ }
1139
+ }
1140
+ const skillCall = finalToolCalls.find(c => c.name === "skill");
1141
+ if (skillCall) {
1142
+ try {
1143
+ const name = JSON.parse(skillCall.arguments || "{}").name;
1144
+ if (name)
1145
+ activeSkill = name;
1146
+ }
1147
+ catch { /* malformed skill args — leave activeSkill unchanged */ }
1148
+ }
1082
1149
  }
1083
1150
  else if (action.kind === "execute_tool") {
1084
1151
  const allCalls = action.calls;
@@ -1214,6 +1281,22 @@ export class RuntimeRunner {
1214
1281
  this.pendingSpoolOutputs.set(call.id, { tool: call.name, output: result.output });
1215
1282
  }
1216
1283
  }
1284
+ // P1-B B3: a `skill` call that resolved successfully activates that skill in the kernel, so
1285
+ // the next `call_provider` narrows the toolset to its declared tools. Fed before `tool_results`
1286
+ // (which computes the next action). Errs-open: a failed/missing skill load doesn't activate.
1287
+ for (const call of allCalls) {
1288
+ if (call.name !== "skill")
1289
+ continue;
1290
+ const res = toolResults.find(r => r.callId === call.id);
1291
+ if (!res || res.isError)
1292
+ continue;
1293
+ try {
1294
+ const name = JSON.parse(call.arguments || "{}").name;
1295
+ if (name)
1296
+ kernelApply(runtime, this.pendingObservations, { kind: "skill_activated", name });
1297
+ }
1298
+ catch { /* malformed skill args — skip activation */ }
1299
+ }
1217
1300
  action = kernelAction(runtime, this.pendingObservations, {
1218
1301
  kind: "tool_results",
1219
1302
  results: toolResults.map(toolResultToKernel),
@@ -4,6 +4,10 @@ export interface SkillMetadata {
4
4
  whenToUse?: string;
5
5
  effort?: number;
6
6
  estimatedTokens?: number;
7
+ /** P1-B tool gating: tool ids this skill needs. When the skill is active, the kernel narrows the
8
+ * exposed toolset to `stable-core ∪ allowedTools`. Parsed from `allowed_tools:` frontmatter
9
+ * (comma-separated or `[a, b]`). Absent ⇒ the skill does not narrow (back-compat). */
10
+ allowedTools?: string[];
7
11
  }
8
12
  /** Read one skill file and return its body (frontmatter stripped). */
9
13
  export declare function readSkillFile(skillDir: string, name: string): Promise<string | null>;
@@ -1,5 +1,13 @@
1
1
  import { readFile, readdir } from "fs/promises";
2
2
  import path from "path";
3
+ /** Parse a frontmatter tool list: `read, write` or `[read, write]` → ["read","write"]. */
4
+ function parseToolList(v) {
5
+ if (v == null || v === "")
6
+ return undefined;
7
+ const ids = String(v).trim().replace(/^\[|\]$/g, "").split(",")
8
+ .map(x => x.trim().replace(/^["']|["']$/g, "")).filter(Boolean);
9
+ return ids.length ? ids : undefined;
10
+ }
3
11
  function parseFrontmatter(content) {
4
12
  const match = content.match(/^---\n([\s\S]*?)\n---\n?([\s\S]*)$/);
5
13
  if (!match)
@@ -38,6 +46,7 @@ export async function scanSkillDir(skillDir) {
38
46
  whenToUse: meta.when_to_use ? String(meta.when_to_use) : undefined,
39
47
  effort: meta.effort ? Number(meta.effort) : undefined,
40
48
  estimatedTokens: meta.estimated_tokens ? Number(meta.estimated_tokens) : undefined,
49
+ allowedTools: parseToolList(meta.allowed_tools),
41
50
  });
42
51
  }
43
52
  return results;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@deepstrike/sdk",
3
- "version": "0.2.18",
3
+ "version": "0.2.20",
4
4
  "description": "DeepStrike Node.js SDK",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
@@ -20,7 +20,7 @@
20
20
  },
21
21
  "dependencies": {
22
22
  "@anthropic-ai/sdk": "^0.99.0",
23
- "@deepstrike/core": "0.2.18",
23
+ "@deepstrike/core": "0.2.20",
24
24
  "@google/generative-ai": "^0.24.1",
25
25
  "openai": "^5.23.2"
26
26
  },