@deepstrike/sdk 0.1.3 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -1
- package/dist/agent.d.ts +11 -0
- package/dist/agent.js +26 -2
- package/dist/harness/harness.d.ts +47 -12
- package/dist/harness/harness.js +55 -40
- package/dist/knowledge/source.d.ts +2 -0
- package/dist/memory/protocols.d.ts +2 -0
- package/dist/skills/loader.d.ts +1 -1
- package/dist/skills/loader.js +3 -2
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -8,7 +8,23 @@ Agent framework built on a Rust kernel. The kernel handles loop control, context
|
|
|
8
8
|
npm install @deepstrike/sdk
|
|
9
9
|
```
|
|
10
10
|
|
|
11
|
-
Requires Node.js 18+.
|
|
11
|
+
Requires Node.js 18+.
|
|
12
|
+
|
|
13
|
+
### Platform support
|
|
14
|
+
|
|
15
|
+
Pre-built native addons are available for the following platforms:
|
|
16
|
+
|
|
17
|
+
| Platform | Package |
|
|
18
|
+
| -------- | ------- |
|
|
19
|
+
| Linux x64 (glibc) | `@deepstrike/core-linux-x64-gnu` |
|
|
20
|
+
| Linux ARM64 (glibc) | `@deepstrike/core-linux-arm64-gnu` |
|
|
21
|
+
| macOS x64 | `@deepstrike/core-darwin-x64` |
|
|
22
|
+
| macOS ARM64 (Apple Silicon) | `@deepstrike/core-darwin-arm64` |
|
|
23
|
+
| Windows x64 | `@deepstrike/core-win32-x64-msvc` |
|
|
24
|
+
|
|
25
|
+
The correct platform package is selected and installed automatically via `optionalDependencies`.
|
|
26
|
+
|
|
27
|
+
> **Note:** `@deepstrike/core` is the low-level native addon package and is not intended for direct use. It is an internal dependency automatically managed by `@deepstrike/sdk`. Direct installation is only relevant when building from Rust source.
|
|
12
28
|
|
|
13
29
|
---
|
|
14
30
|
|
package/dist/agent.d.ts
CHANGED
|
@@ -8,6 +8,17 @@ export interface AgentOptions {
|
|
|
8
8
|
maxTurns?: number;
|
|
9
9
|
timeoutMs?: number;
|
|
10
10
|
extensions?: Record<string, unknown>;
|
|
11
|
+
/**
|
|
12
|
+
* System-level instructions prepended to every context render.
|
|
13
|
+
* Passed to the kernel's `system` partition before the first LLM call.
|
|
14
|
+
*/
|
|
15
|
+
systemPrompt?: string;
|
|
16
|
+
/**
|
|
17
|
+
* Long-term memory snippets pre-seeded into the context before the first LLM call.
|
|
18
|
+
* Each string is pushed to the kernel's `memory` partition (highest-priority context
|
|
19
|
+
* after system). Use to inject memories retrieved from a DreamStore before a run.
|
|
20
|
+
*/
|
|
21
|
+
initialMemory?: string[];
|
|
11
22
|
/**
|
|
12
23
|
* Directory containing skill `.md` files. The kernel auto-injects a `skill`
|
|
13
24
|
* meta-tool so the model can load any skill by name on demand.
|
package/dist/agent.js
CHANGED
|
@@ -63,6 +63,9 @@ export class Agent {
|
|
|
63
63
|
this.pendingInterrupt = false;
|
|
64
64
|
this._turn = 0;
|
|
65
65
|
this._pressure = 0;
|
|
66
|
+
if (this.knowledgeSource) {
|
|
67
|
+
await this.knowledgeSource.init();
|
|
68
|
+
}
|
|
66
69
|
const kernel = await loadKernel();
|
|
67
70
|
const ext = { ...this.extensions, ...(extensions ?? {}) };
|
|
68
71
|
const sm = new kernel.LoopStateMachine({
|
|
@@ -74,6 +77,13 @@ export class Agent {
|
|
|
74
77
|
const router = new kernel.SignalRouter(256);
|
|
75
78
|
const toolSchemas = Array.from(this.tools.values()).map(t => t.schema);
|
|
76
79
|
sm.setTools(toolSchemas);
|
|
80
|
+
if (this.options.systemPrompt) {
|
|
81
|
+
const tokens = Math.max(1, Math.ceil(this.options.systemPrompt.length / 4));
|
|
82
|
+
sm.addSystemMessage(this.options.systemPrompt, tokens);
|
|
83
|
+
}
|
|
84
|
+
for (const mem of this.options.initialMemory ?? []) {
|
|
85
|
+
sm.addMemoryMessage(mem, Math.max(1, Math.ceil(mem.length / 4)));
|
|
86
|
+
}
|
|
77
87
|
if (this.skillDir) {
|
|
78
88
|
const skillMetas = await scanSkillDir(this.skillDir);
|
|
79
89
|
sm.setAvailableSkills(skillMetas.map((m) => ({
|
|
@@ -91,6 +101,8 @@ export class Agent {
|
|
|
91
101
|
sm.setKnowledgeEnabled(true);
|
|
92
102
|
}
|
|
93
103
|
let action = sm.start({ goal, criteria: criteria ?? [] });
|
|
104
|
+
const sessionStart = Date.now();
|
|
105
|
+
const sessionMsgs = [{ role: "user", content: goal }];
|
|
94
106
|
while (!sm.isTerminal()) {
|
|
95
107
|
// Update telemetry
|
|
96
108
|
this._turn = sm.turn;
|
|
@@ -172,6 +184,7 @@ export class Agent {
|
|
|
172
184
|
break;
|
|
173
185
|
}
|
|
174
186
|
action = sm.feedLlmResponse({ role: "assistant", content: finalText, toolCalls: finalToolCalls, tokenCount: turnTokens || undefined });
|
|
187
|
+
sessionMsgs.push({ role: "assistant", content: finalText, toolCalls: finalToolCalls });
|
|
175
188
|
}
|
|
176
189
|
else if (action.kind === "execute_tools") {
|
|
177
190
|
const allCalls = action.calls ?? [];
|
|
@@ -258,9 +271,20 @@ export class Agent {
|
|
|
258
271
|
this._turn = sm.turn;
|
|
259
272
|
this._pressure = sm.pressure();
|
|
260
273
|
const status = result?.termination === "completed" ? "success" : (result?.termination ?? "error");
|
|
261
|
-
// turnsUsed counts tool execution rounds; for single-turn text-only runs it's 0.
|
|
262
|
-
// Map to iterations: at least 1 if we got a result.
|
|
263
274
|
const iterations = result ? Math.max(1, result.turnsUsed) : 0;
|
|
275
|
+
if (this.options.dreamStore && this.options.agentId && sessionMsgs.length > 1) {
|
|
276
|
+
try {
|
|
277
|
+
await this.options.dreamStore.saveSession({
|
|
278
|
+
sessionId: crypto.randomUUID(),
|
|
279
|
+
agentId: this.options.agentId,
|
|
280
|
+
messages: sessionMsgs,
|
|
281
|
+
metadata: null,
|
|
282
|
+
createdAtMs: sessionStart,
|
|
283
|
+
updatedAtMs: Date.now(),
|
|
284
|
+
});
|
|
285
|
+
}
|
|
286
|
+
catch { /* session save failure must not surface to caller */ }
|
|
287
|
+
}
|
|
264
288
|
yield {
|
|
265
289
|
type: "done",
|
|
266
290
|
iterations,
|
|
@@ -1,7 +1,18 @@
|
|
|
1
1
|
import type { Agent } from "../agent.js";
|
|
2
|
+
export interface Criterion {
|
|
3
|
+
text: string;
|
|
4
|
+
required: boolean;
|
|
5
|
+
weight?: number;
|
|
6
|
+
}
|
|
7
|
+
export interface CriterionResult {
|
|
8
|
+
criterion: string;
|
|
9
|
+
passed: boolean;
|
|
10
|
+
score: number;
|
|
11
|
+
feedback: string;
|
|
12
|
+
}
|
|
2
13
|
export interface HarnessRequest {
|
|
3
14
|
goal: string;
|
|
4
|
-
criteria?:
|
|
15
|
+
criteria?: Criterion[];
|
|
5
16
|
extensions?: Record<string, unknown>;
|
|
6
17
|
}
|
|
7
18
|
export interface HarnessOutcome {
|
|
@@ -10,9 +21,42 @@ export interface HarnessOutcome {
|
|
|
10
21
|
iterations: number;
|
|
11
22
|
totalTokens: number;
|
|
12
23
|
status: string;
|
|
13
|
-
|
|
24
|
+
overallScore?: number;
|
|
14
25
|
feedback?: string;
|
|
26
|
+
details?: CriterionResult[];
|
|
27
|
+
}
|
|
28
|
+
export interface Verdict {
|
|
29
|
+
passed: boolean;
|
|
30
|
+
overallScore: number;
|
|
31
|
+
feedback: string;
|
|
32
|
+
details: CriterionResult[];
|
|
15
33
|
}
|
|
34
|
+
export type HarnessEvent = {
|
|
35
|
+
type: "token";
|
|
36
|
+
text: string;
|
|
37
|
+
} | {
|
|
38
|
+
type: "tool_call";
|
|
39
|
+
id: string;
|
|
40
|
+
name: string;
|
|
41
|
+
} | {
|
|
42
|
+
type: "tool_result";
|
|
43
|
+
callId: string;
|
|
44
|
+
content: string;
|
|
45
|
+
isError: boolean;
|
|
46
|
+
} | {
|
|
47
|
+
type: "supervising";
|
|
48
|
+
} | {
|
|
49
|
+
type: "revising";
|
|
50
|
+
verdict: Verdict;
|
|
51
|
+
} | {
|
|
52
|
+
type: "done";
|
|
53
|
+
verdict: Verdict;
|
|
54
|
+
iterations: number;
|
|
55
|
+
totalTokens: number;
|
|
56
|
+
status: string;
|
|
57
|
+
} | {
|
|
58
|
+
type: "max_attempts_reached";
|
|
59
|
+
};
|
|
16
60
|
export interface QualityGate {
|
|
17
61
|
evaluate(request: HarnessRequest, outcome: HarnessOutcome): Promise<boolean>;
|
|
18
62
|
}
|
|
@@ -21,7 +65,6 @@ export declare class SinglePassHarness {
|
|
|
21
65
|
constructor(agent: Agent);
|
|
22
66
|
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
23
67
|
}
|
|
24
|
-
/** Retry loop driven by a pluggable QualityGate (not LLM-as-judge). */
|
|
25
68
|
export declare class EvalLoopHarness {
|
|
26
69
|
private agent;
|
|
27
70
|
private gate;
|
|
@@ -31,21 +74,13 @@ export declare class EvalLoopHarness {
|
|
|
31
74
|
}
|
|
32
75
|
export interface HarnessLoopOptions {
|
|
33
76
|
maxAttempts?: number;
|
|
34
|
-
/** Directory to write distilled skills into. Requires the agent to have skillDir set. */
|
|
35
77
|
skillDir?: string;
|
|
36
78
|
}
|
|
37
|
-
/**
|
|
38
|
-
* Eval loop with LLM-as-judge and feedback injection.
|
|
39
|
-
*
|
|
40
|
-
* Each failed attempt feeds the evaluator's feedback back into the next goal,
|
|
41
|
-
* so the agent knows *why* it failed. On success, if the evaluator proposes a
|
|
42
|
-
* skill candidate it is written to `skillDir` for future sessions to reuse.
|
|
43
|
-
*/
|
|
44
79
|
export declare class HarnessLoop {
|
|
45
80
|
private agent;
|
|
46
81
|
private evalProvider;
|
|
47
82
|
private maxAttempts;
|
|
48
83
|
private skillDir?;
|
|
49
84
|
constructor(agent: Agent, evalProvider: import("../types.js").LLMProvider, options?: HarnessLoopOptions);
|
|
50
|
-
|
|
85
|
+
runStreaming(request: HarnessRequest): AsyncIterable<HarnessEvent>;
|
|
51
86
|
}
|
package/dist/harness/harness.js
CHANGED
|
@@ -9,19 +9,13 @@ async function loadKernel() {
|
|
|
9
9
|
async function runOnce(agent, req) {
|
|
10
10
|
let text = "";
|
|
11
11
|
let done;
|
|
12
|
-
for await (const evt of agent.runStreaming(req.goal, req.criteria, req.extensions)) {
|
|
12
|
+
for await (const evt of agent.runStreaming(req.goal, req.criteria?.map(c => c.text), req.extensions)) {
|
|
13
13
|
if (evt.type === "text_delta")
|
|
14
14
|
text += evt.delta;
|
|
15
15
|
else if (evt.type === "done")
|
|
16
16
|
done = evt;
|
|
17
17
|
}
|
|
18
|
-
return {
|
|
19
|
-
result: text,
|
|
20
|
-
passed: false,
|
|
21
|
-
iterations: done?.iterations ?? 0,
|
|
22
|
-
totalTokens: done?.totalTokens ?? 0,
|
|
23
|
-
status: done?.status ?? "error",
|
|
24
|
-
};
|
|
18
|
+
return { result: text, passed: false, iterations: done?.iterations ?? 0, totalTokens: done?.totalTokens ?? 0, status: done?.status ?? "error" };
|
|
25
19
|
}
|
|
26
20
|
export class SinglePassHarness {
|
|
27
21
|
agent;
|
|
@@ -32,7 +26,6 @@ export class SinglePassHarness {
|
|
|
32
26
|
return { ...await runOnce(this.agent, request), passed: true };
|
|
33
27
|
}
|
|
34
28
|
}
|
|
35
|
-
/** Retry loop driven by a pluggable QualityGate (not LLM-as-judge). */
|
|
36
29
|
export class EvalLoopHarness {
|
|
37
30
|
agent;
|
|
38
31
|
gate;
|
|
@@ -46,20 +39,12 @@ export class EvalLoopHarness {
|
|
|
46
39
|
let outcome = { result: "", passed: false, iterations: 0, totalTokens: 0, status: "error" };
|
|
47
40
|
for (let i = 0; i < this.maxAttempts; i++) {
|
|
48
41
|
outcome = await runOnce(this.agent, request);
|
|
49
|
-
if (await this.gate.evaluate(request, outcome))
|
|
42
|
+
if (await this.gate.evaluate(request, outcome))
|
|
50
43
|
return { ...outcome, passed: true };
|
|
51
|
-
}
|
|
52
44
|
}
|
|
53
45
|
return outcome;
|
|
54
46
|
}
|
|
55
47
|
}
|
|
56
|
-
/**
|
|
57
|
-
* Eval loop with LLM-as-judge and feedback injection.
|
|
58
|
-
*
|
|
59
|
-
* Each failed attempt feeds the evaluator's feedback back into the next goal,
|
|
60
|
-
* so the agent knows *why* it failed. On success, if the evaluator proposes a
|
|
61
|
-
* skill candidate it is written to `skillDir` for future sessions to reuse.
|
|
62
|
-
*/
|
|
63
48
|
export class HarnessLoop {
|
|
64
49
|
agent;
|
|
65
50
|
evalProvider;
|
|
@@ -71,47 +56,77 @@ export class HarnessLoop {
|
|
|
71
56
|
this.maxAttempts = options.maxAttempts ?? 3;
|
|
72
57
|
this.skillDir = options.skillDir;
|
|
73
58
|
}
|
|
74
|
-
async
|
|
59
|
+
async *runStreaming(request) {
|
|
75
60
|
const kernel = await loadKernel();
|
|
76
61
|
const pipeline = new kernel.EvalPipeline({ extractSkillOnPass: true });
|
|
77
|
-
|
|
62
|
+
const criteria = request.criteria ?? [];
|
|
63
|
+
// history tracks the full conversation; agent sees only the current goal message
|
|
64
|
+
// but we inject feedback as user messages to guide revision without discarding prior output
|
|
78
65
|
let currentGoal = request.goal;
|
|
66
|
+
let lastIterations = 0;
|
|
67
|
+
let lastTotalTokens = 0;
|
|
68
|
+
let lastStatus = "error";
|
|
69
|
+
let lastResult = "";
|
|
79
70
|
for (let attempt = 1; attempt <= this.maxAttempts; attempt++) {
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
71
|
+
// Stream agent output directly to caller
|
|
72
|
+
for await (const evt of this.agent.runStreaming(currentGoal, criteria.map(c => c.text), request.extensions)) {
|
|
73
|
+
if (evt.type === "text_delta") {
|
|
74
|
+
lastResult += evt.delta;
|
|
75
|
+
yield { type: "token", text: evt.delta };
|
|
76
|
+
}
|
|
77
|
+
else if (evt.type === "tool_call") {
|
|
78
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
79
|
+
const tc = evt;
|
|
80
|
+
yield { type: "tool_call", id: tc.id, name: tc.name };
|
|
81
|
+
}
|
|
82
|
+
else if (evt.type === "tool_result") {
|
|
83
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
84
|
+
const tr = evt;
|
|
85
|
+
yield { type: "tool_result", callId: tr.callId, content: tr.content, isError: tr.isError };
|
|
86
|
+
}
|
|
87
|
+
else if (evt.type === "done") {
|
|
88
|
+
const d = evt;
|
|
89
|
+
lastIterations = d.iterations;
|
|
90
|
+
lastTotalTokens = d.totalTokens;
|
|
91
|
+
lastStatus = d.status;
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
// Checkpoint: evaluator judges the completed output
|
|
95
|
+
yield { type: "supervising" };
|
|
96
|
+
const evalAction = pipeline.feedOutcome(request.goal, criteria, lastResult, attempt);
|
|
83
97
|
if (evalAction.kind !== "evaluate")
|
|
84
98
|
break;
|
|
85
|
-
// Phase 2: SDK calls evaluator LLM
|
|
86
99
|
let evalText = "";
|
|
87
100
|
for await (const evt of this.evalProvider.stream(evalAction.messages ?? [], [], undefined)) {
|
|
88
101
|
if (evt.type === "text_delta")
|
|
89
102
|
evalText += evt.delta;
|
|
90
103
|
}
|
|
91
|
-
// Phase 3: kernel parses verdict
|
|
92
104
|
const doneAction = pipeline.feedEvalResult(evalText);
|
|
93
105
|
if (doneAction.kind !== "done")
|
|
94
106
|
break;
|
|
95
|
-
|
|
96
|
-
|
|
107
|
+
const verdict = {
|
|
108
|
+
passed: doneAction.passed ?? false,
|
|
109
|
+
overallScore: doneAction.overallScore ?? 0,
|
|
110
|
+
feedback: doneAction.feedback ?? "",
|
|
111
|
+
details: doneAction.details ?? [],
|
|
112
|
+
};
|
|
113
|
+
if (verdict.passed) {
|
|
97
114
|
if (doneAction.skill_candidate && this.skillDir) {
|
|
98
115
|
const { name, description, whenToUse, content } = doneAction.skill_candidate;
|
|
99
|
-
const
|
|
100
|
-
"---",
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
whenToUse ? `when_to_use: ${whenToUse}` : null,
|
|
104
|
-
"---",
|
|
105
|
-
"",
|
|
106
|
-
].filter(l => l !== null).join("\n");
|
|
107
|
-
await writeFile(path.join(this.skillDir, `${name}.md`), frontmatter + content, "utf8");
|
|
116
|
+
const fm = ["---", `name: ${name}`, `description: ${description}`,
|
|
117
|
+
whenToUse ? `when_to_use: ${whenToUse}` : null, "---", ""]
|
|
118
|
+
.filter(Boolean).join("\n");
|
|
119
|
+
await writeFile(path.join(this.skillDir, `${name}.md`), fm + content, "utf8");
|
|
108
120
|
}
|
|
109
|
-
|
|
121
|
+
yield { type: "done", verdict, iterations: lastIterations, totalTokens: lastTotalTokens, status: lastStatus };
|
|
122
|
+
return;
|
|
110
123
|
}
|
|
111
|
-
|
|
112
|
-
|
|
124
|
+
yield { type: "revising", verdict };
|
|
125
|
+
// Inject feedback as next turn's goal — agent sees its prior output + evaluator notes
|
|
126
|
+
currentGoal = `${request.goal}\n\n[Attempt ${attempt} feedback: ${verdict.feedback}]`;
|
|
127
|
+
lastResult = "";
|
|
113
128
|
pipeline.reset();
|
|
114
129
|
}
|
|
115
|
-
|
|
130
|
+
yield { type: "max_attempts_reached" };
|
|
116
131
|
}
|
|
117
132
|
}
|
|
@@ -39,6 +39,8 @@ export interface DreamStore {
|
|
|
39
39
|
commit(agentId: string, result: CurationResult, existing: MemoryEntry[]): Promise<void>;
|
|
40
40
|
/** Semantic search over the agent's long-term memories. Called on demand during a run. */
|
|
41
41
|
search(agentId: string, query: string, topK?: number): Promise<MemoryEntry[]>;
|
|
42
|
+
/** Persist a completed session for future consolidation via `Agent.dream()`. */
|
|
43
|
+
saveSession(data: SessionData): Promise<void>;
|
|
42
44
|
}
|
|
43
45
|
export interface DreamResult {
|
|
44
46
|
sessionsProcessed: number;
|
package/dist/skills/loader.d.ts
CHANGED
|
@@ -5,7 +5,7 @@ export interface SkillMetadata {
|
|
|
5
5
|
effort?: number;
|
|
6
6
|
estimatedTokens?: number;
|
|
7
7
|
}
|
|
8
|
-
/** Read one skill file and return its
|
|
8
|
+
/** Read one skill file and return its body (frontmatter stripped). */
|
|
9
9
|
export declare function readSkillFile(skillDir: string, name: string): Promise<string | null>;
|
|
10
10
|
/** Scan a skill directory and return frontmatter-only metadata for all `.md` files. */
|
|
11
11
|
export declare function scanSkillDir(skillDir: string): Promise<SkillMetadata[]>;
|
package/dist/skills/loader.js
CHANGED
|
@@ -12,10 +12,11 @@ function parseFrontmatter(content) {
|
|
|
12
12
|
}
|
|
13
13
|
return { meta, body: match[2] };
|
|
14
14
|
}
|
|
15
|
-
/** Read one skill file and return its
|
|
15
|
+
/** Read one skill file and return its body (frontmatter stripped). */
|
|
16
16
|
export async function readSkillFile(skillDir, name) {
|
|
17
17
|
try {
|
|
18
|
-
|
|
18
|
+
const raw = await readFile(path.join(skillDir, `${name}.md`), "utf8");
|
|
19
|
+
return parseFrontmatter(raw).body;
|
|
19
20
|
}
|
|
20
21
|
catch {
|
|
21
22
|
return null;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@deepstrike/sdk",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.5",
|
|
4
4
|
"description": "DeepStrike Node.js SDK",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
},
|
|
13
13
|
"dependencies": {
|
|
14
14
|
"@anthropic-ai/sdk": "^0.39.0",
|
|
15
|
-
"@deepstrike/core": "0.1.
|
|
15
|
+
"@deepstrike/core": "0.1.5",
|
|
16
16
|
"openai": "^4.77.0"
|
|
17
17
|
},
|
|
18
18
|
"devDependencies": {
|