@deepstrike/sdk 0.1.1 → 0.1.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +150 -167
- package/dist/agent.d.ts +37 -13
- package/dist/agent.js +121 -62
- package/dist/harness/harness.d.ts +24 -10
- package/dist/harness/harness.js +42 -42
- package/dist/index.d.ts +7 -5
- package/dist/index.js +9 -2
- package/dist/knowledge/source.d.ts +2 -0
- package/dist/memory/protocols.d.ts +2 -0
- package/dist/providers/anthropic.js +3 -3
- package/dist/providers/base.d.ts +3 -0
- package/dist/providers/base.js +34 -0
- package/dist/providers/ollama.js +10 -1
- package/dist/providers/openai.d.ts +7 -0
- package/dist/providers/openai.js +63 -6
- package/dist/safety/permissions.d.ts +16 -3
- package/dist/safety/permissions.js +27 -10
- package/dist/signals/gateway.d.ts +21 -17
- package/dist/signals/gateway.js +31 -25
- package/dist/skills/loader.d.ts +1 -1
- package/dist/skills/loader.js +3 -2
- package/dist/types.d.ts +36 -0
- package/package.json +2 -2
package/dist/agent.js
CHANGED
|
@@ -2,13 +2,15 @@ import { executeTools } from "./tools/index.js";
|
|
|
2
2
|
import { readSkillFile, scanSkillDir } from "./skills/loader.js";
|
|
3
3
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
4
4
|
async function loadKernel() {
|
|
5
|
-
|
|
5
|
+
const mod = await import("@deepstrike/core");
|
|
6
|
+
// CJS modules imported via ESM dynamic import expose exports under `.default`
|
|
7
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
8
|
+
return mod.default ?? mod;
|
|
6
9
|
}
|
|
7
10
|
export class Agent {
|
|
8
11
|
provider;
|
|
9
12
|
options;
|
|
10
13
|
tools = new Map();
|
|
11
|
-
blockedTools = new Set();
|
|
12
14
|
extensions;
|
|
13
15
|
skillDir;
|
|
14
16
|
knowledgeSource;
|
|
@@ -16,6 +18,9 @@ export class Agent {
|
|
|
16
18
|
dreamStore;
|
|
17
19
|
interrupted = false;
|
|
18
20
|
pendingInterrupt = false;
|
|
21
|
+
// Live telemetry — updated each runStreaming call
|
|
22
|
+
_turn = 0;
|
|
23
|
+
_pressure = 0;
|
|
19
24
|
constructor(provider, options) {
|
|
20
25
|
this.provider = provider;
|
|
21
26
|
this.options = options;
|
|
@@ -25,6 +30,10 @@ export class Agent {
|
|
|
25
30
|
this.signalSource = options.signalSource;
|
|
26
31
|
this.dreamStore = options.dreamStore;
|
|
27
32
|
}
|
|
33
|
+
/** Current turn index within the active run (0 before a run starts). */
|
|
34
|
+
get turn() { return this._turn; }
|
|
35
|
+
/** Context pressure ratio [0–1] from the kernel. Values > 0.8 trigger compression. */
|
|
36
|
+
get pressure() { return this._pressure; }
|
|
28
37
|
interrupt() {
|
|
29
38
|
this.interrupted = true;
|
|
30
39
|
}
|
|
@@ -37,21 +46,26 @@ export class Agent {
|
|
|
37
46
|
this.tools.delete(name);
|
|
38
47
|
return this;
|
|
39
48
|
}
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
49
|
+
/**
|
|
50
|
+
* Collect the full text response and return it.
|
|
51
|
+
* For richer control (streaming, tool events, token counts) use `runStreaming`.
|
|
52
|
+
*/
|
|
44
53
|
async run(goal, criteria, extensions) {
|
|
45
|
-
let
|
|
54
|
+
let content = "";
|
|
46
55
|
for await (const evt of this.runStreaming(goal, criteria, extensions)) {
|
|
47
|
-
if (evt.type === "
|
|
48
|
-
|
|
56
|
+
if (evt.type === "text_delta")
|
|
57
|
+
content += evt.delta;
|
|
49
58
|
}
|
|
50
|
-
return
|
|
59
|
+
return content;
|
|
51
60
|
}
|
|
52
61
|
async *runStreaming(goal, criteria, extensions) {
|
|
53
62
|
this.interrupted = false;
|
|
54
63
|
this.pendingInterrupt = false;
|
|
64
|
+
this._turn = 0;
|
|
65
|
+
this._pressure = 0;
|
|
66
|
+
if (this.knowledgeSource) {
|
|
67
|
+
await this.knowledgeSource.init();
|
|
68
|
+
}
|
|
55
69
|
const kernel = await loadKernel();
|
|
56
70
|
const ext = { ...this.extensions, ...(extensions ?? {}) };
|
|
57
71
|
const sm = new kernel.LoopStateMachine({
|
|
@@ -59,15 +73,20 @@ export class Agent {
|
|
|
59
73
|
maxTurns: this.options.maxTurns ?? 25,
|
|
60
74
|
timeoutMs: this.options.timeoutMs,
|
|
61
75
|
});
|
|
62
|
-
//
|
|
76
|
+
// Per-run SignalRouter — dedup state never leaks between runs.
|
|
63
77
|
const router = new kernel.SignalRouter(256);
|
|
64
78
|
const toolSchemas = Array.from(this.tools.values()).map(t => t.schema);
|
|
65
79
|
sm.setTools(toolSchemas);
|
|
66
|
-
|
|
67
|
-
|
|
80
|
+
if (this.options.systemPrompt) {
|
|
81
|
+
const tokens = Math.max(1, Math.ceil(this.options.systemPrompt.length / 4));
|
|
82
|
+
sm.addSystemMessage(this.options.systemPrompt, tokens);
|
|
83
|
+
}
|
|
84
|
+
for (const mem of this.options.initialMemory ?? []) {
|
|
85
|
+
sm.addMemoryMessage(mem, Math.max(1, Math.ceil(mem.length / 4)));
|
|
86
|
+
}
|
|
68
87
|
if (this.skillDir) {
|
|
69
88
|
const skillMetas = await scanSkillDir(this.skillDir);
|
|
70
|
-
sm.setAvailableSkills(skillMetas.map(m => ({
|
|
89
|
+
sm.setAvailableSkills(skillMetas.map((m) => ({
|
|
71
90
|
name: m.name,
|
|
72
91
|
description: m.description,
|
|
73
92
|
whenToUse: m.whenToUse,
|
|
@@ -75,17 +94,20 @@ export class Agent {
|
|
|
75
94
|
estimatedTokens: m.estimatedTokens ?? 0,
|
|
76
95
|
})));
|
|
77
96
|
}
|
|
78
|
-
// Enable the memory meta-tool when both dreamStore and agentId are provided.
|
|
79
97
|
if (this.dreamStore && this.options.agentId) {
|
|
80
98
|
sm.setMemoryEnabled(true);
|
|
81
99
|
}
|
|
82
|
-
// Enable the knowledge meta-tool when a KnowledgeSource is configured.
|
|
83
100
|
if (this.knowledgeSource) {
|
|
84
101
|
sm.setKnowledgeEnabled(true);
|
|
85
102
|
}
|
|
86
103
|
let action = sm.start({ goal, criteria: criteria ?? [] });
|
|
87
|
-
|
|
104
|
+
const sessionStart = Date.now();
|
|
105
|
+
const sessionMsgs = [{ role: "user", content: goal }];
|
|
88
106
|
while (!sm.isTerminal()) {
|
|
107
|
+
// Update telemetry
|
|
108
|
+
this._turn = sm.turn;
|
|
109
|
+
this._pressure = sm.pressure();
|
|
110
|
+
// Hard interrupt
|
|
89
111
|
if (this.interrupted) {
|
|
90
112
|
action = sm.feedTimeout();
|
|
91
113
|
break;
|
|
@@ -95,18 +117,22 @@ export class Agent {
|
|
|
95
117
|
action = sm.feedTimeout();
|
|
96
118
|
break;
|
|
97
119
|
}
|
|
120
|
+
// Drain context-compression observations
|
|
121
|
+
sm.takeObservations();
|
|
98
122
|
// Poll signal source and route through kernel SignalRouter
|
|
99
123
|
if (this.signalSource) {
|
|
100
124
|
const sig = await this.signalSource.nextSignal();
|
|
101
125
|
if (sig) {
|
|
126
|
+
const sigAny = sig;
|
|
102
127
|
const kernelSig = {
|
|
103
128
|
id: crypto.randomUUID(),
|
|
104
|
-
source:
|
|
105
|
-
signalType:
|
|
106
|
-
urgency:
|
|
129
|
+
source: sigAny.source ?? "custom",
|
|
130
|
+
signalType: sigAny.signalType ?? "event",
|
|
131
|
+
urgency: sigAny.urgency
|
|
132
|
+
?? (sig.kind === "interrupt" ? "critical" : "normal"),
|
|
107
133
|
summary: String(sig.payload?.goal ?? sig.kind),
|
|
108
134
|
payload: JSON.stringify(sig.payload ?? {}),
|
|
109
|
-
dedupeKey:
|
|
135
|
+
dedupeKey: sigAny.dedupeKey ?? null,
|
|
110
136
|
timestampMs: Date.now(),
|
|
111
137
|
};
|
|
112
138
|
const disposition = router.ingest(kernelSig, action.kind === "execute_tools");
|
|
@@ -114,20 +140,35 @@ export class Agent {
|
|
|
114
140
|
action = sm.feedTimeout();
|
|
115
141
|
break;
|
|
116
142
|
}
|
|
117
|
-
if (disposition === "interrupt")
|
|
143
|
+
if (disposition === "interrupt")
|
|
118
144
|
this.pendingInterrupt = true;
|
|
119
|
-
}
|
|
120
|
-
// "queue" → buffered; "observe" / "ignore" / "dropped" → no action
|
|
121
145
|
}
|
|
122
146
|
}
|
|
123
|
-
|
|
147
|
+
// Drain previously queued signals — apply any high-urgency ones
|
|
148
|
+
let queued = router.next();
|
|
149
|
+
while (queued) {
|
|
150
|
+
if (queued.urgency === "critical") {
|
|
151
|
+
action = sm.feedTimeout();
|
|
152
|
+
break;
|
|
153
|
+
}
|
|
154
|
+
if (queued.urgency === "high")
|
|
155
|
+
this.pendingInterrupt = true;
|
|
156
|
+
queued = router.next();
|
|
157
|
+
}
|
|
158
|
+
if (this.interrupted || (sm.isTerminal()))
|
|
159
|
+
break;
|
|
124
160
|
if (action.kind === "call_llm") {
|
|
125
|
-
finalText = "";
|
|
126
161
|
const finalToolCalls = [];
|
|
162
|
+
let finalText = "";
|
|
127
163
|
const messages = (action.messages ?? []);
|
|
128
164
|
const tools = (action.tools ?? []);
|
|
165
|
+
let turnTokens = 0;
|
|
129
166
|
try {
|
|
130
167
|
for await (const evt of this.provider.stream(messages, tools, Object.keys(ext).length ? ext : undefined)) {
|
|
168
|
+
if (evt.type === "usage") {
|
|
169
|
+
turnTokens = evt.totalTokens;
|
|
170
|
+
continue;
|
|
171
|
+
}
|
|
131
172
|
yield evt;
|
|
132
173
|
if (evt.type === "text_delta")
|
|
133
174
|
finalText += evt.delta;
|
|
@@ -142,45 +183,47 @@ export class Agent {
|
|
|
142
183
|
action = sm.feedTimeout();
|
|
143
184
|
break;
|
|
144
185
|
}
|
|
145
|
-
action = sm.feedLlmResponse({ role: "assistant", content: finalText, toolCalls: finalToolCalls });
|
|
186
|
+
action = sm.feedLlmResponse({ role: "assistant", content: finalText, toolCalls: finalToolCalls, tokenCount: turnTokens || undefined });
|
|
187
|
+
sessionMsgs.push({ role: "assistant", content: finalText, toolCalls: finalToolCalls });
|
|
146
188
|
}
|
|
147
189
|
else if (action.kind === "execute_tools") {
|
|
148
190
|
const allCalls = action.calls ?? [];
|
|
149
|
-
// Governance
|
|
191
|
+
// Governance evaluation
|
|
150
192
|
const permittedCalls = [];
|
|
193
|
+
const deniedResults = [];
|
|
151
194
|
for (const c of allCalls) {
|
|
152
|
-
if (this.blockedTools.has(c.name)) {
|
|
153
|
-
yield { type: "error", message: `tool blocked: ${c.name}` };
|
|
154
|
-
continue;
|
|
155
|
-
}
|
|
156
195
|
if (this.options.governance) {
|
|
157
196
|
const verdict = this.options.governance.evaluate(c.name, c.arguments);
|
|
158
197
|
if (verdict.kind === "deny") {
|
|
159
|
-
|
|
198
|
+
const msg = `permission denied: ${c.name} — ${verdict.reason ?? ""}`;
|
|
199
|
+
yield { type: "error", message: msg };
|
|
200
|
+
deniedResults.push({ callId: c.id, name: c.name, output: msg, isError: true });
|
|
201
|
+
continue;
|
|
202
|
+
}
|
|
203
|
+
if (verdict.kind === "rate_limited") {
|
|
204
|
+
const msg = `rate limited: ${c.name} — retry after ${verdict.retryAfterMs ?? 0}ms`;
|
|
205
|
+
yield { type: "error", message: msg };
|
|
206
|
+
deniedResults.push({ callId: c.id, name: c.name, output: msg, isError: true });
|
|
160
207
|
continue;
|
|
161
208
|
}
|
|
162
209
|
if (verdict.kind === "ask_user") {
|
|
163
|
-
yield { type: "permission_request", callId: c.id, toolName: c.name, arguments: c.arguments, reason: verdict.reason };
|
|
210
|
+
yield { type: "permission_request", callId: c.id, toolName: c.name, arguments: c.arguments, reason: verdict.reason ?? "" };
|
|
211
|
+
deniedResults.push({ callId: c.id, name: c.name, output: `awaiting user approval: ${c.name}`, isError: true });
|
|
164
212
|
continue;
|
|
165
213
|
}
|
|
166
214
|
}
|
|
167
215
|
permittedCalls.push(c);
|
|
168
216
|
}
|
|
169
|
-
const
|
|
170
|
-
|
|
171
|
-
const
|
|
172
|
-
|
|
173
|
-
const memoryCalls = calls.filter((c) => c.name === "memory");
|
|
174
|
-
// Intercept `knowledge` meta-tool calls: retrieve from KnowledgeSource.
|
|
175
|
-
const knowledgeCalls = calls.filter((c) => c.name === "knowledge");
|
|
176
|
-
const regularCalls = calls.filter((c) => c.name !== "skill" && c.name !== "memory" && c.name !== "knowledge");
|
|
217
|
+
const skillCalls = permittedCalls.filter((c) => c.name === "skill");
|
|
218
|
+
const memoryCalls = permittedCalls.filter((c) => c.name === "memory");
|
|
219
|
+
const knowledgeCalls = permittedCalls.filter((c) => c.name === "knowledge");
|
|
220
|
+
const regularCalls = permittedCalls.filter((c) => c.name !== "skill" && c.name !== "memory" && c.name !== "knowledge");
|
|
177
221
|
const skillResults = this.skillDir
|
|
178
222
|
? await Promise.all(skillCalls.map(async (c) => {
|
|
179
223
|
const args = tryParseJson(c.arguments);
|
|
180
224
|
const name = String(args?.name ?? "");
|
|
181
225
|
const content = await readSkillFile(this.skillDir, name);
|
|
182
|
-
|
|
183
|
-
return { callId: c.id, name: c.name, output, isError: !content };
|
|
226
|
+
return { callId: c.id, name: c.name, output: content ?? `Skill "${name}" not found.`, isError: !content };
|
|
184
227
|
}))
|
|
185
228
|
: skillCalls.map((c) => ({ callId: c.id, name: c.name, output: "No skill directory configured.", isError: true }));
|
|
186
229
|
const memoryResults = (this.dreamStore && this.options.agentId)
|
|
@@ -190,7 +233,7 @@ export class Agent {
|
|
|
190
233
|
const topK = typeof args?.top_k === "number" ? args.top_k : 5;
|
|
191
234
|
const entries = await this.dreamStore.search(this.options.agentId, query, topK);
|
|
192
235
|
const output = entries.length
|
|
193
|
-
? entries.map(e => `[score=${e.score.toFixed(3)}] ${e.text}`).join("\n---\n")
|
|
236
|
+
? entries.map((e) => `[score=${e.score.toFixed(3)}] ${e.text}`).join("\n---\n")
|
|
194
237
|
: "No relevant memories found.";
|
|
195
238
|
return { callId: c.id, name: c.name, output, isError: false };
|
|
196
239
|
}))
|
|
@@ -205,18 +248,18 @@ export class Agent {
|
|
|
205
248
|
return { callId: c.id, name: c.name, output, isError: false };
|
|
206
249
|
}))
|
|
207
250
|
: knowledgeCalls.map((c) => ({ callId: c.id, name: c.name, output: "Knowledge source not configured.", isError: true }));
|
|
208
|
-
// Yield all meta-tool results
|
|
209
251
|
for (const r of [...skillResults, ...memoryResults, ...knowledgeResults])
|
|
210
252
|
yield { type: "tool_result", callId: r.callId, name: r.name, content: r.output, isError: r.isError };
|
|
211
253
|
const results = await executeTools(regularCalls, this.tools);
|
|
212
254
|
for (const r of results) {
|
|
213
|
-
const name = regularCalls.find(c => c.id === r.callId)?.name ?? "";
|
|
255
|
+
const name = regularCalls.find((c) => c.id === r.callId)?.name ?? "";
|
|
214
256
|
yield { type: "tool_result", callId: r.callId, name, content: r.output, isError: r.isError };
|
|
215
257
|
}
|
|
216
258
|
action = sm.feedToolResults([
|
|
217
|
-
...
|
|
218
|
-
...
|
|
219
|
-
...
|
|
259
|
+
...deniedResults.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
260
|
+
...skillResults.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
261
|
+
...memoryResults.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
262
|
+
...knowledgeResults.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
220
263
|
...results.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
221
264
|
]);
|
|
222
265
|
}
|
|
@@ -225,20 +268,38 @@ export class Agent {
|
|
|
225
268
|
}
|
|
226
269
|
}
|
|
227
270
|
const result = action.result;
|
|
271
|
+
this._turn = sm.turn;
|
|
272
|
+
this._pressure = sm.pressure();
|
|
273
|
+
const status = result?.termination === "completed" ? "success" : (result?.termination ?? "error");
|
|
274
|
+
const iterations = result ? Math.max(1, result.turnsUsed) : 0;
|
|
275
|
+
if (this.options.dreamStore && this.options.agentId && sessionMsgs.length > 1) {
|
|
276
|
+
try {
|
|
277
|
+
await this.options.dreamStore.saveSession({
|
|
278
|
+
sessionId: crypto.randomUUID(),
|
|
279
|
+
agentId: this.options.agentId,
|
|
280
|
+
messages: sessionMsgs,
|
|
281
|
+
metadata: null,
|
|
282
|
+
createdAtMs: sessionStart,
|
|
283
|
+
updatedAtMs: Date.now(),
|
|
284
|
+
});
|
|
285
|
+
}
|
|
286
|
+
catch { /* session save failure must not surface to caller */ }
|
|
287
|
+
}
|
|
228
288
|
yield {
|
|
229
289
|
type: "done",
|
|
230
|
-
iterations
|
|
231
|
-
totalTokens: Number(result
|
|
232
|
-
status
|
|
290
|
+
iterations,
|
|
291
|
+
totalTokens: result?.totalTokensUsed ? Number(result.totalTokensUsed) : 0,
|
|
292
|
+
status,
|
|
233
293
|
};
|
|
234
294
|
}
|
|
235
295
|
/**
|
|
236
|
-
* Trigger
|
|
296
|
+
* Trigger the idle dreaming cycle for this agent.
|
|
297
|
+
* Requires `dreamStore` and `agentId` to be configured.
|
|
237
298
|
*
|
|
238
|
-
* Phase 1 — kernel rule-based analysis + LLM prompt assembly
|
|
239
|
-
* Phase 2 — LLM synthesis call (
|
|
240
|
-
* Phase 3 — kernel parses + curates results
|
|
241
|
-
* Phase 4 — commit delta to DreamStore (
|
|
299
|
+
* Phase 1 — kernel rule-based analysis + LLM prompt assembly
|
|
300
|
+
* Phase 2 — LLM synthesis call (I/O)
|
|
301
|
+
* Phase 3 — kernel parses + curates results
|
|
302
|
+
* Phase 4 — commit delta to DreamStore (I/O)
|
|
242
303
|
*/
|
|
243
304
|
async dream(agentId, nowMs = Date.now()) {
|
|
244
305
|
if (!this.dreamStore)
|
|
@@ -256,9 +317,7 @@ export class Agent {
|
|
|
256
317
|
role: m.role,
|
|
257
318
|
content: m.content,
|
|
258
319
|
tokenCount: m.tokenCount,
|
|
259
|
-
toolCalls: (m.toolCalls ?? []).map(tc => ({
|
|
260
|
-
id: tc.id, name: tc.name, arguments: tc.arguments,
|
|
261
|
-
})),
|
|
320
|
+
toolCalls: (m.toolCalls ?? []).map(tc => ({ id: tc.id, name: tc.name, arguments: tc.arguments })),
|
|
262
321
|
})),
|
|
263
322
|
metadata: JSON.stringify(s.metadata ?? null),
|
|
264
323
|
createdAtMs: s.createdAtMs,
|
|
@@ -271,7 +330,7 @@ export class Agent {
|
|
|
271
330
|
}));
|
|
272
331
|
const pipeline = new kernel.IdlePipeline(agentId);
|
|
273
332
|
const action1 = pipeline.feedTrigger(kernelSessions, kernelMemories, nowMs);
|
|
274
|
-
if (action1.kind === "noop") {
|
|
333
|
+
if (action1.kind === "noop" || action1.kind === "aborted") {
|
|
275
334
|
return { sessionsProcessed: 0, insightsExtracted: 0, entriesAdded: 0, entriesRemoved: 0 };
|
|
276
335
|
}
|
|
277
336
|
if (action1.kind !== "synthesize_insights") {
|
|
@@ -1,7 +1,18 @@
|
|
|
1
1
|
import type { Agent } from "../agent.js";
|
|
2
|
+
export interface Criterion {
|
|
3
|
+
text: string;
|
|
4
|
+
required: boolean;
|
|
5
|
+
weight?: number;
|
|
6
|
+
}
|
|
7
|
+
export interface CriterionResult {
|
|
8
|
+
criterion: string;
|
|
9
|
+
passed: boolean;
|
|
10
|
+
score: number;
|
|
11
|
+
feedback: string;
|
|
12
|
+
}
|
|
2
13
|
export interface HarnessRequest {
|
|
3
14
|
goal: string;
|
|
4
|
-
criteria?:
|
|
15
|
+
criteria?: Criterion[];
|
|
5
16
|
extensions?: Record<string, unknown>;
|
|
6
17
|
}
|
|
7
18
|
export interface HarnessOutcome {
|
|
@@ -10,26 +21,29 @@ export interface HarnessOutcome {
|
|
|
10
21
|
iterations: number;
|
|
11
22
|
totalTokens: number;
|
|
12
23
|
status: string;
|
|
13
|
-
|
|
24
|
+
overallScore?: number;
|
|
14
25
|
feedback?: string;
|
|
26
|
+
details?: CriterionResult[];
|
|
27
|
+
}
|
|
28
|
+
export interface QualityGate {
|
|
29
|
+
evaluate(request: HarnessRequest, outcome: HarnessOutcome): Promise<boolean>;
|
|
15
30
|
}
|
|
16
31
|
export declare class SinglePassHarness {
|
|
17
32
|
private agent;
|
|
18
33
|
constructor(agent: Agent);
|
|
19
34
|
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
20
35
|
}
|
|
36
|
+
export declare class EvalLoopHarness {
|
|
37
|
+
private agent;
|
|
38
|
+
private gate;
|
|
39
|
+
private maxAttempts;
|
|
40
|
+
constructor(agent: Agent, gate: QualityGate, maxAttempts?: number);
|
|
41
|
+
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
42
|
+
}
|
|
21
43
|
export interface HarnessLoopOptions {
|
|
22
44
|
maxAttempts?: number;
|
|
23
|
-
/** Directory to write distilled skills into. Requires the agent to have skillDir set. */
|
|
24
45
|
skillDir?: string;
|
|
25
46
|
}
|
|
26
|
-
/**
|
|
27
|
-
* Eval loop with LLM-as-judge and feedback injection.
|
|
28
|
-
*
|
|
29
|
-
* Each failed attempt feeds the evaluator's feedback back into the next goal,
|
|
30
|
-
* so the agent knows *why* it failed. On success, if the evaluator proposes a
|
|
31
|
-
* skill candidate it is written to `skillDir` for future sessions to reuse.
|
|
32
|
-
*/
|
|
33
47
|
export declare class HarnessLoop {
|
|
34
48
|
private agent;
|
|
35
49
|
private evalProvider;
|
package/dist/harness/harness.js
CHANGED
|
@@ -2,24 +2,20 @@ import { writeFile } from "fs/promises";
|
|
|
2
2
|
import path from "path";
|
|
3
3
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
4
4
|
async function loadKernel() {
|
|
5
|
-
|
|
5
|
+
const mod = await import("@deepstrike/core");
|
|
6
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
7
|
+
return mod.default ?? mod;
|
|
6
8
|
}
|
|
7
9
|
async function runOnce(agent, req) {
|
|
8
10
|
let text = "";
|
|
9
11
|
let done;
|
|
10
|
-
for await (const evt of agent.runStreaming(req.goal, req.criteria, req.extensions)) {
|
|
12
|
+
for await (const evt of agent.runStreaming(req.goal, req.criteria?.map(c => c.text), req.extensions)) {
|
|
11
13
|
if (evt.type === "text_delta")
|
|
12
14
|
text += evt.delta;
|
|
13
15
|
else if (evt.type === "done")
|
|
14
16
|
done = evt;
|
|
15
17
|
}
|
|
16
|
-
return {
|
|
17
|
-
result: text,
|
|
18
|
-
passed: false,
|
|
19
|
-
iterations: done?.iterations ?? 0,
|
|
20
|
-
totalTokens: done?.totalTokens ?? 0,
|
|
21
|
-
status: done?.status ?? "error",
|
|
22
|
-
};
|
|
18
|
+
return { result: text, passed: false, iterations: done?.iterations ?? 0, totalTokens: done?.totalTokens ?? 0, status: done?.status ?? "error" };
|
|
23
19
|
}
|
|
24
20
|
export class SinglePassHarness {
|
|
25
21
|
agent;
|
|
@@ -30,13 +26,25 @@ export class SinglePassHarness {
|
|
|
30
26
|
return { ...await runOnce(this.agent, request), passed: true };
|
|
31
27
|
}
|
|
32
28
|
}
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
29
|
+
export class EvalLoopHarness {
|
|
30
|
+
agent;
|
|
31
|
+
gate;
|
|
32
|
+
maxAttempts;
|
|
33
|
+
constructor(agent, gate, maxAttempts = 3) {
|
|
34
|
+
this.agent = agent;
|
|
35
|
+
this.gate = gate;
|
|
36
|
+
this.maxAttempts = maxAttempts;
|
|
37
|
+
}
|
|
38
|
+
async run(request) {
|
|
39
|
+
let outcome = { result: "", passed: false, iterations: 0, totalTokens: 0, status: "error" };
|
|
40
|
+
for (let i = 0; i < this.maxAttempts; i++) {
|
|
41
|
+
outcome = await runOnce(this.agent, request);
|
|
42
|
+
if (await this.gate.evaluate(request, outcome))
|
|
43
|
+
return { ...outcome, passed: true };
|
|
44
|
+
}
|
|
45
|
+
return outcome;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
40
48
|
export class HarnessLoop {
|
|
41
49
|
agent;
|
|
42
50
|
evalProvider;
|
|
@@ -51,48 +59,40 @@ export class HarnessLoop {
|
|
|
51
59
|
async run(request) {
|
|
52
60
|
const kernel = await loadKernel();
|
|
53
61
|
const pipeline = new kernel.EvalPipeline({ extractSkillOnPass: true });
|
|
62
|
+
const criteria = request.criteria ?? [];
|
|
54
63
|
let outcome = { result: "", passed: false, iterations: 0, totalTokens: 0, status: "error" };
|
|
55
64
|
let currentGoal = request.goal;
|
|
56
65
|
for (let attempt = 1; attempt <= this.maxAttempts; attempt++) {
|
|
57
66
|
outcome = await runOnce(this.agent, { ...request, goal: currentGoal });
|
|
58
|
-
|
|
59
|
-
const evalAction = pipeline.feedOutcome({
|
|
60
|
-
goal: request.goal,
|
|
61
|
-
criteria: request.criteria ?? [],
|
|
62
|
-
result: outcome.result,
|
|
63
|
-
attempt,
|
|
64
|
-
});
|
|
67
|
+
const evalAction = pipeline.feedOutcome(request.goal, criteria, outcome.result, attempt);
|
|
65
68
|
if (evalAction.kind !== "evaluate")
|
|
66
69
|
break;
|
|
67
|
-
// Phase 2: SDK calls evaluator LLM
|
|
68
70
|
let evalText = "";
|
|
69
71
|
for await (const evt of this.evalProvider.stream(evalAction.messages ?? [], [], undefined)) {
|
|
70
72
|
if (evt.type === "text_delta")
|
|
71
73
|
evalText += evt.delta;
|
|
72
74
|
}
|
|
73
|
-
|
|
74
|
-
const doneAction = pipeline.feedEvalResult({ content: evalText });
|
|
75
|
+
const doneAction = pipeline.feedEvalResult(evalText);
|
|
75
76
|
if (doneAction.kind !== "done")
|
|
76
77
|
break;
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
"",
|
|
89
|
-
|
|
90
|
-
await writeFile(path.join(this.skillDir, `${name}.md`),
|
|
78
|
+
outcome = {
|
|
79
|
+
...outcome,
|
|
80
|
+
passed: doneAction.passed ?? false,
|
|
81
|
+
overallScore: doneAction.overallScore ?? undefined,
|
|
82
|
+
feedback: doneAction.feedback ?? undefined,
|
|
83
|
+
details: doneAction.details ?? undefined,
|
|
84
|
+
};
|
|
85
|
+
if (doneAction.passed) {
|
|
86
|
+
if (doneAction.skill_candidate && this.skillDir) {
|
|
87
|
+
const { name, description, whenToUse, content } = doneAction.skill_candidate;
|
|
88
|
+
const fm = ["---", `name: ${name}`, `description: ${description}`,
|
|
89
|
+
whenToUse ? `when_to_use: ${whenToUse}` : null, "---", ""]
|
|
90
|
+
.filter(Boolean).join("\n");
|
|
91
|
+
await writeFile(path.join(this.skillDir, `${name}.md`), fm + content, "utf8");
|
|
91
92
|
}
|
|
92
93
|
return outcome;
|
|
93
94
|
}
|
|
94
|
-
|
|
95
|
-
currentGoal = `${request.goal}\n\n[Previous attempt ${attempt} failed: ${evalResult.feedback}]`;
|
|
95
|
+
currentGoal = `${request.goal}\n\n[Previous attempt ${attempt} failed: ${doneAction.feedback}]`;
|
|
96
96
|
pipeline.reset();
|
|
97
97
|
}
|
|
98
98
|
return outcome;
|
package/dist/index.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
export { Agent } from "./agent.js";
|
|
2
2
|
export type { AgentOptions } from "./agent.js";
|
|
3
3
|
export { AnthropicProvider } from "./providers/anthropic.js";
|
|
4
|
-
export { OpenAIProvider, QwenProvider, DeepSeekProvider, MiniMaxProvider } from "./providers/openai.js";
|
|
4
|
+
export { OpenAIProvider, QwenProvider, DeepSeekProvider, MiniMaxProvider, KimiProvider } from "./providers/openai.js";
|
|
5
5
|
export { OllamaProvider } from "./providers/ollama.js";
|
|
6
6
|
export { CircuitBreaker, normalizeToolCall } from "./providers/base.js";
|
|
7
7
|
export { tool, executeTools, readFile } from "./tools/index.js";
|
|
@@ -11,11 +11,13 @@ export type { SkillMetadata } from "./skills/loader.js";
|
|
|
11
11
|
export { WorkingMemory } from "./memory/working.js";
|
|
12
12
|
export type { DreamStore, DreamResult, SessionData, SessionMessage, MemoryEntry, CurationResult, CurationStats, } from "./memory/protocols.js";
|
|
13
13
|
export type { KnowledgeSource } from "./knowledge/source.js";
|
|
14
|
-
export { SinglePassHarness, HarnessLoop } from "./harness/harness.js";
|
|
15
|
-
export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions } from "./harness/harness.js";
|
|
14
|
+
export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
|
|
15
|
+
export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions, QualityGate } from "./harness/harness.js";
|
|
16
16
|
export { ScheduledPrompt } from "./signals/scheduled.js";
|
|
17
17
|
export { SignalGateway } from "./signals/gateway.js";
|
|
18
18
|
export type { RuntimeSignal, SignalSource } from "./signals/types.js";
|
|
19
19
|
export { PermissionManager, PermissionMode } from "./safety/permissions.js";
|
|
20
|
-
export type { PermissionDecision } from "./safety/permissions.js";
|
|
21
|
-
export
|
|
20
|
+
export type { PermissionDecision, Permission } from "./safety/permissions.js";
|
|
21
|
+
export declare const Governance: typeof import("@deepstrike/core").Governance;
|
|
22
|
+
export type { GovernanceVerdictObj as GovernanceVerdict } from "@deepstrike/core";
|
|
23
|
+
export type { Message, ToolCall, ToolResult, ToolSchema, ContentPart, TextPart, ImagePart, AudioPart, StreamEvent, TextDelta, ThinkingDelta, ToolCallEvent, ToolResultEvent, DoneEvent, ErrorEvent, PermissionRequestEvent, LLMProvider, RetryConfig, TokenUsage, ProviderToolSpec, } from "./types.js";
|
package/dist/index.js
CHANGED
|
@@ -1,12 +1,19 @@
|
|
|
1
1
|
export { Agent } from "./agent.js";
|
|
2
2
|
export { AnthropicProvider } from "./providers/anthropic.js";
|
|
3
|
-
export { OpenAIProvider, QwenProvider, DeepSeekProvider, MiniMaxProvider } from "./providers/openai.js";
|
|
3
|
+
export { OpenAIProvider, QwenProvider, DeepSeekProvider, MiniMaxProvider, KimiProvider } from "./providers/openai.js";
|
|
4
4
|
export { OllamaProvider } from "./providers/ollama.js";
|
|
5
5
|
export { CircuitBreaker, normalizeToolCall } from "./providers/base.js";
|
|
6
6
|
export { tool, executeTools, readFile } from "./tools/index.js";
|
|
7
7
|
export { scanSkillDir, readSkillFile } from "./skills/loader.js";
|
|
8
8
|
export { WorkingMemory } from "./memory/working.js";
|
|
9
|
-
export { SinglePassHarness, HarnessLoop } from "./harness/harness.js";
|
|
9
|
+
export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
|
|
10
10
|
export { ScheduledPrompt } from "./signals/scheduled.js";
|
|
11
11
|
export { SignalGateway } from "./signals/gateway.js";
|
|
12
12
|
export { PermissionManager, PermissionMode } from "./safety/permissions.js";
|
|
13
|
+
// Kernel Governance — full pipeline (Permission → Veto → RateLimit → Constraint → Audit)
|
|
14
|
+
// @deepstrike/core is a CJS native addon; static ESM named re-export doesn't work,
|
|
15
|
+
// so we load it via createRequire and re-export with proper types preserved.
|
|
16
|
+
import { createRequire } from "module";
|
|
17
|
+
const _cjsRequire = createRequire(import.meta.url);
|
|
18
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
19
|
+
export const Governance = _cjsRequire("@deepstrike/core").Governance;
|
|
@@ -39,6 +39,8 @@ export interface DreamStore {
|
|
|
39
39
|
commit(agentId: string, result: CurationResult, existing: MemoryEntry[]): Promise<void>;
|
|
40
40
|
/** Semantic search over the agent's long-term memories. Called on demand during a run. */
|
|
41
41
|
search(agentId: string, query: string, topK?: number): Promise<MemoryEntry[]>;
|
|
42
|
+
/** Persist a completed session for future consolidation via `Agent.dream()`. */
|
|
43
|
+
saveSession(data: SessionData): Promise<void>;
|
|
42
44
|
}
|
|
43
45
|
export interface DreamResult {
|
|
44
46
|
sessionsProcessed: number;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import Anthropic from "@anthropic-ai/sdk";
|
|
2
|
-
import { CircuitBreaker, normalizeToolCall } from "./base.js";
|
|
2
|
+
import { CircuitBreaker, normalizeToolCall, toAnthropicContent } from "./base.js";
|
|
3
3
|
export class AnthropicProvider {
|
|
4
4
|
model;
|
|
5
5
|
client;
|
|
@@ -24,7 +24,7 @@ export class AnthropicProvider {
|
|
|
24
24
|
if (this.circuit.isOpen())
|
|
25
25
|
throw new Error("Circuit breaker open");
|
|
26
26
|
const system = messages.filter(m => m.role === "system").map(m => m.content).join("\n\n");
|
|
27
|
-
const msgs = messages.filter(m => m.role !== "system").map(m => ({ role: m.role, content: m
|
|
27
|
+
const msgs = messages.filter(m => m.role !== "system").map(m => ({ role: m.role, content: toAnthropicContent(m) }));
|
|
28
28
|
let lastErr;
|
|
29
29
|
for (let i = 0; i < this.maxRetries; i++) {
|
|
30
30
|
try {
|
|
@@ -60,7 +60,7 @@ export class AnthropicProvider {
|
|
|
60
60
|
}
|
|
61
61
|
async *stream(messages, tools, extensions) {
|
|
62
62
|
const system = messages.filter(m => m.role === "system").map(m => m.content).join("\n\n");
|
|
63
|
-
const msgs = messages.filter(m => m.role !== "system").map(m => ({ role: m.role, content: m
|
|
63
|
+
const msgs = messages.filter(m => m.role !== "system").map(m => ({ role: m.role, content: toAnthropicContent(m) }));
|
|
64
64
|
const toolBlocks = {};
|
|
65
65
|
const stream = this.client.messages.stream({
|
|
66
66
|
model: this.model,
|
package/dist/providers/base.d.ts
CHANGED
|
@@ -13,3 +13,6 @@ export declare function normalizeToolCall(id: string, name: string, args: unknow
|
|
|
13
13
|
name: string;
|
|
14
14
|
arguments: string;
|
|
15
15
|
} | null;
|
|
16
|
+
import type { Message } from "../types.js";
|
|
17
|
+
export declare function toAnthropicContent(msg: Message): string | Array<Record<string, unknown>>;
|
|
18
|
+
export declare function toOpenAIContent(msg: Message): string | Array<Record<string, unknown>>;
|