@deepstrike/sdk 0.1.1 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +133 -166
- package/dist/agent.d.ts +26 -13
- package/dist/agent.js +97 -62
- package/dist/harness/harness.d.ts +11 -0
- package/dist/harness/harness.js +32 -15
- package/dist/index.d.ts +7 -5
- package/dist/index.js +9 -2
- package/dist/providers/anthropic.js +3 -3
- package/dist/providers/base.d.ts +3 -0
- package/dist/providers/base.js +34 -0
- package/dist/providers/ollama.js +10 -1
- package/dist/providers/openai.d.ts +7 -0
- package/dist/providers/openai.js +63 -6
- package/dist/safety/permissions.d.ts +16 -3
- package/dist/safety/permissions.js +27 -10
- package/dist/signals/gateway.d.ts +21 -17
- package/dist/signals/gateway.js +31 -25
- package/dist/types.d.ts +36 -0
- package/package.json +2 -2
package/dist/agent.js
CHANGED
|
@@ -2,13 +2,15 @@ import { executeTools } from "./tools/index.js";
|
|
|
2
2
|
import { readSkillFile, scanSkillDir } from "./skills/loader.js";
|
|
3
3
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
4
4
|
async function loadKernel() {
|
|
5
|
-
|
|
5
|
+
const mod = await import("@deepstrike/core");
|
|
6
|
+
// CJS modules imported via ESM dynamic import expose exports under `.default`
|
|
7
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
8
|
+
return mod.default ?? mod;
|
|
6
9
|
}
|
|
7
10
|
export class Agent {
|
|
8
11
|
provider;
|
|
9
12
|
options;
|
|
10
13
|
tools = new Map();
|
|
11
|
-
blockedTools = new Set();
|
|
12
14
|
extensions;
|
|
13
15
|
skillDir;
|
|
14
16
|
knowledgeSource;
|
|
@@ -16,6 +18,9 @@ export class Agent {
|
|
|
16
18
|
dreamStore;
|
|
17
19
|
interrupted = false;
|
|
18
20
|
pendingInterrupt = false;
|
|
21
|
+
// Live telemetry — updated each runStreaming call
|
|
22
|
+
_turn = 0;
|
|
23
|
+
_pressure = 0;
|
|
19
24
|
constructor(provider, options) {
|
|
20
25
|
this.provider = provider;
|
|
21
26
|
this.options = options;
|
|
@@ -25,6 +30,10 @@ export class Agent {
|
|
|
25
30
|
this.signalSource = options.signalSource;
|
|
26
31
|
this.dreamStore = options.dreamStore;
|
|
27
32
|
}
|
|
33
|
+
/** Current turn index within the active run (0 before a run starts). */
|
|
34
|
+
get turn() { return this._turn; }
|
|
35
|
+
/** Context pressure ratio [0–1] from the kernel. Values > 0.8 trigger compression. */
|
|
36
|
+
get pressure() { return this._pressure; }
|
|
28
37
|
interrupt() {
|
|
29
38
|
this.interrupted = true;
|
|
30
39
|
}
|
|
@@ -37,21 +46,23 @@ export class Agent {
|
|
|
37
46
|
this.tools.delete(name);
|
|
38
47
|
return this;
|
|
39
48
|
}
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
49
|
+
/**
|
|
50
|
+
* Collect the full text response and return it.
|
|
51
|
+
* For richer control (streaming, tool events, token counts) use `runStreaming`.
|
|
52
|
+
*/
|
|
44
53
|
async run(goal, criteria, extensions) {
|
|
45
|
-
let
|
|
54
|
+
let content = "";
|
|
46
55
|
for await (const evt of this.runStreaming(goal, criteria, extensions)) {
|
|
47
|
-
if (evt.type === "
|
|
48
|
-
|
|
56
|
+
if (evt.type === "text_delta")
|
|
57
|
+
content += evt.delta;
|
|
49
58
|
}
|
|
50
|
-
return
|
|
59
|
+
return content;
|
|
51
60
|
}
|
|
52
61
|
async *runStreaming(goal, criteria, extensions) {
|
|
53
62
|
this.interrupted = false;
|
|
54
63
|
this.pendingInterrupt = false;
|
|
64
|
+
this._turn = 0;
|
|
65
|
+
this._pressure = 0;
|
|
55
66
|
const kernel = await loadKernel();
|
|
56
67
|
const ext = { ...this.extensions, ...(extensions ?? {}) };
|
|
57
68
|
const sm = new kernel.LoopStateMachine({
|
|
@@ -59,15 +70,13 @@ export class Agent {
|
|
|
59
70
|
maxTurns: this.options.maxTurns ?? 25,
|
|
60
71
|
timeoutMs: this.options.timeoutMs,
|
|
61
72
|
});
|
|
62
|
-
//
|
|
73
|
+
// Per-run SignalRouter — dedup state never leaks between runs.
|
|
63
74
|
const router = new kernel.SignalRouter(256);
|
|
64
75
|
const toolSchemas = Array.from(this.tools.values()).map(t => t.schema);
|
|
65
76
|
sm.setTools(toolSchemas);
|
|
66
|
-
// Scan skill directory and register metadata with the kernel.
|
|
67
|
-
// The kernel will auto-inject the `skill` meta-tool into every CallLLM action.
|
|
68
77
|
if (this.skillDir) {
|
|
69
78
|
const skillMetas = await scanSkillDir(this.skillDir);
|
|
70
|
-
sm.setAvailableSkills(skillMetas.map(m => ({
|
|
79
|
+
sm.setAvailableSkills(skillMetas.map((m) => ({
|
|
71
80
|
name: m.name,
|
|
72
81
|
description: m.description,
|
|
73
82
|
whenToUse: m.whenToUse,
|
|
@@ -75,17 +84,18 @@ export class Agent {
|
|
|
75
84
|
estimatedTokens: m.estimatedTokens ?? 0,
|
|
76
85
|
})));
|
|
77
86
|
}
|
|
78
|
-
// Enable the memory meta-tool when both dreamStore and agentId are provided.
|
|
79
87
|
if (this.dreamStore && this.options.agentId) {
|
|
80
88
|
sm.setMemoryEnabled(true);
|
|
81
89
|
}
|
|
82
|
-
// Enable the knowledge meta-tool when a KnowledgeSource is configured.
|
|
83
90
|
if (this.knowledgeSource) {
|
|
84
91
|
sm.setKnowledgeEnabled(true);
|
|
85
92
|
}
|
|
86
93
|
let action = sm.start({ goal, criteria: criteria ?? [] });
|
|
87
|
-
let finalText = "";
|
|
88
94
|
while (!sm.isTerminal()) {
|
|
95
|
+
// Update telemetry
|
|
96
|
+
this._turn = sm.turn;
|
|
97
|
+
this._pressure = sm.pressure();
|
|
98
|
+
// Hard interrupt
|
|
89
99
|
if (this.interrupted) {
|
|
90
100
|
action = sm.feedTimeout();
|
|
91
101
|
break;
|
|
@@ -95,18 +105,22 @@ export class Agent {
|
|
|
95
105
|
action = sm.feedTimeout();
|
|
96
106
|
break;
|
|
97
107
|
}
|
|
108
|
+
// Drain context-compression observations
|
|
109
|
+
sm.takeObservations();
|
|
98
110
|
// Poll signal source and route through kernel SignalRouter
|
|
99
111
|
if (this.signalSource) {
|
|
100
112
|
const sig = await this.signalSource.nextSignal();
|
|
101
113
|
if (sig) {
|
|
114
|
+
const sigAny = sig;
|
|
102
115
|
const kernelSig = {
|
|
103
116
|
id: crypto.randomUUID(),
|
|
104
|
-
source:
|
|
105
|
-
signalType:
|
|
106
|
-
urgency:
|
|
117
|
+
source: sigAny.source ?? "custom",
|
|
118
|
+
signalType: sigAny.signalType ?? "event",
|
|
119
|
+
urgency: sigAny.urgency
|
|
120
|
+
?? (sig.kind === "interrupt" ? "critical" : "normal"),
|
|
107
121
|
summary: String(sig.payload?.goal ?? sig.kind),
|
|
108
122
|
payload: JSON.stringify(sig.payload ?? {}),
|
|
109
|
-
dedupeKey:
|
|
123
|
+
dedupeKey: sigAny.dedupeKey ?? null,
|
|
110
124
|
timestampMs: Date.now(),
|
|
111
125
|
};
|
|
112
126
|
const disposition = router.ingest(kernelSig, action.kind === "execute_tools");
|
|
@@ -114,20 +128,35 @@ export class Agent {
|
|
|
114
128
|
action = sm.feedTimeout();
|
|
115
129
|
break;
|
|
116
130
|
}
|
|
117
|
-
if (disposition === "interrupt")
|
|
131
|
+
if (disposition === "interrupt")
|
|
118
132
|
this.pendingInterrupt = true;
|
|
119
|
-
}
|
|
120
|
-
// "queue" → buffered; "observe" / "ignore" / "dropped" → no action
|
|
121
133
|
}
|
|
122
134
|
}
|
|
123
|
-
|
|
135
|
+
// Drain previously queued signals — apply any high-urgency ones
|
|
136
|
+
let queued = router.next();
|
|
137
|
+
while (queued) {
|
|
138
|
+
if (queued.urgency === "critical") {
|
|
139
|
+
action = sm.feedTimeout();
|
|
140
|
+
break;
|
|
141
|
+
}
|
|
142
|
+
if (queued.urgency === "high")
|
|
143
|
+
this.pendingInterrupt = true;
|
|
144
|
+
queued = router.next();
|
|
145
|
+
}
|
|
146
|
+
if (this.interrupted || (sm.isTerminal()))
|
|
147
|
+
break;
|
|
124
148
|
if (action.kind === "call_llm") {
|
|
125
|
-
finalText = "";
|
|
126
149
|
const finalToolCalls = [];
|
|
150
|
+
let finalText = "";
|
|
127
151
|
const messages = (action.messages ?? []);
|
|
128
152
|
const tools = (action.tools ?? []);
|
|
153
|
+
let turnTokens = 0;
|
|
129
154
|
try {
|
|
130
155
|
for await (const evt of this.provider.stream(messages, tools, Object.keys(ext).length ? ext : undefined)) {
|
|
156
|
+
if (evt.type === "usage") {
|
|
157
|
+
turnTokens = evt.totalTokens;
|
|
158
|
+
continue;
|
|
159
|
+
}
|
|
131
160
|
yield evt;
|
|
132
161
|
if (evt.type === "text_delta")
|
|
133
162
|
finalText += evt.delta;
|
|
@@ -142,45 +171,46 @@ export class Agent {
|
|
|
142
171
|
action = sm.feedTimeout();
|
|
143
172
|
break;
|
|
144
173
|
}
|
|
145
|
-
action = sm.feedLlmResponse({ role: "assistant", content: finalText, toolCalls: finalToolCalls });
|
|
174
|
+
action = sm.feedLlmResponse({ role: "assistant", content: finalText, toolCalls: finalToolCalls, tokenCount: turnTokens || undefined });
|
|
146
175
|
}
|
|
147
176
|
else if (action.kind === "execute_tools") {
|
|
148
177
|
const allCalls = action.calls ?? [];
|
|
149
|
-
// Governance
|
|
178
|
+
// Governance evaluation
|
|
150
179
|
const permittedCalls = [];
|
|
180
|
+
const deniedResults = [];
|
|
151
181
|
for (const c of allCalls) {
|
|
152
|
-
if (this.blockedTools.has(c.name)) {
|
|
153
|
-
yield { type: "error", message: `tool blocked: ${c.name}` };
|
|
154
|
-
continue;
|
|
155
|
-
}
|
|
156
182
|
if (this.options.governance) {
|
|
157
183
|
const verdict = this.options.governance.evaluate(c.name, c.arguments);
|
|
158
184
|
if (verdict.kind === "deny") {
|
|
159
|
-
|
|
185
|
+
const msg = `permission denied: ${c.name} — ${verdict.reason ?? ""}`;
|
|
186
|
+
yield { type: "error", message: msg };
|
|
187
|
+
deniedResults.push({ callId: c.id, name: c.name, output: msg, isError: true });
|
|
188
|
+
continue;
|
|
189
|
+
}
|
|
190
|
+
if (verdict.kind === "rate_limited") {
|
|
191
|
+
const msg = `rate limited: ${c.name} — retry after ${verdict.retryAfterMs ?? 0}ms`;
|
|
192
|
+
yield { type: "error", message: msg };
|
|
193
|
+
deniedResults.push({ callId: c.id, name: c.name, output: msg, isError: true });
|
|
160
194
|
continue;
|
|
161
195
|
}
|
|
162
196
|
if (verdict.kind === "ask_user") {
|
|
163
|
-
yield { type: "permission_request", callId: c.id, toolName: c.name, arguments: c.arguments, reason: verdict.reason };
|
|
197
|
+
yield { type: "permission_request", callId: c.id, toolName: c.name, arguments: c.arguments, reason: verdict.reason ?? "" };
|
|
198
|
+
deniedResults.push({ callId: c.id, name: c.name, output: `awaiting user approval: ${c.name}`, isError: true });
|
|
164
199
|
continue;
|
|
165
200
|
}
|
|
166
201
|
}
|
|
167
202
|
permittedCalls.push(c);
|
|
168
203
|
}
|
|
169
|
-
const
|
|
170
|
-
|
|
171
|
-
const
|
|
172
|
-
|
|
173
|
-
const memoryCalls = calls.filter((c) => c.name === "memory");
|
|
174
|
-
// Intercept `knowledge` meta-tool calls: retrieve from KnowledgeSource.
|
|
175
|
-
const knowledgeCalls = calls.filter((c) => c.name === "knowledge");
|
|
176
|
-
const regularCalls = calls.filter((c) => c.name !== "skill" && c.name !== "memory" && c.name !== "knowledge");
|
|
204
|
+
const skillCalls = permittedCalls.filter((c) => c.name === "skill");
|
|
205
|
+
const memoryCalls = permittedCalls.filter((c) => c.name === "memory");
|
|
206
|
+
const knowledgeCalls = permittedCalls.filter((c) => c.name === "knowledge");
|
|
207
|
+
const regularCalls = permittedCalls.filter((c) => c.name !== "skill" && c.name !== "memory" && c.name !== "knowledge");
|
|
177
208
|
const skillResults = this.skillDir
|
|
178
209
|
? await Promise.all(skillCalls.map(async (c) => {
|
|
179
210
|
const args = tryParseJson(c.arguments);
|
|
180
211
|
const name = String(args?.name ?? "");
|
|
181
212
|
const content = await readSkillFile(this.skillDir, name);
|
|
182
|
-
|
|
183
|
-
return { callId: c.id, name: c.name, output, isError: !content };
|
|
213
|
+
return { callId: c.id, name: c.name, output: content ?? `Skill "${name}" not found.`, isError: !content };
|
|
184
214
|
}))
|
|
185
215
|
: skillCalls.map((c) => ({ callId: c.id, name: c.name, output: "No skill directory configured.", isError: true }));
|
|
186
216
|
const memoryResults = (this.dreamStore && this.options.agentId)
|
|
@@ -190,7 +220,7 @@ export class Agent {
|
|
|
190
220
|
const topK = typeof args?.top_k === "number" ? args.top_k : 5;
|
|
191
221
|
const entries = await this.dreamStore.search(this.options.agentId, query, topK);
|
|
192
222
|
const output = entries.length
|
|
193
|
-
? entries.map(e => `[score=${e.score.toFixed(3)}] ${e.text}`).join("\n---\n")
|
|
223
|
+
? entries.map((e) => `[score=${e.score.toFixed(3)}] ${e.text}`).join("\n---\n")
|
|
194
224
|
: "No relevant memories found.";
|
|
195
225
|
return { callId: c.id, name: c.name, output, isError: false };
|
|
196
226
|
}))
|
|
@@ -205,18 +235,18 @@ export class Agent {
|
|
|
205
235
|
return { callId: c.id, name: c.name, output, isError: false };
|
|
206
236
|
}))
|
|
207
237
|
: knowledgeCalls.map((c) => ({ callId: c.id, name: c.name, output: "Knowledge source not configured.", isError: true }));
|
|
208
|
-
// Yield all meta-tool results
|
|
209
238
|
for (const r of [...skillResults, ...memoryResults, ...knowledgeResults])
|
|
210
239
|
yield { type: "tool_result", callId: r.callId, name: r.name, content: r.output, isError: r.isError };
|
|
211
240
|
const results = await executeTools(regularCalls, this.tools);
|
|
212
241
|
for (const r of results) {
|
|
213
|
-
const name = regularCalls.find(c => c.id === r.callId)?.name ?? "";
|
|
242
|
+
const name = regularCalls.find((c) => c.id === r.callId)?.name ?? "";
|
|
214
243
|
yield { type: "tool_result", callId: r.callId, name, content: r.output, isError: r.isError };
|
|
215
244
|
}
|
|
216
245
|
action = sm.feedToolResults([
|
|
217
|
-
...
|
|
218
|
-
...
|
|
219
|
-
...
|
|
246
|
+
...deniedResults.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
247
|
+
...skillResults.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
248
|
+
...memoryResults.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
249
|
+
...knowledgeResults.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
220
250
|
...results.map(r => ({ callId: r.callId, output: r.output, isError: r.isError })),
|
|
221
251
|
]);
|
|
222
252
|
}
|
|
@@ -225,20 +255,27 @@ export class Agent {
|
|
|
225
255
|
}
|
|
226
256
|
}
|
|
227
257
|
const result = action.result;
|
|
258
|
+
this._turn = sm.turn;
|
|
259
|
+
this._pressure = sm.pressure();
|
|
260
|
+
const status = result?.termination === "completed" ? "success" : (result?.termination ?? "error");
|
|
261
|
+
// turnsUsed counts tool execution rounds; for single-turn text-only runs it's 0.
|
|
262
|
+
// Map to iterations: at least 1 if we got a result.
|
|
263
|
+
const iterations = result ? Math.max(1, result.turnsUsed) : 0;
|
|
228
264
|
yield {
|
|
229
265
|
type: "done",
|
|
230
|
-
iterations
|
|
231
|
-
totalTokens: Number(result
|
|
232
|
-
status
|
|
266
|
+
iterations,
|
|
267
|
+
totalTokens: result?.totalTokensUsed ? Number(result.totalTokensUsed) : 0,
|
|
268
|
+
status,
|
|
233
269
|
};
|
|
234
270
|
}
|
|
235
271
|
/**
|
|
236
|
-
* Trigger
|
|
272
|
+
* Trigger the idle dreaming cycle for this agent.
|
|
273
|
+
* Requires `dreamStore` and `agentId` to be configured.
|
|
237
274
|
*
|
|
238
|
-
* Phase 1 — kernel rule-based analysis + LLM prompt assembly
|
|
239
|
-
* Phase 2 — LLM synthesis call (
|
|
240
|
-
* Phase 3 — kernel parses + curates results
|
|
241
|
-
* Phase 4 — commit delta to DreamStore (
|
|
275
|
+
* Phase 1 — kernel rule-based analysis + LLM prompt assembly
|
|
276
|
+
* Phase 2 — LLM synthesis call (I/O)
|
|
277
|
+
* Phase 3 — kernel parses + curates results
|
|
278
|
+
* Phase 4 — commit delta to DreamStore (I/O)
|
|
242
279
|
*/
|
|
243
280
|
async dream(agentId, nowMs = Date.now()) {
|
|
244
281
|
if (!this.dreamStore)
|
|
@@ -256,9 +293,7 @@ export class Agent {
|
|
|
256
293
|
role: m.role,
|
|
257
294
|
content: m.content,
|
|
258
295
|
tokenCount: m.tokenCount,
|
|
259
|
-
toolCalls: (m.toolCalls ?? []).map(tc => ({
|
|
260
|
-
id: tc.id, name: tc.name, arguments: tc.arguments,
|
|
261
|
-
})),
|
|
296
|
+
toolCalls: (m.toolCalls ?? []).map(tc => ({ id: tc.id, name: tc.name, arguments: tc.arguments })),
|
|
262
297
|
})),
|
|
263
298
|
metadata: JSON.stringify(s.metadata ?? null),
|
|
264
299
|
createdAtMs: s.createdAtMs,
|
|
@@ -271,7 +306,7 @@ export class Agent {
|
|
|
271
306
|
}));
|
|
272
307
|
const pipeline = new kernel.IdlePipeline(agentId);
|
|
273
308
|
const action1 = pipeline.feedTrigger(kernelSessions, kernelMemories, nowMs);
|
|
274
|
-
if (action1.kind === "noop") {
|
|
309
|
+
if (action1.kind === "noop" || action1.kind === "aborted") {
|
|
275
310
|
return { sessionsProcessed: 0, insightsExtracted: 0, entriesAdded: 0, entriesRemoved: 0 };
|
|
276
311
|
}
|
|
277
312
|
if (action1.kind !== "synthesize_insights") {
|
|
@@ -13,11 +13,22 @@ export interface HarnessOutcome {
|
|
|
13
13
|
/** Feedback from the evaluator LLM — injected into the next attempt's goal. */
|
|
14
14
|
feedback?: string;
|
|
15
15
|
}
|
|
16
|
+
export interface QualityGate {
|
|
17
|
+
evaluate(request: HarnessRequest, outcome: HarnessOutcome): Promise<boolean>;
|
|
18
|
+
}
|
|
16
19
|
export declare class SinglePassHarness {
|
|
17
20
|
private agent;
|
|
18
21
|
constructor(agent: Agent);
|
|
19
22
|
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
20
23
|
}
|
|
24
|
+
/** Retry loop driven by a pluggable QualityGate (not LLM-as-judge). */
|
|
25
|
+
export declare class EvalLoopHarness {
|
|
26
|
+
private agent;
|
|
27
|
+
private gate;
|
|
28
|
+
private maxAttempts;
|
|
29
|
+
constructor(agent: Agent, gate: QualityGate, maxAttempts?: number);
|
|
30
|
+
run(request: HarnessRequest): Promise<HarnessOutcome>;
|
|
31
|
+
}
|
|
21
32
|
export interface HarnessLoopOptions {
|
|
22
33
|
maxAttempts?: number;
|
|
23
34
|
/** Directory to write distilled skills into. Requires the agent to have skillDir set. */
|
package/dist/harness/harness.js
CHANGED
|
@@ -2,7 +2,9 @@ import { writeFile } from "fs/promises";
|
|
|
2
2
|
import path from "path";
|
|
3
3
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
4
4
|
async function loadKernel() {
|
|
5
|
-
|
|
5
|
+
const mod = await import("@deepstrike/core");
|
|
6
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
7
|
+
return mod.default ?? mod;
|
|
6
8
|
}
|
|
7
9
|
async function runOnce(agent, req) {
|
|
8
10
|
let text = "";
|
|
@@ -30,6 +32,27 @@ export class SinglePassHarness {
|
|
|
30
32
|
return { ...await runOnce(this.agent, request), passed: true };
|
|
31
33
|
}
|
|
32
34
|
}
|
|
35
|
+
/** Retry loop driven by a pluggable QualityGate (not LLM-as-judge). */
|
|
36
|
+
export class EvalLoopHarness {
|
|
37
|
+
agent;
|
|
38
|
+
gate;
|
|
39
|
+
maxAttempts;
|
|
40
|
+
constructor(agent, gate, maxAttempts = 3) {
|
|
41
|
+
this.agent = agent;
|
|
42
|
+
this.gate = gate;
|
|
43
|
+
this.maxAttempts = maxAttempts;
|
|
44
|
+
}
|
|
45
|
+
async run(request) {
|
|
46
|
+
let outcome = { result: "", passed: false, iterations: 0, totalTokens: 0, status: "error" };
|
|
47
|
+
for (let i = 0; i < this.maxAttempts; i++) {
|
|
48
|
+
outcome = await runOnce(this.agent, request);
|
|
49
|
+
if (await this.gate.evaluate(request, outcome)) {
|
|
50
|
+
return { ...outcome, passed: true };
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
return outcome;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
33
56
|
/**
|
|
34
57
|
* Eval loop with LLM-as-judge and feedback injection.
|
|
35
58
|
*
|
|
@@ -55,13 +78,8 @@ export class HarnessLoop {
|
|
|
55
78
|
let currentGoal = request.goal;
|
|
56
79
|
for (let attempt = 1; attempt <= this.maxAttempts; attempt++) {
|
|
57
80
|
outcome = await runOnce(this.agent, { ...request, goal: currentGoal });
|
|
58
|
-
// Phase 1: kernel builds eval prompt
|
|
59
|
-
const evalAction = pipeline.feedOutcome(
|
|
60
|
-
goal: request.goal,
|
|
61
|
-
criteria: request.criteria ?? [],
|
|
62
|
-
result: outcome.result,
|
|
63
|
-
attempt,
|
|
64
|
-
});
|
|
81
|
+
// Phase 1: kernel builds eval prompt (positional args per kernel API)
|
|
82
|
+
const evalAction = pipeline.feedOutcome(request.goal, request.criteria ?? [], outcome.result, attempt);
|
|
65
83
|
if (evalAction.kind !== "evaluate")
|
|
66
84
|
break;
|
|
67
85
|
// Phase 2: SDK calls evaluator LLM
|
|
@@ -71,14 +89,13 @@ export class HarnessLoop {
|
|
|
71
89
|
evalText += evt.delta;
|
|
72
90
|
}
|
|
73
91
|
// Phase 3: kernel parses verdict
|
|
74
|
-
const doneAction = pipeline.feedEvalResult(
|
|
92
|
+
const doneAction = pipeline.feedEvalResult(evalText);
|
|
75
93
|
if (doneAction.kind !== "done")
|
|
76
94
|
break;
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
const { name, description, whenToUse, content } = evalResult.skillCandidate;
|
|
95
|
+
outcome = { ...outcome, passed: doneAction.passed, feedback: doneAction.feedback };
|
|
96
|
+
if (doneAction.passed) {
|
|
97
|
+
if (doneAction.skill_candidate && this.skillDir) {
|
|
98
|
+
const { name, description, whenToUse, content } = doneAction.skill_candidate;
|
|
82
99
|
const frontmatter = [
|
|
83
100
|
"---",
|
|
84
101
|
`name: ${name}`,
|
|
@@ -92,7 +109,7 @@ export class HarnessLoop {
|
|
|
92
109
|
return outcome;
|
|
93
110
|
}
|
|
94
111
|
// Inject feedback into next attempt's goal
|
|
95
|
-
currentGoal = `${request.goal}\n\n[Previous attempt ${attempt} failed: ${
|
|
112
|
+
currentGoal = `${request.goal}\n\n[Previous attempt ${attempt} failed: ${doneAction.feedback}]`;
|
|
96
113
|
pipeline.reset();
|
|
97
114
|
}
|
|
98
115
|
return outcome;
|
package/dist/index.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
export { Agent } from "./agent.js";
|
|
2
2
|
export type { AgentOptions } from "./agent.js";
|
|
3
3
|
export { AnthropicProvider } from "./providers/anthropic.js";
|
|
4
|
-
export { OpenAIProvider, QwenProvider, DeepSeekProvider, MiniMaxProvider } from "./providers/openai.js";
|
|
4
|
+
export { OpenAIProvider, QwenProvider, DeepSeekProvider, MiniMaxProvider, KimiProvider } from "./providers/openai.js";
|
|
5
5
|
export { OllamaProvider } from "./providers/ollama.js";
|
|
6
6
|
export { CircuitBreaker, normalizeToolCall } from "./providers/base.js";
|
|
7
7
|
export { tool, executeTools, readFile } from "./tools/index.js";
|
|
@@ -11,11 +11,13 @@ export type { SkillMetadata } from "./skills/loader.js";
|
|
|
11
11
|
export { WorkingMemory } from "./memory/working.js";
|
|
12
12
|
export type { DreamStore, DreamResult, SessionData, SessionMessage, MemoryEntry, CurationResult, CurationStats, } from "./memory/protocols.js";
|
|
13
13
|
export type { KnowledgeSource } from "./knowledge/source.js";
|
|
14
|
-
export { SinglePassHarness, HarnessLoop } from "./harness/harness.js";
|
|
15
|
-
export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions } from "./harness/harness.js";
|
|
14
|
+
export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
|
|
15
|
+
export type { HarnessRequest, HarnessOutcome, HarnessLoopOptions, QualityGate } from "./harness/harness.js";
|
|
16
16
|
export { ScheduledPrompt } from "./signals/scheduled.js";
|
|
17
17
|
export { SignalGateway } from "./signals/gateway.js";
|
|
18
18
|
export type { RuntimeSignal, SignalSource } from "./signals/types.js";
|
|
19
19
|
export { PermissionManager, PermissionMode } from "./safety/permissions.js";
|
|
20
|
-
export type { PermissionDecision } from "./safety/permissions.js";
|
|
21
|
-
export
|
|
20
|
+
export type { PermissionDecision, Permission } from "./safety/permissions.js";
|
|
21
|
+
export declare const Governance: typeof import("@deepstrike/core").Governance;
|
|
22
|
+
export type { GovernanceVerdictObj as GovernanceVerdict } from "@deepstrike/core";
|
|
23
|
+
export type { Message, ToolCall, ToolResult, ToolSchema, ContentPart, TextPart, ImagePart, AudioPart, StreamEvent, TextDelta, ThinkingDelta, ToolCallEvent, ToolResultEvent, DoneEvent, ErrorEvent, PermissionRequestEvent, LLMProvider, RetryConfig, TokenUsage, ProviderToolSpec, } from "./types.js";
|
package/dist/index.js
CHANGED
|
@@ -1,12 +1,19 @@
|
|
|
1
1
|
export { Agent } from "./agent.js";
|
|
2
2
|
export { AnthropicProvider } from "./providers/anthropic.js";
|
|
3
|
-
export { OpenAIProvider, QwenProvider, DeepSeekProvider, MiniMaxProvider } from "./providers/openai.js";
|
|
3
|
+
export { OpenAIProvider, QwenProvider, DeepSeekProvider, MiniMaxProvider, KimiProvider } from "./providers/openai.js";
|
|
4
4
|
export { OllamaProvider } from "./providers/ollama.js";
|
|
5
5
|
export { CircuitBreaker, normalizeToolCall } from "./providers/base.js";
|
|
6
6
|
export { tool, executeTools, readFile } from "./tools/index.js";
|
|
7
7
|
export { scanSkillDir, readSkillFile } from "./skills/loader.js";
|
|
8
8
|
export { WorkingMemory } from "./memory/working.js";
|
|
9
|
-
export { SinglePassHarness, HarnessLoop } from "./harness/harness.js";
|
|
9
|
+
export { SinglePassHarness, EvalLoopHarness, HarnessLoop } from "./harness/harness.js";
|
|
10
10
|
export { ScheduledPrompt } from "./signals/scheduled.js";
|
|
11
11
|
export { SignalGateway } from "./signals/gateway.js";
|
|
12
12
|
export { PermissionManager, PermissionMode } from "./safety/permissions.js";
|
|
13
|
+
// Kernel Governance — full pipeline (Permission → Veto → RateLimit → Constraint → Audit)
|
|
14
|
+
// @deepstrike/core is a CJS native addon; static ESM named re-export doesn't work,
|
|
15
|
+
// so we load it via createRequire and re-export with proper types preserved.
|
|
16
|
+
import { createRequire } from "module";
|
|
17
|
+
const _cjsRequire = createRequire(import.meta.url);
|
|
18
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
19
|
+
export const Governance = _cjsRequire("@deepstrike/core").Governance;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import Anthropic from "@anthropic-ai/sdk";
|
|
2
|
-
import { CircuitBreaker, normalizeToolCall } from "./base.js";
|
|
2
|
+
import { CircuitBreaker, normalizeToolCall, toAnthropicContent } from "./base.js";
|
|
3
3
|
export class AnthropicProvider {
|
|
4
4
|
model;
|
|
5
5
|
client;
|
|
@@ -24,7 +24,7 @@ export class AnthropicProvider {
|
|
|
24
24
|
if (this.circuit.isOpen())
|
|
25
25
|
throw new Error("Circuit breaker open");
|
|
26
26
|
const system = messages.filter(m => m.role === "system").map(m => m.content).join("\n\n");
|
|
27
|
-
const msgs = messages.filter(m => m.role !== "system").map(m => ({ role: m.role, content: m
|
|
27
|
+
const msgs = messages.filter(m => m.role !== "system").map(m => ({ role: m.role, content: toAnthropicContent(m) }));
|
|
28
28
|
let lastErr;
|
|
29
29
|
for (let i = 0; i < this.maxRetries; i++) {
|
|
30
30
|
try {
|
|
@@ -60,7 +60,7 @@ export class AnthropicProvider {
|
|
|
60
60
|
}
|
|
61
61
|
async *stream(messages, tools, extensions) {
|
|
62
62
|
const system = messages.filter(m => m.role === "system").map(m => m.content).join("\n\n");
|
|
63
|
-
const msgs = messages.filter(m => m.role !== "system").map(m => ({ role: m.role, content: m
|
|
63
|
+
const msgs = messages.filter(m => m.role !== "system").map(m => ({ role: m.role, content: toAnthropicContent(m) }));
|
|
64
64
|
const toolBlocks = {};
|
|
65
65
|
const stream = this.client.messages.stream({
|
|
66
66
|
model: this.model,
|
package/dist/providers/base.d.ts
CHANGED
|
@@ -13,3 +13,6 @@ export declare function normalizeToolCall(id: string, name: string, args: unknow
|
|
|
13
13
|
name: string;
|
|
14
14
|
arguments: string;
|
|
15
15
|
} | null;
|
|
16
|
+
import type { Message } from "../types.js";
|
|
17
|
+
export declare function toAnthropicContent(msg: Message): string | Array<Record<string, unknown>>;
|
|
18
|
+
export declare function toOpenAIContent(msg: Message): string | Array<Record<string, unknown>>;
|
package/dist/providers/base.js
CHANGED
|
@@ -44,3 +44,37 @@ export function normalizeToolCall(id, name, args) {
|
|
|
44
44
|
}
|
|
45
45
|
return { id: String(id ?? ""), name: n, arguments: JSON.stringify(parsed) };
|
|
46
46
|
}
|
|
47
|
+
export function toAnthropicContent(msg) {
|
|
48
|
+
if (!msg.contentParts?.length)
|
|
49
|
+
return msg.content;
|
|
50
|
+
return msg.contentParts.map(p => {
|
|
51
|
+
if (p.type === "text")
|
|
52
|
+
return { type: "text", text: p.text };
|
|
53
|
+
if (p.type === "image") {
|
|
54
|
+
if (p.data) {
|
|
55
|
+
return { type: "image", source: { type: "base64", media_type: p.mediaType ?? "image/png", data: p.data } };
|
|
56
|
+
}
|
|
57
|
+
return { type: "image", source: { type: "url", url: p.url } };
|
|
58
|
+
}
|
|
59
|
+
if (p.type === "audio") {
|
|
60
|
+
return { type: "text", text: `[audio: ${p.mediaType}]` };
|
|
61
|
+
}
|
|
62
|
+
return { type: "text", text: "" };
|
|
63
|
+
});
|
|
64
|
+
}
|
|
65
|
+
export function toOpenAIContent(msg) {
|
|
66
|
+
if (!msg.contentParts?.length)
|
|
67
|
+
return msg.content;
|
|
68
|
+
return msg.contentParts.map(p => {
|
|
69
|
+
if (p.type === "text")
|
|
70
|
+
return { type: "text", text: p.text };
|
|
71
|
+
if (p.type === "image") {
|
|
72
|
+
const url = p.data ? `data:${p.mediaType ?? "image/png"};base64,${p.data}` : p.url;
|
|
73
|
+
return { type: "image_url", image_url: { url, ...(p.detail ? { detail: p.detail } : {}) } };
|
|
74
|
+
}
|
|
75
|
+
if (p.type === "audio") {
|
|
76
|
+
return { type: "input_audio", input_audio: { data: p.data, format: p.mediaType?.split("/")[1] ?? "wav" } };
|
|
77
|
+
}
|
|
78
|
+
return { type: "text", text: "" };
|
|
79
|
+
});
|
|
80
|
+
}
|
package/dist/providers/ollama.js
CHANGED
|
@@ -7,7 +7,16 @@ export class OllamaProvider {
|
|
|
7
7
|
this.baseUrl = baseUrl;
|
|
8
8
|
}
|
|
9
9
|
toOllamaMessages(messages) {
|
|
10
|
-
return messages.map(m =>
|
|
10
|
+
return messages.map(m => {
|
|
11
|
+
const images = [];
|
|
12
|
+
if (m.contentParts?.length) {
|
|
13
|
+
for (const p of m.contentParts) {
|
|
14
|
+
if (p.type === "image" && p.data)
|
|
15
|
+
images.push(p.data);
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
return { role: m.role, content: m.content, ...(images.length ? { images } : {}) };
|
|
19
|
+
});
|
|
11
20
|
}
|
|
12
21
|
async complete(messages, tools) {
|
|
13
22
|
const resp = await fetch(`${this.baseUrl}/api/chat`, {
|
|
@@ -27,6 +27,7 @@ export declare class QwenProvider extends OpenAIProvider {
|
|
|
27
27
|
maxRetries: number;
|
|
28
28
|
baseDelay: number;
|
|
29
29
|
});
|
|
30
|
+
stream(messages: Message[], tools: ToolSchema[], extensions?: Record<string, unknown>): AsyncIterable<StreamEvent>;
|
|
30
31
|
}
|
|
31
32
|
export declare class DeepSeekProvider extends OpenAIProvider {
|
|
32
33
|
constructor(apiKey: string, model?: string, retry?: {
|
|
@@ -42,3 +43,9 @@ export declare class MiniMaxProvider extends OpenAIProvider {
|
|
|
42
43
|
});
|
|
43
44
|
stream(messages: Message[], tools: ToolSchema[], extensions?: Record<string, unknown>): AsyncIterable<StreamEvent>;
|
|
44
45
|
}
|
|
46
|
+
export declare class KimiProvider extends OpenAIProvider {
|
|
47
|
+
constructor(apiKey: string, model?: string, retry?: {
|
|
48
|
+
maxRetries: number;
|
|
49
|
+
baseDelay: number;
|
|
50
|
+
});
|
|
51
|
+
}
|