@deepstrike/sdk 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/harness/harness.d.ts +33 -1
- package/dist/harness/harness.js +45 -13
- package/package.json +2 -2
|
@@ -25,6 +25,38 @@ export interface HarnessOutcome {
|
|
|
25
25
|
feedback?: string;
|
|
26
26
|
details?: CriterionResult[];
|
|
27
27
|
}
|
|
28
|
+
export interface Verdict {
|
|
29
|
+
passed: boolean;
|
|
30
|
+
overallScore: number;
|
|
31
|
+
feedback: string;
|
|
32
|
+
details: CriterionResult[];
|
|
33
|
+
}
|
|
34
|
+
export type HarnessEvent = {
|
|
35
|
+
type: "token";
|
|
36
|
+
text: string;
|
|
37
|
+
} | {
|
|
38
|
+
type: "tool_call";
|
|
39
|
+
id: string;
|
|
40
|
+
name: string;
|
|
41
|
+
} | {
|
|
42
|
+
type: "tool_result";
|
|
43
|
+
callId: string;
|
|
44
|
+
content: string;
|
|
45
|
+
isError: boolean;
|
|
46
|
+
} | {
|
|
47
|
+
type: "supervising";
|
|
48
|
+
} | {
|
|
49
|
+
type: "revising";
|
|
50
|
+
verdict: Verdict;
|
|
51
|
+
} | {
|
|
52
|
+
type: "done";
|
|
53
|
+
verdict: Verdict;
|
|
54
|
+
iterations: number;
|
|
55
|
+
totalTokens: number;
|
|
56
|
+
status: string;
|
|
57
|
+
} | {
|
|
58
|
+
type: "max_attempts_reached";
|
|
59
|
+
};
|
|
28
60
|
export interface QualityGate {
|
|
29
61
|
evaluate(request: HarnessRequest, outcome: HarnessOutcome): Promise<boolean>;
|
|
30
62
|
}
|
|
@@ -50,5 +82,5 @@ export declare class HarnessLoop {
|
|
|
50
82
|
private maxAttempts;
|
|
51
83
|
private skillDir?;
|
|
52
84
|
constructor(agent: Agent, evalProvider: import("../types.js").LLMProvider, options?: HarnessLoopOptions);
|
|
53
|
-
|
|
85
|
+
runStreaming(request: HarnessRequest): AsyncIterable<HarnessEvent>;
|
|
54
86
|
}
|
package/dist/harness/harness.js
CHANGED
|
@@ -56,15 +56,44 @@ export class HarnessLoop {
|
|
|
56
56
|
this.maxAttempts = options.maxAttempts ?? 3;
|
|
57
57
|
this.skillDir = options.skillDir;
|
|
58
58
|
}
|
|
59
|
-
async
|
|
59
|
+
async *runStreaming(request) {
|
|
60
60
|
const kernel = await loadKernel();
|
|
61
61
|
const pipeline = new kernel.EvalPipeline({ extractSkillOnPass: true });
|
|
62
62
|
const criteria = request.criteria ?? [];
|
|
63
|
-
|
|
63
|
+
// history tracks the full conversation; agent sees only the current goal message
|
|
64
|
+
// but we inject feedback as user messages to guide revision without discarding prior output
|
|
64
65
|
let currentGoal = request.goal;
|
|
66
|
+
let lastIterations = 0;
|
|
67
|
+
let lastTotalTokens = 0;
|
|
68
|
+
let lastStatus = "error";
|
|
69
|
+
let lastResult = "";
|
|
65
70
|
for (let attempt = 1; attempt <= this.maxAttempts; attempt++) {
|
|
66
|
-
|
|
67
|
-
const
|
|
71
|
+
// Stream agent output directly to caller
|
|
72
|
+
for await (const evt of this.agent.runStreaming(currentGoal, criteria.map(c => c.text), request.extensions)) {
|
|
73
|
+
if (evt.type === "text_delta") {
|
|
74
|
+
lastResult += evt.delta;
|
|
75
|
+
yield { type: "token", text: evt.delta };
|
|
76
|
+
}
|
|
77
|
+
else if (evt.type === "tool_call") {
|
|
78
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
79
|
+
const tc = evt;
|
|
80
|
+
yield { type: "tool_call", id: tc.id, name: tc.name };
|
|
81
|
+
}
|
|
82
|
+
else if (evt.type === "tool_result") {
|
|
83
|
+
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
84
|
+
const tr = evt;
|
|
85
|
+
yield { type: "tool_result", callId: tr.callId, content: tr.content, isError: tr.isError };
|
|
86
|
+
}
|
|
87
|
+
else if (evt.type === "done") {
|
|
88
|
+
const d = evt;
|
|
89
|
+
lastIterations = d.iterations;
|
|
90
|
+
lastTotalTokens = d.totalTokens;
|
|
91
|
+
lastStatus = d.status;
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
// Checkpoint: evaluator judges the completed output
|
|
95
|
+
yield { type: "supervising" };
|
|
96
|
+
const evalAction = pipeline.feedOutcome(request.goal, criteria, lastResult, attempt);
|
|
68
97
|
if (evalAction.kind !== "evaluate")
|
|
69
98
|
break;
|
|
70
99
|
let evalText = "";
|
|
@@ -75,14 +104,13 @@ export class HarnessLoop {
|
|
|
75
104
|
const doneAction = pipeline.feedEvalResult(evalText);
|
|
76
105
|
if (doneAction.kind !== "done")
|
|
77
106
|
break;
|
|
78
|
-
|
|
79
|
-
...outcome,
|
|
107
|
+
const verdict = {
|
|
80
108
|
passed: doneAction.passed ?? false,
|
|
81
|
-
overallScore: doneAction.overallScore ??
|
|
82
|
-
feedback: doneAction.feedback ??
|
|
83
|
-
details: doneAction.details ??
|
|
109
|
+
overallScore: doneAction.overallScore ?? 0,
|
|
110
|
+
feedback: doneAction.feedback ?? "",
|
|
111
|
+
details: doneAction.details ?? [],
|
|
84
112
|
};
|
|
85
|
-
if (
|
|
113
|
+
if (verdict.passed) {
|
|
86
114
|
if (doneAction.skill_candidate && this.skillDir) {
|
|
87
115
|
const { name, description, whenToUse, content } = doneAction.skill_candidate;
|
|
88
116
|
const fm = ["---", `name: ${name}`, `description: ${description}`,
|
|
@@ -90,11 +118,15 @@ export class HarnessLoop {
|
|
|
90
118
|
.filter(Boolean).join("\n");
|
|
91
119
|
await writeFile(path.join(this.skillDir, `${name}.md`), fm + content, "utf8");
|
|
92
120
|
}
|
|
93
|
-
|
|
121
|
+
yield { type: "done", verdict, iterations: lastIterations, totalTokens: lastTotalTokens, status: lastStatus };
|
|
122
|
+
return;
|
|
94
123
|
}
|
|
95
|
-
|
|
124
|
+
yield { type: "revising", verdict };
|
|
125
|
+
// Inject feedback as next turn's goal — agent sees its prior output + evaluator notes
|
|
126
|
+
currentGoal = `${request.goal}\n\n[Attempt ${attempt} feedback: ${verdict.feedback}]`;
|
|
127
|
+
lastResult = "";
|
|
96
128
|
pipeline.reset();
|
|
97
129
|
}
|
|
98
|
-
|
|
130
|
+
yield { type: "max_attempts_reached" };
|
|
99
131
|
}
|
|
100
132
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@deepstrike/sdk",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.5",
|
|
4
4
|
"description": "DeepStrike Node.js SDK",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
},
|
|
13
13
|
"dependencies": {
|
|
14
14
|
"@anthropic-ai/sdk": "^0.39.0",
|
|
15
|
-
"@deepstrike/core": "0.1.
|
|
15
|
+
"@deepstrike/core": "0.1.5",
|
|
16
16
|
"openai": "^4.77.0"
|
|
17
17
|
},
|
|
18
18
|
"devDependencies": {
|