@deepstrike/sdk 0.1.4 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -25,6 +25,38 @@ export interface HarnessOutcome {
25
25
  feedback?: string;
26
26
  details?: CriterionResult[];
27
27
  }
28
+ export interface Verdict {
29
+ passed: boolean;
30
+ overallScore: number;
31
+ feedback: string;
32
+ details: CriterionResult[];
33
+ }
34
+ export type HarnessEvent = {
35
+ type: "token";
36
+ text: string;
37
+ } | {
38
+ type: "tool_call";
39
+ id: string;
40
+ name: string;
41
+ } | {
42
+ type: "tool_result";
43
+ callId: string;
44
+ content: string;
45
+ isError: boolean;
46
+ } | {
47
+ type: "supervising";
48
+ } | {
49
+ type: "revising";
50
+ verdict: Verdict;
51
+ } | {
52
+ type: "done";
53
+ verdict: Verdict;
54
+ iterations: number;
55
+ totalTokens: number;
56
+ status: string;
57
+ } | {
58
+ type: "max_attempts_reached";
59
+ };
28
60
  export interface QualityGate {
29
61
  evaluate(request: HarnessRequest, outcome: HarnessOutcome): Promise<boolean>;
30
62
  }
@@ -50,5 +82,5 @@ export declare class HarnessLoop {
50
82
  private maxAttempts;
51
83
  private skillDir?;
52
84
  constructor(agent: Agent, evalProvider: import("../types.js").LLMProvider, options?: HarnessLoopOptions);
53
- run(request: HarnessRequest): Promise<HarnessOutcome>;
85
+ runStreaming(request: HarnessRequest): AsyncIterable<HarnessEvent>;
54
86
  }
@@ -56,15 +56,44 @@ export class HarnessLoop {
56
56
  this.maxAttempts = options.maxAttempts ?? 3;
57
57
  this.skillDir = options.skillDir;
58
58
  }
59
- async run(request) {
59
+ async *runStreaming(request) {
60
60
  const kernel = await loadKernel();
61
61
  const pipeline = new kernel.EvalPipeline({ extractSkillOnPass: true });
62
62
  const criteria = request.criteria ?? [];
63
- let outcome = { result: "", passed: false, iterations: 0, totalTokens: 0, status: "error" };
63
+ // history tracks the full conversation; agent sees only the current goal message
64
+ // but we inject feedback as user messages to guide revision without discarding prior output
64
65
  let currentGoal = request.goal;
66
+ let lastIterations = 0;
67
+ let lastTotalTokens = 0;
68
+ let lastStatus = "error";
69
+ let lastResult = "";
65
70
  for (let attempt = 1; attempt <= this.maxAttempts; attempt++) {
66
- outcome = await runOnce(this.agent, { ...request, goal: currentGoal });
67
- const evalAction = pipeline.feedOutcome(request.goal, criteria, outcome.result, attempt);
71
+ // Stream agent output directly to caller
72
+ for await (const evt of this.agent.runStreaming(currentGoal, criteria.map(c => c.text), request.extensions)) {
73
+ if (evt.type === "text_delta") {
74
+ lastResult += evt.delta;
75
+ yield { type: "token", text: evt.delta };
76
+ }
77
+ else if (evt.type === "tool_call") {
78
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
79
+ const tc = evt;
80
+ yield { type: "tool_call", id: tc.id, name: tc.name };
81
+ }
82
+ else if (evt.type === "tool_result") {
83
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
84
+ const tr = evt;
85
+ yield { type: "tool_result", callId: tr.callId, content: tr.content, isError: tr.isError };
86
+ }
87
+ else if (evt.type === "done") {
88
+ const d = evt;
89
+ lastIterations = d.iterations;
90
+ lastTotalTokens = d.totalTokens;
91
+ lastStatus = d.status;
92
+ }
93
+ }
94
+ // Checkpoint: evaluator judges the completed output
95
+ yield { type: "supervising" };
96
+ const evalAction = pipeline.feedOutcome(request.goal, criteria, lastResult, attempt);
68
97
  if (evalAction.kind !== "evaluate")
69
98
  break;
70
99
  let evalText = "";
@@ -75,14 +104,13 @@ export class HarnessLoop {
75
104
  const doneAction = pipeline.feedEvalResult(evalText);
76
105
  if (doneAction.kind !== "done")
77
106
  break;
78
- outcome = {
79
- ...outcome,
107
+ const verdict = {
80
108
  passed: doneAction.passed ?? false,
81
- overallScore: doneAction.overallScore ?? undefined,
82
- feedback: doneAction.feedback ?? undefined,
83
- details: doneAction.details ?? undefined,
109
+ overallScore: doneAction.overallScore ?? 0,
110
+ feedback: doneAction.feedback ?? "",
111
+ details: doneAction.details ?? [],
84
112
  };
85
- if (doneAction.passed) {
113
+ if (verdict.passed) {
86
114
  if (doneAction.skill_candidate && this.skillDir) {
87
115
  const { name, description, whenToUse, content } = doneAction.skill_candidate;
88
116
  const fm = ["---", `name: ${name}`, `description: ${description}`,
@@ -90,11 +118,15 @@ export class HarnessLoop {
90
118
  .filter(Boolean).join("\n");
91
119
  await writeFile(path.join(this.skillDir, `${name}.md`), fm + content, "utf8");
92
120
  }
93
- return outcome;
121
+ yield { type: "done", verdict, iterations: lastIterations, totalTokens: lastTotalTokens, status: lastStatus };
122
+ return;
94
123
  }
95
- currentGoal = `${request.goal}\n\n[Previous attempt ${attempt} failed: ${doneAction.feedback}]`;
124
+ yield { type: "revising", verdict };
125
+ // Inject feedback as next turn's goal — agent sees its prior output + evaluator notes
126
+ currentGoal = `${request.goal}\n\n[Attempt ${attempt} feedback: ${verdict.feedback}]`;
127
+ lastResult = "";
96
128
  pipeline.reset();
97
129
  }
98
- return outcome;
130
+ yield { type: "max_attempts_reached" };
99
131
  }
100
132
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@deepstrike/sdk",
3
- "version": "0.1.4",
3
+ "version": "0.1.5",
4
4
  "description": "DeepStrike Node.js SDK",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
@@ -12,7 +12,7 @@
12
12
  },
13
13
  "dependencies": {
14
14
  "@anthropic-ai/sdk": "^0.39.0",
15
- "@deepstrike/core": "0.1.4",
15
+ "@deepstrike/core": "0.1.5",
16
16
  "openai": "^4.77.0"
17
17
  },
18
18
  "devDependencies": {