@deepstrike/sdk 0.1.10 → 0.1.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -0
- package/dist/agent.d.ts +10 -9
- package/dist/agent.js +64 -12
- package/dist/collaboration/contract.d.ts +55 -0
- package/dist/collaboration/contract.js +65 -0
- package/dist/collaboration/handoff.d.ts +79 -0
- package/dist/collaboration/handoff.js +98 -0
- package/dist/collaboration/harness.d.ts +55 -0
- package/dist/collaboration/harness.js +150 -0
- package/dist/collaboration/index.d.ts +10 -0
- package/dist/collaboration/index.js +9 -0
- package/dist/collaboration/modes/creator-verifier.d.ts +69 -0
- package/dist/collaboration/modes/creator-verifier.js +124 -0
- package/dist/collaboration/pool.d.ts +46 -0
- package/dist/collaboration/pool.js +80 -0
- package/dist/harness/harness.js +7 -1
- package/dist/index.d.ts +12 -2
- package/dist/index.js +5 -0
- package/dist/kernel.d.ts +5 -2
- package/dist/memory/protocols.d.ts +8 -1
- package/dist/providers/anthropic.d.ts +3 -3
- package/dist/providers/anthropic.js +9 -9
- package/dist/providers/base.d.ts +7 -4
- package/dist/providers/base.js +41 -38
- package/dist/providers/deepseek.d.ts +2 -2
- package/dist/providers/deepseek.js +2 -2
- package/dist/providers/gemini.d.ts +3 -3
- package/dist/providers/gemini.js +8 -13
- package/dist/providers/ollama.d.ts +3 -3
- package/dist/providers/ollama.js +12 -8
- package/dist/providers/openai-chat.d.ts +2 -2
- package/dist/providers/openai-chat.js +6 -4
- package/dist/providers/openai-responses.d.ts +5 -5
- package/dist/providers/openai-responses.js +13 -19
- package/dist/providers/openai.d.ts +3 -3
- package/dist/providers/openai.js +4 -4
- package/dist/providers/qwen.d.ts +3 -3
- package/dist/providers/qwen.js +4 -4
- package/dist/types.d.ts +12 -2
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -50,6 +50,15 @@ const result = await agent.run("What is 17 + 28?")
|
|
|
50
50
|
console.log(result)
|
|
51
51
|
```
|
|
52
52
|
|
|
53
|
+
Same-session conversation continuity is explicit via `sessionId`:
|
|
54
|
+
|
|
55
|
+
```typescript
|
|
56
|
+
await agent.run("My name is Ada.", undefined, undefined, "chat-1")
|
|
57
|
+
const reply = await agent.run("What is my name?", undefined, undefined, "chat-1")
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
By default, the agent keeps session transcripts in memory for the lifetime of that `Agent` instance. Provide a `sessionStore` when the transcript must survive process restarts or be shared across workers.
|
|
61
|
+
|
|
53
62
|
Streaming:
|
|
54
63
|
|
|
55
64
|
```typescript
|
|
@@ -196,6 +205,22 @@ const agent = new Agent(provider, {
|
|
|
196
205
|
const result = await agent.dream("my-agent", Date.now())
|
|
197
206
|
```
|
|
198
207
|
|
|
208
|
+
### SessionStore (same-session transcript continuity)
|
|
209
|
+
|
|
210
|
+
```typescript
|
|
211
|
+
import type { SessionStore } from "@deepstrike/sdk"
|
|
212
|
+
|
|
213
|
+
class MySessionStore implements SessionStore {
|
|
214
|
+
async loadSession(sessionId) { return db.sessions.get(sessionId) }
|
|
215
|
+
async saveSession(session) { await db.sessions.put(session.sessionId, session) }
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
const agent = new Agent(provider, {
|
|
219
|
+
maxTokens: 4096,
|
|
220
|
+
sessionStore: new MySessionStore(),
|
|
221
|
+
})
|
|
222
|
+
```
|
|
223
|
+
|
|
199
224
|
---
|
|
200
225
|
|
|
201
226
|
## Governance
|
package/dist/agent.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { LLMProvider, StreamEvent } from "./types.js";
|
|
2
2
|
import type { RegisteredTool } from "./tools/index.js";
|
|
3
|
-
import type { DreamStore, DreamResult } from "./memory/protocols.js";
|
|
3
|
+
import type { DreamStore, DreamResult, SessionStore } from "./memory/protocols.js";
|
|
4
4
|
import type { KnowledgeSource } from "./knowledge/source.js";
|
|
5
5
|
import type { SignalSource } from "./signals/types.js";
|
|
6
6
|
export interface AgentOptions {
|
|
@@ -13,12 +13,6 @@ export interface AgentOptions {
|
|
|
13
13
|
* Passed to the kernel's `system` partition before the first LLM call.
|
|
14
14
|
*/
|
|
15
15
|
systemPrompt?: string;
|
|
16
|
-
/**
|
|
17
|
-
* Long-term memory snippets pre-seeded into the context before the first LLM call.
|
|
18
|
-
* Each string is pushed to the kernel's `memory` partition (highest-priority context
|
|
19
|
-
* after system). Use to inject memories retrieved from a DreamStore before a run.
|
|
20
|
-
*/
|
|
21
|
-
initialMemory?: string[];
|
|
22
16
|
/**
|
|
23
17
|
* Directory containing skill `.md` files. The kernel auto-injects a `skill`
|
|
24
18
|
* meta-tool so the model can load any skill by name on demand.
|
|
@@ -28,6 +22,8 @@ export interface AgentOptions {
|
|
|
28
22
|
signalSource?: SignalSource;
|
|
29
23
|
/** Backing store for the idle dreaming pipeline. Required to call `Agent.dream()`. */
|
|
30
24
|
dreamStore?: DreamStore;
|
|
25
|
+
/** Optional durable transcript store for same-session conversational continuity. */
|
|
26
|
+
sessionStore?: SessionStore;
|
|
31
27
|
/**
|
|
32
28
|
* Stable identifier for this agent. Required to enable in-session memory retrieval
|
|
33
29
|
* when `dreamStore` is configured.
|
|
@@ -56,6 +52,7 @@ export declare class Agent {
|
|
|
56
52
|
private knowledgeSource?;
|
|
57
53
|
private signalSource?;
|
|
58
54
|
private dreamStore?;
|
|
55
|
+
private readonly inMemorySessions;
|
|
59
56
|
private interrupted;
|
|
60
57
|
private pendingInterrupt;
|
|
61
58
|
private _turn;
|
|
@@ -72,8 +69,12 @@ export declare class Agent {
|
|
|
72
69
|
* Collect the full text response and return it.
|
|
73
70
|
* For richer control (streaming, tool events, token counts) use `runStreaming`.
|
|
74
71
|
*/
|
|
75
|
-
run(goal: string, criteria?: string[], extensions?: Record<string, unknown
|
|
76
|
-
runStreaming(goal: string, criteria?: string[], extensions?: Record<string, unknown
|
|
72
|
+
run(goal: string, criteria?: string[], extensions?: Record<string, unknown>, sessionId?: string): Promise<string>;
|
|
73
|
+
runStreaming(goal: string, criteria?: string[], extensions?: Record<string, unknown>, sessionId?: string): AsyncIterable<StreamEvent>;
|
|
74
|
+
private loadSession;
|
|
75
|
+
private saveSession;
|
|
76
|
+
private toMessage;
|
|
77
|
+
private kernelMsgToSessionMsg;
|
|
77
78
|
/**
|
|
78
79
|
* Trigger the idle dreaming cycle for this agent.
|
|
79
80
|
* Requires `dreamStore` and `agentId` to be configured.
|
package/dist/agent.js
CHANGED
|
@@ -10,6 +10,7 @@ export class Agent {
|
|
|
10
10
|
knowledgeSource;
|
|
11
11
|
signalSource;
|
|
12
12
|
dreamStore;
|
|
13
|
+
inMemorySessions = new Map();
|
|
13
14
|
interrupted = false;
|
|
14
15
|
pendingInterrupt = false;
|
|
15
16
|
// Live telemetry — updated each runStreaming call
|
|
@@ -44,15 +45,15 @@ export class Agent {
|
|
|
44
45
|
* Collect the full text response and return it.
|
|
45
46
|
* For richer control (streaming, tool events, token counts) use `runStreaming`.
|
|
46
47
|
*/
|
|
47
|
-
async run(goal, criteria, extensions) {
|
|
48
|
+
async run(goal, criteria, extensions, sessionId) {
|
|
48
49
|
let content = "";
|
|
49
|
-
for await (const evt of this.runStreaming(goal, criteria, extensions)) {
|
|
50
|
+
for await (const evt of this.runStreaming(goal, criteria, extensions, sessionId)) {
|
|
50
51
|
if (evt.type === "text_delta")
|
|
51
52
|
content += evt.delta;
|
|
52
53
|
}
|
|
53
54
|
return content;
|
|
54
55
|
}
|
|
55
|
-
async *runStreaming(goal, criteria, extensions) {
|
|
56
|
+
async *runStreaming(goal, criteria, extensions, sessionId) {
|
|
56
57
|
this.interrupted = false;
|
|
57
58
|
this.pendingInterrupt = false;
|
|
58
59
|
this._turn = 0;
|
|
@@ -76,8 +77,10 @@ export class Agent {
|
|
|
76
77
|
const tokens = Math.max(1, Math.ceil(this.options.systemPrompt.length / 4));
|
|
77
78
|
sm.addSystemMessage(this.options.systemPrompt, tokens);
|
|
78
79
|
}
|
|
79
|
-
|
|
80
|
-
|
|
80
|
+
const previousSession = sessionId ? await this.loadSession(sessionId) : undefined;
|
|
81
|
+
const previousMsgs = previousSession?.messages ?? [];
|
|
82
|
+
if (previousMsgs.length > 0) {
|
|
83
|
+
sm.preloadHistory(previousMsgs.map(m => this.toMessage(m)));
|
|
81
84
|
}
|
|
82
85
|
if (this.skillDir) {
|
|
83
86
|
const skillMetas = await scanSkillDir(this.skillDir);
|
|
@@ -97,7 +100,6 @@ export class Agent {
|
|
|
97
100
|
}
|
|
98
101
|
let action = sm.start({ goal, criteria: criteria ?? [] });
|
|
99
102
|
const sessionStart = Date.now();
|
|
100
|
-
const sessionMsgs = [{ role: "user", content: goal }];
|
|
101
103
|
while (!sm.isTerminal()) {
|
|
102
104
|
// Update telemetry
|
|
103
105
|
this._turn = sm.turn;
|
|
@@ -153,11 +155,11 @@ export class Agent {
|
|
|
153
155
|
if (action.kind === "call_llm") {
|
|
154
156
|
const finalToolCalls = [];
|
|
155
157
|
let finalText = "";
|
|
156
|
-
const
|
|
158
|
+
const context = action.context;
|
|
157
159
|
const tools = (action.tools ?? []);
|
|
158
160
|
let turnTokens = 0;
|
|
159
161
|
try {
|
|
160
|
-
for await (const evt of this.provider.stream(
|
|
162
|
+
for await (const evt of this.provider.stream(context, tools, Object.keys(ext).length ? ext : undefined, providerState)) {
|
|
161
163
|
if (evt.type === "usage") {
|
|
162
164
|
turnTokens = evt.totalTokens;
|
|
163
165
|
continue;
|
|
@@ -177,7 +179,6 @@ export class Agent {
|
|
|
177
179
|
break;
|
|
178
180
|
}
|
|
179
181
|
action = sm.feedLlmResponse({ role: "assistant", content: finalText, toolCalls: finalToolCalls, tokenCount: turnTokens || undefined });
|
|
180
|
-
sessionMsgs.push({ role: "assistant", content: finalText, toolCalls: finalToolCalls });
|
|
181
182
|
}
|
|
182
183
|
else if (action.kind === "execute_tools") {
|
|
183
184
|
const allCalls = action.calls ?? [];
|
|
@@ -266,12 +267,13 @@ export class Agent {
|
|
|
266
267
|
this._pressure = sm.pressure();
|
|
267
268
|
const status = result?.termination ?? "error";
|
|
268
269
|
const iterations = result ? Math.max(1, result.turnsUsed) : 0;
|
|
269
|
-
|
|
270
|
+
const newMsgs = sm.drainNewMessages().map(m => this.kernelMsgToSessionMsg(m));
|
|
271
|
+
if (this.options.dreamStore && this.options.agentId && newMsgs.length > 0) {
|
|
270
272
|
try {
|
|
271
273
|
await this.options.dreamStore.saveSession({
|
|
272
274
|
sessionId: crypto.randomUUID(),
|
|
273
275
|
agentId: this.options.agentId,
|
|
274
|
-
messages:
|
|
276
|
+
messages: newMsgs,
|
|
275
277
|
metadata: null,
|
|
276
278
|
createdAtMs: sessionStart,
|
|
277
279
|
updatedAtMs: Date.now(),
|
|
@@ -279,6 +281,17 @@ export class Agent {
|
|
|
279
281
|
}
|
|
280
282
|
catch { /* session save failure must not surface to caller */ }
|
|
281
283
|
}
|
|
284
|
+
if (sessionId) {
|
|
285
|
+
const now = Date.now();
|
|
286
|
+
await this.saveSession({
|
|
287
|
+
sessionId,
|
|
288
|
+
agentId: this.options.agentId ?? "default",
|
|
289
|
+
messages: [...previousMsgs, ...newMsgs],
|
|
290
|
+
metadata: previousSession?.metadata ?? null,
|
|
291
|
+
createdAtMs: previousSession?.createdAtMs ?? sessionStart,
|
|
292
|
+
updatedAtMs: now,
|
|
293
|
+
});
|
|
294
|
+
}
|
|
282
295
|
yield {
|
|
283
296
|
type: "done",
|
|
284
297
|
iterations,
|
|
@@ -286,6 +299,36 @@ export class Agent {
|
|
|
286
299
|
status,
|
|
287
300
|
};
|
|
288
301
|
}
|
|
302
|
+
async loadSession(sessionId) {
|
|
303
|
+
return this.options.sessionStore
|
|
304
|
+
? this.options.sessionStore.loadSession(sessionId)
|
|
305
|
+
: this.inMemorySessions.get(sessionId);
|
|
306
|
+
}
|
|
307
|
+
async saveSession(data) {
|
|
308
|
+
if (this.options.sessionStore) {
|
|
309
|
+
await this.options.sessionStore.saveSession(data);
|
|
310
|
+
return;
|
|
311
|
+
}
|
|
312
|
+
this.inMemorySessions.set(data.sessionId, data);
|
|
313
|
+
}
|
|
314
|
+
toMessage(message) {
|
|
315
|
+
return {
|
|
316
|
+
role: message.role,
|
|
317
|
+
content: message.content,
|
|
318
|
+
contentParts: message.contentParts,
|
|
319
|
+
tokenCount: message.tokenCount,
|
|
320
|
+
toolCalls: message.toolCalls ?? [],
|
|
321
|
+
};
|
|
322
|
+
}
|
|
323
|
+
kernelMsgToSessionMsg(msg) {
|
|
324
|
+
return {
|
|
325
|
+
role: msg.role,
|
|
326
|
+
content: msg.content,
|
|
327
|
+
contentParts: msg.contentParts,
|
|
328
|
+
tokenCount: msg.tokenCount,
|
|
329
|
+
toolCalls: msg.toolCalls?.length ? msg.toolCalls : undefined,
|
|
330
|
+
};
|
|
331
|
+
}
|
|
289
332
|
/**
|
|
290
333
|
* Trigger the idle dreaming cycle for this agent.
|
|
291
334
|
* Requires `dreamStore` and `agentId` to be configured.
|
|
@@ -332,7 +375,16 @@ export class Agent {
|
|
|
332
375
|
}
|
|
333
376
|
let synthesisText = "";
|
|
334
377
|
const providerState = this.provider.createRunState?.();
|
|
335
|
-
|
|
378
|
+
// IdlePipeline produces raw messages for synthesis; wrap them in a RenderedContext.
|
|
379
|
+
// The first system message (if any) becomes systemText; the rest are turns.
|
|
380
|
+
const synthMsgs = (action1.messages ?? []);
|
|
381
|
+
const synthSystemMsgs = synthMsgs.filter(m => m.role === "system");
|
|
382
|
+
const synthTurns = synthMsgs.filter(m => m.role !== "system");
|
|
383
|
+
const synthContext = {
|
|
384
|
+
systemText: synthSystemMsgs.map(m => m.content).join("\n\n"),
|
|
385
|
+
turns: synthTurns,
|
|
386
|
+
};
|
|
387
|
+
for await (const evt of this.provider.stream(synthContext, [], undefined, providerState)) {
|
|
336
388
|
if (evt.type === "text_delta")
|
|
337
389
|
synthesisText += evt.delta;
|
|
338
390
|
}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* VerificationContract — first-class type for contract-driven development.
|
|
3
|
+
*
|
|
4
|
+
* Contracts travel in the executor's system partition (never compressed) and
|
|
5
|
+
* are given to the verifier alongside the artifact. The verifier never sees
|
|
6
|
+
* the executor's implementation history — only the goal, contract, and artifact.
|
|
7
|
+
*/
|
|
8
|
+
export interface AcceptanceCriterion {
|
|
9
|
+
/** Stable id — matched in ContractCheckResult. */
|
|
10
|
+
id: string;
|
|
11
|
+
/** Human-readable statement of what correct looks like. */
|
|
12
|
+
text: string;
|
|
13
|
+
/** If true, failure here fails the entire contract regardless of score. */
|
|
14
|
+
required: boolean;
|
|
15
|
+
/** Contribution to the weighted overall score [0.0–1.0]. */
|
|
16
|
+
weight: number;
|
|
17
|
+
/** If true the SDK can verify this deterministically (e.g. word count, schema). */
|
|
18
|
+
machineCheckable: boolean;
|
|
19
|
+
}
|
|
20
|
+
export interface VerificationContract {
|
|
21
|
+
/** Stable id — doubles as the skill name on successful extraction. */
|
|
22
|
+
id: string;
|
|
23
|
+
/** Goal this contract governs. Injected into the executor's context. */
|
|
24
|
+
goal: string;
|
|
25
|
+
acceptance: AcceptanceCriterion[];
|
|
26
|
+
/** Patterns the executor must avoid. Checked by the verifier. */
|
|
27
|
+
antiPatterns: string[];
|
|
28
|
+
/** Artifacts that must be present before the verifier runs. */
|
|
29
|
+
evidenceRequired: string[];
|
|
30
|
+
}
|
|
31
|
+
export interface ContractCheckResult {
|
|
32
|
+
criterionId: string;
|
|
33
|
+
passed: boolean;
|
|
34
|
+
evidence?: string;
|
|
35
|
+
}
|
|
36
|
+
/** Build a contract incrementally. */
|
|
37
|
+
export declare class ContractBuilder {
|
|
38
|
+
private contract;
|
|
39
|
+
constructor(id: string, goal: string);
|
|
40
|
+
criterion(id: string, text: string, opts?: {
|
|
41
|
+
required?: boolean;
|
|
42
|
+
weight?: number;
|
|
43
|
+
machineCheckable?: boolean;
|
|
44
|
+
}): this;
|
|
45
|
+
antiPattern(pattern: string): this;
|
|
46
|
+
evidence(item: string): this;
|
|
47
|
+
build(): VerificationContract;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Render the contract as a markdown block for injection into a system prompt.
|
|
51
|
+
* Used by ContractDrivenHarness to inject into the executor's system partition.
|
|
52
|
+
*/
|
|
53
|
+
export declare function formatContractForSystemPrompt(contract: VerificationContract): string;
|
|
54
|
+
/** Derive a flat string[] of criterion texts for the existing HarnessLoop criteria API. */
|
|
55
|
+
export declare function contractToCriteriaStrings(contract: VerificationContract): string[];
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* VerificationContract — first-class type for contract-driven development.
|
|
3
|
+
*
|
|
4
|
+
* Contracts travel in the executor's system partition (never compressed) and
|
|
5
|
+
* are given to the verifier alongside the artifact. The verifier never sees
|
|
6
|
+
* the executor's implementation history — only the goal, contract, and artifact.
|
|
7
|
+
*/
|
|
8
|
+
/** Build a contract incrementally. */
|
|
9
|
+
export class ContractBuilder {
|
|
10
|
+
contract;
|
|
11
|
+
constructor(id, goal) {
|
|
12
|
+
this.contract = { id, goal, acceptance: [], antiPatterns: [], evidenceRequired: [] };
|
|
13
|
+
}
|
|
14
|
+
criterion(id, text, opts = {}) {
|
|
15
|
+
this.contract.acceptance.push({
|
|
16
|
+
id,
|
|
17
|
+
text,
|
|
18
|
+
required: opts.required ?? true,
|
|
19
|
+
weight: Math.min(1, Math.max(0, opts.weight ?? 1.0)),
|
|
20
|
+
machineCheckable: opts.machineCheckable ?? false,
|
|
21
|
+
});
|
|
22
|
+
return this;
|
|
23
|
+
}
|
|
24
|
+
antiPattern(pattern) {
|
|
25
|
+
this.contract.antiPatterns.push(pattern);
|
|
26
|
+
return this;
|
|
27
|
+
}
|
|
28
|
+
evidence(item) {
|
|
29
|
+
this.contract.evidenceRequired.push(item);
|
|
30
|
+
return this;
|
|
31
|
+
}
|
|
32
|
+
build() {
|
|
33
|
+
return structuredClone(this.contract);
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Render the contract as a markdown block for injection into a system prompt.
|
|
38
|
+
* Used by ContractDrivenHarness to inject into the executor's system partition.
|
|
39
|
+
*/
|
|
40
|
+
export function formatContractForSystemPrompt(contract) {
|
|
41
|
+
const lines = [
|
|
42
|
+
`## Verification Contract: ${contract.id}`,
|
|
43
|
+
"",
|
|
44
|
+
`Goal: ${contract.goal}`,
|
|
45
|
+
"",
|
|
46
|
+
"### Acceptance Criteria",
|
|
47
|
+
];
|
|
48
|
+
contract.acceptance.forEach((c, i) => {
|
|
49
|
+
const req = c.required ? "[REQUIRED]" : "[OPTIONAL]";
|
|
50
|
+
lines.push(`${i + 1}. ${req} ${c.text} (id: \`${c.id}\`, weight: ${c.weight.toFixed(1)})`);
|
|
51
|
+
});
|
|
52
|
+
if (contract.antiPatterns.length > 0) {
|
|
53
|
+
lines.push("", "### Anti-Patterns (must avoid)");
|
|
54
|
+
contract.antiPatterns.forEach(p => lines.push(`- ${p}`));
|
|
55
|
+
}
|
|
56
|
+
if (contract.evidenceRequired.length > 0) {
|
|
57
|
+
lines.push("", "### Required Evidence");
|
|
58
|
+
contract.evidenceRequired.forEach(e => lines.push(`- ${e}`));
|
|
59
|
+
}
|
|
60
|
+
return lines.join("\n");
|
|
61
|
+
}
|
|
62
|
+
/** Derive a flat string[] of criterion texts for the existing HarnessLoop criteria API. */
|
|
63
|
+
export function contractToCriteriaStrings(contract) {
|
|
64
|
+
return contract.acceptance.map(c => c.text);
|
|
65
|
+
}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import type { ContractCheckResult, VerificationContract } from "./contract.js";
|
|
2
|
+
import type { DreamResult } from "../memory/protocols.js";
|
|
3
|
+
/**
|
|
4
|
+
* HandoffArtifact — the single exchange token between sprints and agent instances.
|
|
5
|
+
*
|
|
6
|
+
* All handoff paths converge here:
|
|
7
|
+
* - ContractDrivenHarness completion → HandoffBus.fromContractOutcome()
|
|
8
|
+
* - Sub-agent completion → HandoffBus.fromSubAgentResult()
|
|
9
|
+
* - Dream consolidation → HandoffBus.fromDream()
|
|
10
|
+
* - Context renewal (kernel) → carried in kernel's HandoffArtifact type
|
|
11
|
+
*
|
|
12
|
+
* The invariant: a HandoffArtifact tells the next agent not only *what was done*
|
|
13
|
+
* but *what has been proven*. The contract_status field is never discarded on renewal.
|
|
14
|
+
*/
|
|
15
|
+
export interface HandoffArtifact {
|
|
16
|
+
goal: string;
|
|
17
|
+
sprint: number;
|
|
18
|
+
progressSummary: string;
|
|
19
|
+
openTasks: string[];
|
|
20
|
+
/** Per-criterion verification results from the most recent contract run. */
|
|
21
|
+
contractStatus: ContractCheckResult[];
|
|
22
|
+
/** Ratio of verification failures over 24 h (failed / total). 0.0 if no data. */
|
|
23
|
+
driftRate24h: number;
|
|
24
|
+
/** Issues blocking completion — require human or orchestrator attention. */
|
|
25
|
+
blockedOn: string[];
|
|
26
|
+
}
|
|
27
|
+
/** Input to HandoffBus.fromContractOutcome */
|
|
28
|
+
export interface ContractOutcomeInput {
|
|
29
|
+
contract: VerificationContract;
|
|
30
|
+
checkResults: ContractCheckResult[];
|
|
31
|
+
artifact: string;
|
|
32
|
+
success: boolean;
|
|
33
|
+
blockedOn?: string[];
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* HandoffBus — the canonical factory for HandoffArtifact.
|
|
37
|
+
*
|
|
38
|
+
* Every transition between agent contexts goes through one of these static methods.
|
|
39
|
+
* This ensures that the resulting artifact always carries contract_status, never just
|
|
40
|
+
* a prose summary.
|
|
41
|
+
*/
|
|
42
|
+
export declare class HandoffBus {
|
|
43
|
+
/**
|
|
44
|
+
* Build a HandoffArtifact from a ContractDrivenHarness outcome.
|
|
45
|
+
* The artifact field is used as the progress summary.
|
|
46
|
+
*/
|
|
47
|
+
static fromContractOutcome(input: ContractOutcomeInput): HandoffArtifact;
|
|
48
|
+
/**
|
|
49
|
+
* Build a HandoffArtifact from a sub-agent's final message.
|
|
50
|
+
* Matches the convention established by `report_sub_agent` (only final message injected).
|
|
51
|
+
*/
|
|
52
|
+
static fromSubAgentResult(opts: {
|
|
53
|
+
goal: string;
|
|
54
|
+
finalMessage: string;
|
|
55
|
+
sprint?: number;
|
|
56
|
+
}): HandoffArtifact;
|
|
57
|
+
/**
|
|
58
|
+
* Build a HandoffArtifact from a dream consolidation result.
|
|
59
|
+
* Used when the idle pipeline produces new memories that should be
|
|
60
|
+
* carried into the next sprint's context.
|
|
61
|
+
*/
|
|
62
|
+
static fromDream(opts: {
|
|
63
|
+
goal: string;
|
|
64
|
+
dreamResult: DreamResult;
|
|
65
|
+
sprint?: number;
|
|
66
|
+
}): HandoffArtifact;
|
|
67
|
+
/**
|
|
68
|
+
* Render the artifact as a compact injection string for the next agent's
|
|
69
|
+
* working partition (not system — this is a handoff note, not a permanent rule).
|
|
70
|
+
*/
|
|
71
|
+
static toContextNote(artifact: HandoffArtifact): string;
|
|
72
|
+
/**
|
|
73
|
+
* True when drift rate exceeds threshold or required criteria are blocked.
|
|
74
|
+
* Use to decide whether to pause autonomous delegation and escalate.
|
|
75
|
+
*/
|
|
76
|
+
static requiresEscalation(artifact: HandoffArtifact, opts?: {
|
|
77
|
+
driftThreshold?: number;
|
|
78
|
+
}): boolean;
|
|
79
|
+
}
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* HandoffBus — the canonical factory for HandoffArtifact.
|
|
3
|
+
*
|
|
4
|
+
* Every transition between agent contexts goes through one of these static methods.
|
|
5
|
+
* This ensures that the resulting artifact always carries contract_status, never just
|
|
6
|
+
* a prose summary.
|
|
7
|
+
*/
|
|
8
|
+
export class HandoffBus {
|
|
9
|
+
/**
|
|
10
|
+
* Build a HandoffArtifact from a ContractDrivenHarness outcome.
|
|
11
|
+
* The artifact field is used as the progress summary.
|
|
12
|
+
*/
|
|
13
|
+
static fromContractOutcome(input) {
|
|
14
|
+
const failedRequired = input.checkResults.filter(r => {
|
|
15
|
+
if (r.passed)
|
|
16
|
+
return false;
|
|
17
|
+
const c = input.contract.acceptance.find(a => a.id === r.criterionId);
|
|
18
|
+
return c?.required ?? false;
|
|
19
|
+
});
|
|
20
|
+
return {
|
|
21
|
+
goal: input.contract.goal,
|
|
22
|
+
sprint: 1,
|
|
23
|
+
progressSummary: input.success
|
|
24
|
+
? `Completed: ${input.artifact.slice(0, 200)}${input.artifact.length > 200 ? "…" : ""}`
|
|
25
|
+
: `Incomplete after max attempts. ${failedRequired.length} required criteria failed.`,
|
|
26
|
+
openTasks: input.success ? [] : failedRequired.map(r => `Fix criterion: ${r.criterionId}`),
|
|
27
|
+
contractStatus: input.checkResults,
|
|
28
|
+
driftRate24h: input.checkResults.length > 0
|
|
29
|
+
? input.checkResults.filter(r => !r.passed).length / input.checkResults.length
|
|
30
|
+
: 0,
|
|
31
|
+
blockedOn: input.blockedOn ?? [],
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* Build a HandoffArtifact from a sub-agent's final message.
|
|
36
|
+
* Matches the convention established by `report_sub_agent` (only final message injected).
|
|
37
|
+
*/
|
|
38
|
+
static fromSubAgentResult(opts) {
|
|
39
|
+
return {
|
|
40
|
+
goal: opts.goal,
|
|
41
|
+
sprint: opts.sprint ?? 1,
|
|
42
|
+
progressSummary: opts.finalMessage.slice(0, 500),
|
|
43
|
+
openTasks: [],
|
|
44
|
+
contractStatus: [],
|
|
45
|
+
driftRate24h: 0,
|
|
46
|
+
blockedOn: [],
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Build a HandoffArtifact from a dream consolidation result.
|
|
51
|
+
* Used when the idle pipeline produces new memories that should be
|
|
52
|
+
* carried into the next sprint's context.
|
|
53
|
+
*/
|
|
54
|
+
static fromDream(opts) {
|
|
55
|
+
return {
|
|
56
|
+
goal: opts.goal,
|
|
57
|
+
sprint: opts.sprint ?? 1,
|
|
58
|
+
progressSummary: `Memory consolidated: ${opts.dreamResult.entriesAdded} added, ${opts.dreamResult.entriesRemoved} removed.`,
|
|
59
|
+
openTasks: [],
|
|
60
|
+
contractStatus: [],
|
|
61
|
+
driftRate24h: 0,
|
|
62
|
+
blockedOn: [],
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* Render the artifact as a compact injection string for the next agent's
|
|
67
|
+
* working partition (not system — this is a handoff note, not a permanent rule).
|
|
68
|
+
*/
|
|
69
|
+
static toContextNote(artifact) {
|
|
70
|
+
const lines = [
|
|
71
|
+
`[Handoff from sprint ${artifact.sprint}]`,
|
|
72
|
+
`Goal: ${artifact.goal}`,
|
|
73
|
+
`Progress: ${artifact.progressSummary}`,
|
|
74
|
+
];
|
|
75
|
+
if (artifact.openTasks.length > 0) {
|
|
76
|
+
lines.push(`Open tasks: ${artifact.openTasks.join("; ")}`);
|
|
77
|
+
}
|
|
78
|
+
if (artifact.contractStatus.length > 0) {
|
|
79
|
+
const passed = artifact.contractStatus.filter(r => r.passed).length;
|
|
80
|
+
lines.push(`Contract: ${passed}/${artifact.contractStatus.length} criteria passed`);
|
|
81
|
+
}
|
|
82
|
+
if (artifact.blockedOn.length > 0) {
|
|
83
|
+
lines.push(`BLOCKED ON: ${artifact.blockedOn.join("; ")}`);
|
|
84
|
+
}
|
|
85
|
+
if (artifact.driftRate24h > 0) {
|
|
86
|
+
lines.push(`Drift rate: ${(artifact.driftRate24h * 100).toFixed(1)}%`);
|
|
87
|
+
}
|
|
88
|
+
return lines.join("\n");
|
|
89
|
+
}
|
|
90
|
+
/**
|
|
91
|
+
* True when drift rate exceeds threshold or required criteria are blocked.
|
|
92
|
+
* Use to decide whether to pause autonomous delegation and escalate.
|
|
93
|
+
*/
|
|
94
|
+
static requiresEscalation(artifact, opts = {}) {
|
|
95
|
+
const threshold = opts.driftThreshold ?? 0.05;
|
|
96
|
+
return artifact.driftRate24h > threshold || artifact.blockedOn.length > 0;
|
|
97
|
+
}
|
|
98
|
+
}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
import type { AgentPool } from "./pool.js";
|
|
2
|
+
import type { VerificationContract, ContractCheckResult } from "./contract.js";
|
|
3
|
+
import type { HandoffArtifact } from "./handoff.js";
|
|
4
|
+
export interface Violation {
|
|
5
|
+
criterionId: string;
|
|
6
|
+
text: string;
|
|
7
|
+
detail: string;
|
|
8
|
+
}
|
|
9
|
+
export interface ContractOutcome {
|
|
10
|
+
success: boolean;
|
|
11
|
+
artifact: string;
|
|
12
|
+
checkResults: ContractCheckResult[];
|
|
13
|
+
attemptsUsed: number;
|
|
14
|
+
totalTokensConsumed: number;
|
|
15
|
+
handoff: HandoffArtifact;
|
|
16
|
+
}
|
|
17
|
+
export interface ContractHarnessOptions {
|
|
18
|
+
maxAttempts?: number;
|
|
19
|
+
onViolation?: (violations: Violation[]) => void;
|
|
20
|
+
}
|
|
21
|
+
/**
|
|
22
|
+
* ContractDrivenHarness — the core multi-agent execution primitive.
|
|
23
|
+
*
|
|
24
|
+
* Differs from HarnessLoop in three ways:
|
|
25
|
+
* 1. Executor and verifier are **separate Agent instances** — no shared history.
|
|
26
|
+
* 2. Verifier receives only the artifact + contract, not the implementation transcript.
|
|
27
|
+
* 3. Feedback returned to the executor is a structured list of Violations,
|
|
28
|
+
* not a free-text LLM summary.
|
|
29
|
+
*
|
|
30
|
+
* Protocol per attempt:
|
|
31
|
+
* executor.run(goal, contract) → artifact
|
|
32
|
+
* verifier.runIsolated(artifact, contract) → audit text
|
|
33
|
+
* parse audit text → ContractCheckResult[]
|
|
34
|
+
* all required criteria pass → Done
|
|
35
|
+
* violations remain → inject only violation list into next executor goal
|
|
36
|
+
* maxAttempts exceeded → produce HandoffArtifact with blocked_on
|
|
37
|
+
*/
|
|
38
|
+
export declare class ContractDrivenHarness {
|
|
39
|
+
private pool;
|
|
40
|
+
private contract;
|
|
41
|
+
private maxAttempts;
|
|
42
|
+
private onViolation?;
|
|
43
|
+
constructor(pool: AgentPool, contract: VerificationContract, options?: ContractHarnessOptions);
|
|
44
|
+
run(): Promise<ContractOutcome>;
|
|
45
|
+
private _findViolations;
|
|
46
|
+
private _formatViolationsForFeedback;
|
|
47
|
+
/**
|
|
48
|
+
* Parse the verifier's free-text audit into structured ContractCheckResult[].
|
|
49
|
+
*
|
|
50
|
+
* The verifier is prompted to produce a structured PASS/FAIL per criterion.
|
|
51
|
+
* This parser handles the common patterns; callers can subclass and override
|
|
52
|
+
* for stricter parsing.
|
|
53
|
+
*/
|
|
54
|
+
private _parseAuditText;
|
|
55
|
+
}
|