@llblab/pi-actors 0.39.0 → 0.40.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +10 -3
- package/BACKLOG.md +2 -11
- package/CHANGELOG.md +45 -0
- package/README.md +4 -4
- package/dist/index.js +6 -4
- package/dist/lib/async-runs.d.ts +4 -0
- package/dist/lib/async-runs.js +112 -17
- package/dist/lib/command-templates.d.ts +9 -0
- package/dist/lib/command-templates.js +92 -11
- package/dist/lib/config.js +0 -5
- package/dist/lib/execution.d.ts +31 -0
- package/dist/lib/execution.js +145 -12
- package/dist/lib/file-state.d.ts +1 -0
- package/dist/lib/file-state.js +91 -3
- package/dist/lib/observability.d.ts +1 -1
- package/dist/lib/observability.js +7 -4
- package/dist/lib/pi.d.ts +1 -1
- package/dist/lib/pi.js +2 -2
- package/dist/lib/prompts.d.ts +1 -2
- package/dist/lib/prompts.js +2 -3
- package/dist/lib/recipes-context.js +17 -9
- package/dist/lib/recipes-discovery.js +11 -5
- package/dist/lib/recipes-references.d.ts +1 -1
- package/dist/lib/recipes-references.js +4 -5
- package/dist/lib/recipes-usage.d.ts +2 -0
- package/dist/lib/recipes-usage.js +35 -21
- package/dist/lib/registry.d.ts +0 -2
- package/dist/lib/registry.js +33 -10
- package/dist/lib/runs-ownership.d.ts +7 -0
- package/dist/lib/runs-ownership.js +82 -0
- package/dist/lib/runs-process.d.ts +17 -2
- package/dist/lib/runs-process.js +99 -11
- package/dist/lib/runs-retention.d.ts +3 -0
- package/dist/lib/runs-retention.js +18 -3
- package/dist/lib/runs-start.d.ts +2 -2
- package/dist/lib/runs-start.js +51 -17
- package/dist/lib/runs-status.d.ts +1 -1
- package/dist/lib/runs-status.js +8 -6
- package/dist/lib/runtime.js +69 -13
- package/dist/lib/tools-inspect.d.ts +2 -0
- package/dist/lib/tools-inspect.js +39 -2
- package/dist/lib/tools-register.js +0 -1
- package/dist/lib/tools-spawn.js +3 -2
- package/dist/lib/tools.d.ts +1 -0
- package/dist/lib/tools.js +3 -0
- package/dist/pi-actors/index.js +1 -0
- package/dist/recipes/subagent-judge.json +2 -1
- package/dist/recipes/subagent-merge.json +2 -1
- package/dist/recipes/subagent-normalize.json +2 -1
- package/dist/recipes/subagent-review-coordinator.json +1 -1
- package/dist/recipes/subagent-review.json +2 -1
- package/dist/recipes/subagent-verify.json +2 -1
- package/dist/scripts/async-runner.mjs +274 -6
- package/dist/scripts/build-dist.mjs +14 -1
- package/dist/skills/actors/SKILL.md +11 -7
- package/dist/skills/swarm/SKILL.md +1 -1
- package/docs/actor-messages.md +1 -1
- package/docs/async-runs.md +14 -5
- package/docs/command-templates.md +4 -2
- package/docs/recipe-library.md +1 -0
- package/docs/template-recipes.md +5 -7
- package/docs/tool-registry.md +4 -2
- package/index.ts +18 -7
- package/lib/async-runs.ts +138 -19
- package/lib/command-templates.ts +132 -13
- package/lib/config.ts +0 -4
- package/lib/execution.ts +198 -13
- package/lib/file-state.ts +106 -3
- package/lib/observability.ts +11 -5
- package/lib/pi.ts +3 -3
- package/lib/prompts.ts +2 -4
- package/lib/recipes-context.ts +17 -9
- package/lib/recipes-discovery.ts +10 -5
- package/lib/recipes-references.ts +5 -6
- package/lib/recipes-usage.ts +36 -20
- package/lib/registry.ts +43 -13
- package/lib/runs-ownership.ts +117 -0
- package/lib/runs-process.ts +138 -16
- package/lib/runs-retention.ts +22 -2
- package/lib/runs-start.ts +89 -31
- package/lib/runs-status.ts +15 -6
- package/lib/runtime.ts +64 -12
- package/lib/tools-inspect.ts +46 -4
- package/lib/tools-register.ts +0 -3
- package/lib/tools-spawn.ts +5 -5
- package/lib/tools.ts +8 -0
- package/package.json +2 -2
- package/recipes/subagent-judge.json +2 -1
- package/recipes/subagent-merge.json +2 -1
- package/recipes/subagent-normalize.json +2 -1
- package/recipes/subagent-review-coordinator.json +1 -1
- package/recipes/subagent-review.json +2 -1
- package/recipes/subagent-verify.json +2 -1
- package/scripts/async-runner.mjs +274 -6
- package/scripts/build-dist.mjs +14 -1
- package/skills/actors/SKILL.md +11 -7
- package/skills/swarm/SKILL.md +1 -1
|
@@ -8,8 +8,14 @@
|
|
|
8
8
|
* chasing a one-off lib entrypoint domain.
|
|
9
9
|
*/
|
|
10
10
|
|
|
11
|
-
import {
|
|
12
|
-
|
|
11
|
+
import {
|
|
12
|
+
appendFileSync,
|
|
13
|
+
existsSync,
|
|
14
|
+
mkdirSync,
|
|
15
|
+
readFileSync,
|
|
16
|
+
statSync,
|
|
17
|
+
} from "node:fs";
|
|
18
|
+
import { dirname, join, relative } from "node:path";
|
|
13
19
|
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
14
20
|
|
|
15
21
|
function packageRoot() {
|
|
@@ -30,7 +36,8 @@ const { appendRecipeContextToPiArgs, materializePiPrintPromptArg } =
|
|
|
30
36
|
const { buildReviewPreflightDiagnostic, formatReviewPreflightDiagnostic } =
|
|
31
37
|
await importRuntimeModule("preflight-diagnostics");
|
|
32
38
|
const { execCommandTemplate } = await importRuntimeModule("command-templates");
|
|
33
|
-
const { executeRegisteredTool } =
|
|
39
|
+
const { applyOutputAcceptancePolicy, executeRegisteredTool } =
|
|
40
|
+
await importRuntimeModule("execution");
|
|
34
41
|
const { writeJsonAtomic } = await importRuntimeModule("file-state");
|
|
35
42
|
|
|
36
43
|
function quoteCommandDetailPart(value) {
|
|
@@ -43,6 +50,17 @@ function formatCommandDetail(command, args) {
|
|
|
43
50
|
return [command, ...args].map(quoteCommandDetailPart).join(" ");
|
|
44
51
|
}
|
|
45
52
|
|
|
53
|
+
function captureDetails(result) {
|
|
54
|
+
return {
|
|
55
|
+
...(typeof result.stdoutBytes === "number" ? { stdout_bytes: result.stdoutBytes } : {}),
|
|
56
|
+
...(typeof result.stderrBytes === "number" ? { stderr_bytes: result.stderrBytes } : {}),
|
|
57
|
+
...(result.stdoutFile ? { stdout_file: result.stdoutFile } : {}),
|
|
58
|
+
...(result.stderrFile ? { stderr_file: result.stderrFile } : {}),
|
|
59
|
+
...(result.stdoutTruncated ? { stdout_truncated: true } : {}),
|
|
60
|
+
...(result.stderrTruncated ? { stderr_truncated: true } : {}),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
46
64
|
function summarizeCommandDetail(commandDetail) {
|
|
47
65
|
return commandDetail.length > 160
|
|
48
66
|
? `${commandDetail.slice(0, 157)}...`
|
|
@@ -59,6 +77,7 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
59
77
|
const outboxPath = join(stateDir, "outbox.jsonl");
|
|
60
78
|
const stdoutPath = join(stateDir, "stdout.log");
|
|
61
79
|
const stderrPath = join(stateDir, "stderr.log");
|
|
80
|
+
const evidencePath = join(stateDir, "review-evidence.json");
|
|
62
81
|
const meta = JSON.parse(readFileSync(runPath, "utf8"));
|
|
63
82
|
|
|
64
83
|
function event(name, data = {}) {
|
|
@@ -87,7 +106,171 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
87
106
|
let activeSubagents = 0;
|
|
88
107
|
let completedSubagents = 0;
|
|
89
108
|
let promptCounter = 0;
|
|
109
|
+
let captureCounter = 0;
|
|
90
110
|
const subagentFailures = [];
|
|
111
|
+
const evidenceRecords = [];
|
|
112
|
+
const stageOccurrences = new Map();
|
|
113
|
+
let reportEvidence;
|
|
114
|
+
|
|
115
|
+
function writeEvidenceManifest(status) {
|
|
116
|
+
writeJsonAtomic(evidencePath, {
|
|
117
|
+
version: 1,
|
|
118
|
+
run: meta.run,
|
|
119
|
+
status,
|
|
120
|
+
...(meta.model_policy ? { model_policy: meta.model_policy } : {}),
|
|
121
|
+
commands: [...evidenceRecords].sort((left, right) =>
|
|
122
|
+
left.id.localeCompare(right.id)
|
|
123
|
+
),
|
|
124
|
+
...(reportEvidence ? { report_evidence: reportEvidence } : {}),
|
|
125
|
+
updated_at: new Date().toISOString(),
|
|
126
|
+
});
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function commandEvidenceStartRecord({
|
|
130
|
+
commandDetail,
|
|
131
|
+
commandId,
|
|
132
|
+
materialized,
|
|
133
|
+
options,
|
|
134
|
+
stage,
|
|
135
|
+
occurrence,
|
|
136
|
+
}) {
|
|
137
|
+
const recipeContext = options?.actorRecipeContext;
|
|
138
|
+
return {
|
|
139
|
+
id: commandId,
|
|
140
|
+
stage,
|
|
141
|
+
occurrence,
|
|
142
|
+
status: "running",
|
|
143
|
+
started_at: new Date().toISOString(),
|
|
144
|
+
...(options?.evidenceContext?.label
|
|
145
|
+
? { label: options.evidenceContext.label }
|
|
146
|
+
: {}),
|
|
147
|
+
...(options?.evidenceContext?.repeatIndex !== undefined
|
|
148
|
+
? { branch_index: options.evidenceContext.repeatIndex }
|
|
149
|
+
: {}),
|
|
150
|
+
...(recipeContext ? { recipe_context: recipeContext } : {}),
|
|
151
|
+
command: commandDetail,
|
|
152
|
+
...(materialized.promptFile
|
|
153
|
+
? { prompt_file: relative(stateDir, materialized.promptFile) }
|
|
154
|
+
: {}),
|
|
155
|
+
...(materialized.promptBytes
|
|
156
|
+
? { prompt_bytes: materialized.promptBytes }
|
|
157
|
+
: {}),
|
|
158
|
+
attempts: [],
|
|
159
|
+
semantic_acceptance:
|
|
160
|
+
options?.evidenceContext?.acceptOutput === "review_evidence" ||
|
|
161
|
+
stage === "preflight"
|
|
162
|
+
? "pending"
|
|
163
|
+
: "not_required",
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function commandEvidenceRecord({
|
|
168
|
+
captureDir,
|
|
169
|
+
commandDetail,
|
|
170
|
+
commandId,
|
|
171
|
+
materialized,
|
|
172
|
+
options,
|
|
173
|
+
result,
|
|
174
|
+
rawExitCode,
|
|
175
|
+
stage,
|
|
176
|
+
occurrence,
|
|
177
|
+
startedAt,
|
|
178
|
+
}) {
|
|
179
|
+
const recipeContext = options?.actorRecipeContext;
|
|
180
|
+
const attempts = [];
|
|
181
|
+
const maxAttempts = Math.max(1, options?.retry || 1);
|
|
182
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
|
|
183
|
+
const attemptDir = join(
|
|
184
|
+
captureDir,
|
|
185
|
+
`attempt-${String(attempt).padStart(3, "0")}`,
|
|
186
|
+
);
|
|
187
|
+
const stdoutFile = join(attemptDir, "stdout.log");
|
|
188
|
+
const stderrFile = join(attemptDir, "stderr.log");
|
|
189
|
+
if (!existsSync(stdoutFile) && !existsSync(stderrFile)) continue;
|
|
190
|
+
attempts.push({
|
|
191
|
+
attempt,
|
|
192
|
+
stdout: {
|
|
193
|
+
path: relative(stateDir, stdoutFile),
|
|
194
|
+
bytes: existsSync(stdoutFile) ? statSync(stdoutFile).size : 0,
|
|
195
|
+
},
|
|
196
|
+
stderr: {
|
|
197
|
+
path: relative(stateDir, stderrFile),
|
|
198
|
+
bytes: existsSync(stderrFile) ? statSync(stderrFile).size : 0,
|
|
199
|
+
},
|
|
200
|
+
});
|
|
201
|
+
}
|
|
202
|
+
const expectedReviewMarker =
|
|
203
|
+
options?.evidenceContext?.acceptOutput === "review_evidence";
|
|
204
|
+
const expectedPreflightMarker = stage === "preflight";
|
|
205
|
+
const semanticStdout = result.stdoutTruncated && result.stdoutFile && existsSync(result.stdoutFile)
|
|
206
|
+
? readFileSync(result.stdoutFile, "utf8")
|
|
207
|
+
: result.stdout;
|
|
208
|
+
const firstNonWhitespaceLine = semanticStdout
|
|
209
|
+
.split(/\r?\n/)
|
|
210
|
+
.find((line) => line.trim().length > 0);
|
|
211
|
+
const markerAccepted = expectedReviewMarker
|
|
212
|
+
? firstNonWhitespaceLine?.trim() === "ACTOR_REVIEW_RESULT"
|
|
213
|
+
: expectedPreflightMarker
|
|
214
|
+
? result.stdout.trimStart().startsWith("ACTOR_PREFLIGHT_OK")
|
|
215
|
+
: undefined;
|
|
216
|
+
return {
|
|
217
|
+
id: commandId,
|
|
218
|
+
stage,
|
|
219
|
+
occurrence,
|
|
220
|
+
status: result.code === 0 ? "done" : "failed",
|
|
221
|
+
started_at: startedAt,
|
|
222
|
+
completed_at: new Date().toISOString(),
|
|
223
|
+
...(options?.evidenceContext?.label
|
|
224
|
+
? { label: options.evidenceContext.label }
|
|
225
|
+
: {}),
|
|
226
|
+
...(options?.evidenceContext?.repeatIndex !== undefined
|
|
227
|
+
? { branch_index: options.evidenceContext.repeatIndex }
|
|
228
|
+
: {}),
|
|
229
|
+
...(recipeContext ? { recipe_context: recipeContext } : {}),
|
|
230
|
+
command: commandDetail,
|
|
231
|
+
...(materialized.promptFile
|
|
232
|
+
? { prompt_file: relative(stateDir, materialized.promptFile) }
|
|
233
|
+
: {}),
|
|
234
|
+
...(materialized.promptBytes
|
|
235
|
+
? { prompt_bytes: materialized.promptBytes }
|
|
236
|
+
: {}),
|
|
237
|
+
attempts,
|
|
238
|
+
exit_code: rawExitCode ?? result.code,
|
|
239
|
+
effective_exit_code: result.code,
|
|
240
|
+
killed: result.killed,
|
|
241
|
+
stdout_bytes: result.stdoutBytes ?? Buffer.byteLength(result.stdout),
|
|
242
|
+
stderr_bytes: result.stderrBytes ?? Buffer.byteLength(result.stderr),
|
|
243
|
+
stdout_truncated: result.stdoutTruncated === true,
|
|
244
|
+
stderr_truncated: result.stderrTruncated === true,
|
|
245
|
+
semantic_acceptance:
|
|
246
|
+
markerAccepted === undefined
|
|
247
|
+
? "not_required"
|
|
248
|
+
: markerAccepted && result.code === 0
|
|
249
|
+
? "accepted"
|
|
250
|
+
: "rejected",
|
|
251
|
+
};
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
function auditReviewReport(text) {
|
|
255
|
+
const required = evidenceRecords
|
|
256
|
+
.filter((record) =>
|
|
257
|
+
["reviewer", "verifier", "merger", "judge"].includes(record.stage)
|
|
258
|
+
)
|
|
259
|
+
.map((record) => `review-evidence.json#${record.id}`);
|
|
260
|
+
if (required.length === 0) return undefined;
|
|
261
|
+
const cited = [...new Set(
|
|
262
|
+
text.match(/review-evidence\.json#command-\d{3}/g) || [],
|
|
263
|
+
)].sort();
|
|
264
|
+
const missing = required.filter((reference) => !cited.includes(reference));
|
|
265
|
+
const claimsComplete = /\bStatus\b[\s:*#_-]{0,40}\bcomplete\b/i.test(text);
|
|
266
|
+
return {
|
|
267
|
+
required,
|
|
268
|
+
cited,
|
|
269
|
+
missing,
|
|
270
|
+
claims_complete: claimsComplete,
|
|
271
|
+
complete_allowed: missing.length === 0,
|
|
272
|
+
};
|
|
273
|
+
}
|
|
91
274
|
|
|
92
275
|
function getCommandDoneDelivery(result) {
|
|
93
276
|
return result.code !== 0 || activeSubagents > 0 ? "followup" : "log";
|
|
@@ -131,15 +314,64 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
131
314
|
);
|
|
132
315
|
const execArgs = materialized.args;
|
|
133
316
|
const commandDetail = formatCommandDetail(command, execArgs);
|
|
317
|
+
captureCounter += 1;
|
|
318
|
+
const commandId = `command-${String(captureCounter).padStart(3, "0")}`;
|
|
319
|
+
const stage =
|
|
320
|
+
options?.actorRecipeContext?.alias ||
|
|
321
|
+
options?.actorRecipeContext?.name ||
|
|
322
|
+
"command";
|
|
323
|
+
const occurrence = (stageOccurrences.get(stage) || 0) + 1;
|
|
324
|
+
stageOccurrences.set(stage, occurrence);
|
|
325
|
+
if (
|
|
326
|
+
materialized.promptFile &&
|
|
327
|
+
["verifier", "merger", "judge", "normalizer"].includes(stage)
|
|
328
|
+
) {
|
|
329
|
+
const references = evidenceRecords
|
|
330
|
+
.filter((record) =>
|
|
331
|
+
["reviewer", "verifier", "merger", "judge"].includes(record.stage)
|
|
332
|
+
)
|
|
333
|
+
.map((record) => `ACTOR_EVIDENCE_REF: review-evidence.json#${record.id}`);
|
|
334
|
+
if (references.length > 0) {
|
|
335
|
+
appendFileSync(
|
|
336
|
+
materialized.promptFile,
|
|
337
|
+
`\n\nRetained actor evidence references available for citation:\n${references.join("\n")}\n`,
|
|
338
|
+
);
|
|
339
|
+
materialized.promptBytes = statSync(materialized.promptFile).size;
|
|
340
|
+
}
|
|
341
|
+
}
|
|
134
342
|
activeSubagents += 1;
|
|
343
|
+
const evidenceIndex = evidenceRecords.length;
|
|
344
|
+
const startedEvidence = commandEvidenceStartRecord({
|
|
345
|
+
commandDetail,
|
|
346
|
+
commandId,
|
|
347
|
+
materialized,
|
|
348
|
+
options,
|
|
349
|
+
stage,
|
|
350
|
+
occurrence,
|
|
351
|
+
});
|
|
352
|
+
evidenceRecords.push(startedEvidence);
|
|
353
|
+
writeEvidenceManifest("running");
|
|
135
354
|
event("command.start", {
|
|
136
355
|
activeSubagents,
|
|
356
|
+
command_id: commandId,
|
|
137
357
|
command: commandDetail,
|
|
138
358
|
...(materialized.promptFile ? { prompt_file: materialized.promptFile } : {}),
|
|
139
359
|
...(materialized.promptBytes ? { prompt_bytes: materialized.promptBytes } : {}),
|
|
140
360
|
});
|
|
141
361
|
progressRunning();
|
|
142
|
-
|
|
362
|
+
const captureDir = join(stateDir, "captures", commandId);
|
|
363
|
+
const rawResult = await execCommandTemplate(command, execArgs, {
|
|
364
|
+
...options,
|
|
365
|
+
captureDir,
|
|
366
|
+
});
|
|
367
|
+
let result = await applyOutputAcceptancePolicy(
|
|
368
|
+
rawResult,
|
|
369
|
+
options?.evidenceContext?.acceptOutput,
|
|
370
|
+
);
|
|
371
|
+
result = {
|
|
372
|
+
...result,
|
|
373
|
+
evidenceRef: `review-evidence.json#${commandId}`,
|
|
374
|
+
};
|
|
143
375
|
const preflightDiagnostic = result.code !== 0
|
|
144
376
|
? buildReviewPreflightDiagnostic({
|
|
145
377
|
args: execArgs,
|
|
@@ -160,6 +392,19 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
160
392
|
].filter(Boolean).join("\n"),
|
|
161
393
|
};
|
|
162
394
|
}
|
|
395
|
+
evidenceRecords[evidenceIndex] = commandEvidenceRecord({
|
|
396
|
+
captureDir,
|
|
397
|
+
commandDetail,
|
|
398
|
+
commandId,
|
|
399
|
+
materialized,
|
|
400
|
+
options,
|
|
401
|
+
result,
|
|
402
|
+
rawExitCode: rawResult.code,
|
|
403
|
+
stage,
|
|
404
|
+
occurrence,
|
|
405
|
+
startedAt: startedEvidence.started_at,
|
|
406
|
+
});
|
|
407
|
+
writeEvidenceManifest("running");
|
|
163
408
|
activeSubagents = Math.max(0, activeSubagents - 1);
|
|
164
409
|
completedSubagents += 1;
|
|
165
410
|
if (result.code !== 0) {
|
|
@@ -167,15 +412,18 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
167
412
|
code: result.code,
|
|
168
413
|
command: commandDetail,
|
|
169
414
|
killed: result.killed,
|
|
415
|
+
...captureDetails(result),
|
|
170
416
|
...(materialized.promptFile ? { prompt_file: materialized.promptFile } : {}),
|
|
171
417
|
...(preflightDiagnostic ? { preflight: preflightDiagnostic } : {}),
|
|
172
418
|
});
|
|
173
419
|
}
|
|
174
420
|
event("command.done", {
|
|
175
421
|
activeSubagents,
|
|
422
|
+
command_id: commandId,
|
|
176
423
|
code: result.code,
|
|
177
424
|
command: commandDetail,
|
|
178
425
|
killed: result.killed,
|
|
426
|
+
...captureDetails(result),
|
|
179
427
|
...(materialized.promptFile ? { prompt_file: materialized.promptFile } : {}),
|
|
180
428
|
...(materialized.promptBytes ? { prompt_bytes: materialized.promptBytes } : {}),
|
|
181
429
|
...(preflightDiagnostic ? { preflight: preflightDiagnostic } : {}),
|
|
@@ -186,10 +434,12 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
186
434
|
{
|
|
187
435
|
activeSubagents,
|
|
188
436
|
...(meta.artifacts ? { artifacts: meta.artifacts } : {}),
|
|
189
|
-
run_files: [stdoutPath, stderrPath, resultPath, eventsPath, outboxPath],
|
|
437
|
+
run_files: [stdoutPath, stderrPath, resultPath, eventsPath, outboxPath, evidencePath],
|
|
438
|
+
command_id: commandId,
|
|
190
439
|
code: result.code,
|
|
191
440
|
command: commandDetail,
|
|
192
441
|
killed: result.killed,
|
|
442
|
+
...captureDetails(result),
|
|
193
443
|
...(materialized.promptFile ? { prompt_file: materialized.promptFile } : {}),
|
|
194
444
|
...(materialized.promptBytes ? { prompt_bytes: materialized.promptBytes } : {}),
|
|
195
445
|
...(preflightDiagnostic ? { preflight: preflightDiagnostic } : {}),
|
|
@@ -218,6 +468,20 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
218
468
|
);
|
|
219
469
|
const text = result.content?.[0]?.text || "";
|
|
220
470
|
appendFileSync(stdoutPath, text);
|
|
471
|
+
reportEvidence = auditReviewReport(text);
|
|
472
|
+
if (
|
|
473
|
+
reportEvidence?.claims_complete &&
|
|
474
|
+
reportEvidence.complete_allowed !== true
|
|
475
|
+
) {
|
|
476
|
+
const error = new Error(
|
|
477
|
+
`review report evidence incomplete: missing ${reportEvidence.missing.join(", ")}`,
|
|
478
|
+
);
|
|
479
|
+
error.details = {
|
|
480
|
+
code: 65,
|
|
481
|
+
failureReason: "incomplete review report evidence",
|
|
482
|
+
};
|
|
483
|
+
throw error;
|
|
484
|
+
}
|
|
221
485
|
writeJsonAtomic(resultPath, {
|
|
222
486
|
code: result.details.code,
|
|
223
487
|
command: result.details.command,
|
|
@@ -226,6 +490,7 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
226
490
|
truncated: result.details.truncated,
|
|
227
491
|
completedAt: new Date().toISOString(),
|
|
228
492
|
});
|
|
493
|
+
writeEvidenceManifest("done");
|
|
229
494
|
progress("done", {
|
|
230
495
|
completed: 1,
|
|
231
496
|
failures: result.details.nonCriticalFailures || [],
|
|
@@ -245,11 +510,14 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
245
510
|
...(details?.softQuorum ? { soft_quorum: details.softQuorum } : {}),
|
|
246
511
|
completedAt: new Date().toISOString(),
|
|
247
512
|
});
|
|
513
|
+
writeEvidenceManifest("failed");
|
|
248
514
|
progress("failed", {
|
|
249
515
|
completed: 0,
|
|
250
516
|
failures: Array.isArray(details?.branches) && details.branches.length > 0
|
|
251
517
|
? details.branches
|
|
252
|
-
:
|
|
518
|
+
: subagentFailures.length > 0
|
|
519
|
+
? subagentFailures
|
|
520
|
+
: [{ message }],
|
|
253
521
|
...(details?.failureReason ? { failureReason: details.failureReason } : {}),
|
|
254
522
|
});
|
|
255
523
|
event("run.failed", {
|
|
@@ -9,7 +9,13 @@
|
|
|
9
9
|
*/
|
|
10
10
|
|
|
11
11
|
import { spawnSync } from "node:child_process";
|
|
12
|
-
import {
|
|
12
|
+
import {
|
|
13
|
+
cpSync,
|
|
14
|
+
mkdirSync,
|
|
15
|
+
readdirSync,
|
|
16
|
+
rmSync,
|
|
17
|
+
writeFileSync,
|
|
18
|
+
} from "node:fs";
|
|
13
19
|
import { join } from "node:path";
|
|
14
20
|
|
|
15
21
|
function run(command, args) {
|
|
@@ -22,6 +28,13 @@ mkdirSync("dist", { recursive: true });
|
|
|
22
28
|
|
|
23
29
|
run("tsc", ["-p", "tsconfig.build.json"]);
|
|
24
30
|
|
|
31
|
+
mkdirSync(join("dist", "pi-actors"), { recursive: true });
|
|
32
|
+
writeFileSync(
|
|
33
|
+
join("dist", "pi-actors", "index.js"),
|
|
34
|
+
'export { default } from "../index.js";\n',
|
|
35
|
+
"utf8",
|
|
36
|
+
);
|
|
37
|
+
|
|
25
38
|
for (const dir of ["scripts", "recipes", "fixtures", "skills"]) {
|
|
26
39
|
cpSync(dir, join("dist", dir), { recursive: true });
|
|
27
40
|
}
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
name: actors
|
|
3
3
|
description: Required practical guide for non-trivial pi-actors use. Read before using or changing spawn, message, inspect, actor runs, tools, recipes, command templates, async lifecycle, mailboxes, artifacts, and local orchestration mechanics.
|
|
4
4
|
metadata:
|
|
5
|
-
version: 0.
|
|
5
|
+
version: 0.40.0
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# Actors (pi-actors)
|
|
@@ -68,6 +68,7 @@ Rules:
|
|
|
68
68
|
- Use inline `template` for one-off experiments; promote useful repeats to recipes.
|
|
69
69
|
- When a successful actor follow-up suggests persistence, decide whether the pattern deserves durable tool memory; call `register_tool` yourself only when the evidence is strong, and ask before writing the user recipe root.
|
|
70
70
|
- Use stable `as` names when you will inspect or message the actor later.
|
|
71
|
+
- Public run state is runtime-owned; do not pass custom `state_dir` paths. This keeps `run:<id>` addressability and retention on one boundary.
|
|
71
72
|
- `async: true` on the recipe is the detached run switch.
|
|
72
73
|
|
|
73
74
|
### `message` — send a typed envelope
|
|
@@ -90,6 +91,7 @@ Envelope fields:
|
|
|
90
91
|
- Group posts to `room:<run>` require `from` from the same run (`run:<run>` or `branch:<run>/<branch>`).
|
|
91
92
|
- Runtime termination message: `control.kill` is the only documented actor message that kills a run. `control.stop` and `control.cancel` are actor-local mailbox vocabulary only when a recipe declares and handles them. Terminal retention messages: `control.archive`, `control.prune`.
|
|
92
93
|
- Long-lived child processes should remain in the run-owned process group unless the recipe implements an explicit daemon termination bridge; do not leave unowned detached services behind `control.kill`.
|
|
94
|
+
- Run controls revalidate a persisted cross-platform process identity proof before delivery or signaling. Treat `dead pid`, `owner mismatch`, and `unsupported proof` as distinct fail-closed states; never bypass them with direct pid signals.
|
|
93
95
|
|
|
94
96
|
Check `inspect view=mailbox` before domain-specific messages.
|
|
95
97
|
|
|
@@ -116,8 +118,8 @@ Views:
|
|
|
116
118
|
- Advanced `contacts`: roster-derived direct-message targets without full roster metadata.
|
|
117
119
|
- Advanced `previews`: TUI-ready bounded group-message previews with timestamp/from/to/type/summary/body_preview.
|
|
118
120
|
- `mailbox`: declared accepts/emits contract for runs; queued direct branch inbox messages for `branch:<run>/<branch>` with `id`, status, route/type, and queue/handling timestamps.
|
|
119
|
-
- `files`: run state
|
|
120
|
-
- `artifacts`: declared artifact paths/status.
|
|
121
|
+
- `files`: run state file summary plus a `lines`-bounded `review-evidence.json` manifest when present, including total/truncated command counts and stage capture paths.
|
|
122
|
+
- `artifacts`: declared artifact paths/status plus the same bounded owned review-evidence manifest when present.
|
|
121
123
|
- `recipes` target: registry summary for active, shadowed, invalid, disabled, and diagnostic recipe entries.
|
|
122
124
|
|
|
123
125
|
Actor inspector commands:
|
|
@@ -128,7 +130,7 @@ Actor inspector commands:
|
|
|
128
130
|
|
|
129
131
|
The table is compact and optimistic by default: bounded body previews, capped noisy room rows, branch-local inbox previews, stable event ids in selected-message details, and an inline roster summary in the form `name/role` that wraps only when needed. Use `unread` for queued branch inbox work and `branch <name>` / `current-branch <name>` for one branch's room/direct/inbox traffic. Rows with `metadata.requires_response=true` show a `!` attention marker. `/actors-inspect <number>` marks that row read for the current session filter. Active roster members use the target color; members that sent `actor.leave` stay visible as inactive/muted participants from the current run. Actor display names come from `actor.join` bodies (`display`) or branch addresses, keeping debugger output plain and name-driven.
|
|
130
132
|
|
|
131
|
-
Let terminal notifications arrive
|
|
133
|
+
Let terminal notifications arrive. When a deferred actor result gates the next step, wait for that terminal steering notification instead of scheduling continuation loops, repeatedly inspecting, or mutating the actor's reviewed scope. Busy coordinators receive it at the next safe tool boundary; idle coordinators start a normal turn. Inspect early only for an operator request, a meaningful actor event, or diagnosis of an overdue or stuck run.
|
|
132
134
|
|
|
133
135
|
## Runtime Communication Rules
|
|
134
136
|
|
|
@@ -154,11 +156,13 @@ Controls:
|
|
|
154
156
|
- `args`, `defaults`: public placeholder declarations and defaults.
|
|
155
157
|
- `parallel: true`: fanout child nodes.
|
|
156
158
|
- `when`: conditional execution.
|
|
159
|
+
- `accept_output: review_evidence`: fail closed unless the exact first non-whitespace stdout line is `ACTOR_REVIEW_RESULT`; marker prefixes are rejected and rejected stdout remains diagnostic evidence.
|
|
157
160
|
- `timeout`, `delay`, `retry`: timing and retry controls; string placeholders are allowed where supported.
|
|
158
161
|
- `failure`: `continue`, `branch`, or `root` propagation.
|
|
159
162
|
- `recover`: cleanup between retry attempts.
|
|
160
163
|
- `repeat`: repeated node expansion.
|
|
161
164
|
- `output`: output behavior selection.
|
|
165
|
+
- Command stdout/stderr use bounded tails plus complete spill files; tool/run diagnostics expose byte counts, truncation, and spill paths, while pipelines fail with `incomplete pipeline stdin` rather than consuming a partial tail.
|
|
162
166
|
|
|
163
167
|
Placeholders:
|
|
164
168
|
|
|
@@ -199,7 +203,7 @@ Rules:
|
|
|
199
203
|
7. Declare `mailbox` for actors that accept or emit meaningful messages.
|
|
200
204
|
8. Declare `artifacts` for durable outputs the coordinator should inspect.
|
|
201
205
|
9. File-backed recipe identity comes from the filename basename; legacy top-level `name` fields are ignored by loaders.
|
|
202
|
-
10. File-backed async recipes pass child `pi -p` actors a bounded JSONL recipe context bundle by default: raw entry/import recipe records, derived `name`, import path/alias, and `"you_are_here": true` on the launching recipe node. The runner
|
|
206
|
+
10. File-backed async recipes pass child `pi -p` actors a bounded JSONL recipe context bundle by default: raw entry/import recipe records, derived `name`, import path/alias, and `"you_are_here": true` on the launching recipe node. The runner collapses all natural-language positional fragments into one prompt under `prompts/command-NNN.md`, keeps intentional `@file` attachments separate, and invokes Pi with one authoritative prompt-file arg so large prompts and recipe context stay inspectable and argv-safe. Set `"actor_context": false` or `"off"` to suppress recipe context for minimal prompts.
|
|
203
207
|
11. Keep packaged recipes generic: no machine-local paths, no private companion identities, no project-specific defaults unless the recipe is explicitly project-specific.
|
|
204
208
|
12. Do not ship concrete model-version defaults in packaged recipes. For review-oriented subagent/lens recipes, default model/thinking args through `{current_model}` and `{current_thinking}` so they inherit the selected Pi session policy; keep `model`, `models`, `thinking`, and stage-specific model args explicit so callers can override policy at launch.
|
|
205
209
|
|
|
@@ -219,7 +223,7 @@ Muscle-memory lens: pi-actors has two durable executable-memory layers.
|
|
|
219
223
|
|
|
220
224
|
Agents grow active memory by calling `register_tool` or by deliberate recipe-file edits. They grow draft memory by trying ad hoc actors successfully. Treat both as executable habits: drafts are the workbench/proving ground; root recipes are promoted muscle memory.
|
|
221
225
|
|
|
222
|
-
Usage lens: user
|
|
226
|
+
Usage lens: user recipe launches update extension-maintained `.usage/<recipe-filename>.json` sidecars with fields such as `usage.calls` and `usage.last_called`; authored recipe files are not rewritten for telemetry. Discovery merges the sidecar into inspection. Agents should not hand-edit counters as part of normal recipe maintenance. Treat usage as evidence for usefulness analysis: heavily used recipes are good candidates for promotion, documentation, or stronger tests; unused recipes are cleanup candidates. Do not use failure counts as a primary usefulness signal because failures may reflect bad caller judgment rather than bad recipes. Do not delete or demote solely from counters without operator approval.
|
|
223
227
|
|
|
224
228
|
Promotion lens: successful transient/ad hoc actor runs are evidence, not commands. Inline spawns leave draft recipes as replayable evidence, not active tools. If a draft is repeatable, parameterized, safe enough, and likely useful later, the agent may promote it by moving/copying it into `~/.pi/agent/recipes` or by calling `register_tool` with a concise name, typed args/defaults, and a reviewed template or recipe path. Do not auto-register every success; do not promote temp paths, secrets, one-off prompts, or project-private assumptions without normalization and approval.
|
|
225
229
|
|
|
@@ -284,7 +288,7 @@ The user recipe root is the default tool set by location. It accepts canonical J
|
|
|
284
288
|
|
|
285
289
|
Use packaged recipes by name with `spawn file=<name>` for async actors, or register/call them as tools when repeated use deserves a stable shortcut.
|
|
286
290
|
|
|
287
|
-
Packaged review recipes are directly spawnable. Use `spawn file="pipeline-review-readiness" values={...}` for readiness review or `spawn file="subagent-review" values={...}` for one reviewer; pass model/thinking/tool policy through values, then inspect the run. Review coordinators preflight stage models before fanout; `ACTOR_PREFLIGHT_FAILED` diagnostics identify the failed stage, selected policy, provider error class, prompt file, and override args. Quorum-aware review fanout exposes `subagent_ttl_ms`, `reviewer_concurrency`, `min_successful_reviewers`, and `merge_policy`; partial reviewer evidence is preserved and marked `complete`, `degraded`, or `insufficient_data`. Run status/progress exposes `model_policy` so inherited vs explicit model/thinking choices remain visible. Do not recreate their script commands, call packaged scripts directly, or create wrapper recipes just to launch the maintained recipe.
|
|
291
|
+
Packaged review recipes are directly spawnable. Use `spawn file="pipeline-review-readiness" values={...}` for readiness review or `spawn file="subagent-review" values={...}` for one reviewer; pass model/thinking/tool policy through values, then inspect the run. Review coordinators preflight stage models before fanout; `ACTOR_PREFLIGHT_FAILED` diagnostics identify the failed stage, selected policy, provider error class, prompt file, and override args. Review stages require the `ACTOR_REVIEW_RESULT` evidence marker, so format acknowledgements and input requests fail closed before satisfying quorum or flowing downstream; rejected stdout remains in branch diagnostics. Quorum-aware review fanout exposes `subagent_ttl_ms`, `reviewer_concurrency`, `min_successful_reviewers`, and `merge_policy`; partial reviewer evidence is preserved and marked `complete`, `degraded`, or `insufficient_data`. Run status/progress exposes `model_policy` so inherited vs explicit model/thinking choices remain visible. Do not recreate their script commands, call packaged scripts directly, or create wrapper recipes just to launch the maintained recipe.
|
|
288
292
|
|
|
289
293
|
- [`pipeline-room-swarm`](../../recipes/pipeline-room-swarm.json): room-visible swarm coordination with roles, rounds, optional locker, artifact synthesis, and `subagent_ttl_ms` for hard participant budgets.
|
|
290
294
|
- [`pipeline-repo-health`](../../recipes/pipeline-repo-health.json): git/doc/validation evidence → normalized repository health report.
|
package/docs/actor-messages.md
CHANGED
|
@@ -186,7 +186,7 @@ Recipes can declare their conversational surface:
|
|
|
186
186
|
}
|
|
187
187
|
```
|
|
188
188
|
|
|
189
|
-
`spawn` creates detached `run:<id>` actors from a recipe file/name or inline command template.
|
|
189
|
+
`spawn` creates detached `run:<id>` actors from a recipe file/name or inline command template. Public spawn state directories are runtime-owned so every actor remains addressable and retention cannot target a caller-selected directory; spawn metadata may include named `artifacts` for terminal follow-ups and inspection. Room rosters are durable but burst-safe: repeated messages that only update `last_seen` may be coalesced briefly, while semantic roster changes such as role/status/display still write immediately.
|
|
190
190
|
|
|
191
191
|
## Inspect
|
|
192
192
|
|
package/docs/async-runs.md
CHANGED
|
@@ -81,19 +81,26 @@ Use `run_id` on async recipe tools or `as: "run:<id>"` on `spawn` when the calle
|
|
|
81
81
|
- `{default_room}`: default room address, e.g. `room:review`.
|
|
82
82
|
- `{communication_file}`: compact communication snapshot path.
|
|
83
83
|
|
|
84
|
+
Review commands that require semantic evidence apply marker acceptance before command completion accounting. Rejected code-zero output is reported consistently as a failed command in events, progress, evidence, and outbox delivery; it cannot emit a success-level completion notification. Evidence records are written before command launch and lifecycle cancellation or kill finalizes any running record with its interrupted state, effective exit code, and attempt capture paths. Async attempt stdout/stderr files exist from attempt start, so even small partial streams remain auditable when a command never returns.
|
|
85
|
+
|
|
84
86
|
## State Files
|
|
85
87
|
|
|
86
88
|
Use ordinary files under the extension temp directory so status tools stay simple and inspectable:
|
|
87
89
|
|
|
88
|
-
-
|
|
90
|
+
- `.pi-actors-run-state.json`: runtime ownership marker binding the run id to the canonical state directory; launch reuse and destructive retention fail closed when it is absent, invalid, mismatched, or reached through a symlink alias. State reuse also fails closed whenever the persisted process identity mismatches a still-live pid, preventing corrupted metadata from admitting overlapping runners.
|
|
91
|
+
- `run.json`: pid, cross-platform `process_identity` proof (start time, command, and canonical cwd where available), optional source metadata (`launch_source`, `tool`, `recipe`, `recipe_file`), command-template config, cwd, coordinator owner id, values, named `artifacts`, mailbox metadata, created time, and state dir. Existing launch cwd aliases are resolved through native `realpath` before proof matching, so symlinked working directories do not degrade control to `unsupported_proof`.
|
|
89
92
|
- `communication.json`: compact actor communication snapshot with self/root/parent, default-room, member, and contact hints for room-aware scripts and agents.
|
|
90
93
|
- `progress.json`: phase, active command count, completed count, failures, updated time, and optional `model_policy` provenance for inherited/explicit model and thinking values.
|
|
91
94
|
- `events.jsonl`: append-only implementation lifecycle log.
|
|
92
95
|
- `outbox.jsonl`: implementation storage for actor-message envelopes used by `inspect view=messages`, coordinator notifications, or follow-up context. Coordinator follow-ups preserve bounded `body` previews plus message metadata for decision points.
|
|
93
96
|
- `stdout.log` and `stderr.log`: detached process output.
|
|
94
|
-
- `prompts/command-NNN.md`: state-owned prompt files
|
|
97
|
+
- `prompts/command-NNN.md`: state-owned prompt files that collapse child `pi -p` natural-language positional fragments and appended recipe context into one authoritative `@file` prompt while preserving intentional file/image arguments.
|
|
98
|
+
- `captures/command-NNN/attempt-NNN/{stdout,stderr}.log`: complete byte-exact command streams, retained even below the bounded in-memory capture limit and separated across retries.
|
|
99
|
+
- `review-evidence.json`: stable command/stage manifest linking prompts, repeated branches, capture attempts, byte counts, exit state, semantic marker acceptance, recipe context, and model/thinking policy; terminal status aligns with the run. Review pipelines inject prior-stage `ACTOR_EVIDENCE_REF` values into downstream prompts, record cited/missing report sources, and fail closed if a normalized report claims `complete` without every required reviewer, verifier, merger, and judge reference.
|
|
95
100
|
- `result.json`: final code, killed flag, output selector, and optional full-output path.
|
|
96
101
|
|
|
102
|
+
Public `spawn` always uses the runtime-owned run root; caller-selected state directories are rejected so `run:<id>` addressing and retention share one boundary. Internal adapters may still supply isolated state directories for deterministic fixtures, but those are not part of the public actor contract. Every launched runner also persists a process identity proof and revalidates it for status, state reuse, message delivery, cancellation, kill, and retirement; dead pids, reused-pid owner mismatches, and unavailable platform proofs remain distinct diagnostics and destructive controls fail closed.
|
|
103
|
+
|
|
97
104
|
For pi-actors, actor run state defaults to:
|
|
98
105
|
|
|
99
106
|
```text
|
|
@@ -110,6 +117,8 @@ State files use this shape:
|
|
|
110
117
|
~/.pi/agent/tmp/pi-actors/runs/<run>/outbox.jsonl
|
|
111
118
|
~/.pi/agent/tmp/pi-actors/runs/<run>/stdout.log
|
|
112
119
|
~/.pi/agent/tmp/pi-actors/runs/<run>/stderr.log
|
|
120
|
+
~/.pi/agent/tmp/pi-actors/runs/<run>/review-evidence.json
|
|
121
|
+
~/.pi/agent/tmp/pi-actors/runs/<run>/captures/command-NNN/attempt-NNN/stdout.log
|
|
113
122
|
~/.pi/agent/tmp/pi-actors/runs/<run>/prompts/command-001.md
|
|
114
123
|
~/.pi/agent/tmp/pi-actors/runs/<run>/result.json
|
|
115
124
|
```
|
|
@@ -170,7 +179,7 @@ Low-level async actions map into the actor surface instead of forming a second p
|
|
|
170
179
|
- Send/control → `message`
|
|
171
180
|
- Status/tail/messages/list → `inspect`
|
|
172
181
|
- Force kill → `message` with `control.kill`, with synchronous results
|
|
173
|
-
- Archive/prune terminal state → `message` with `control.archive` or `control.prune`, with active runs rejected fail-closed
|
|
182
|
+
- Archive/prune terminal state → `message` with `control.archive` or `control.prune`, with active runs rejected fail-closed; retained artifacts use collision-safe identity-derived filenames, preserve timestamps, skip missing optional files, and abort prune before source deletion on any copy failure
|
|
174
183
|
|
|
175
184
|
Compact text is returned by default so async management does not flood agent context; use verbose inspection when the full state object is needed. List output intentionally shares one state root across music, subagents, timers, and other async work; source fields such as `tool` and `recipe` distinguish run purpose when the launcher recorded them. The run root may contain a rebuildable `index.json` with run id, state directory, owner, status, update time, and recipe/tool hints; corrupt indexes fall back to recursive scan. Registered tools are the preferred user-facing surface for reusable recipes. `control.prune` accepts `body.preserve_artifacts=true` to copy existing named artifacts beside the run root before deleting terminal state.
|
|
176
185
|
|
|
@@ -207,9 +216,9 @@ Runtime wake notifications are now modeled separately from durable queues. Messa
|
|
|
207
216
|
|
|
208
217
|
## Coordinator Notifications
|
|
209
218
|
|
|
210
|
-
The launching coordinator should not busy-poll long-running async runs. The extension watches run state directories and
|
|
219
|
+
The launching coordinator should not busy-poll long-running async runs. The extension watches run state directories and steers terminal `done`/`failed`/unhandled `killed`/`exited` transitions back to the owning session with `triggerTurn: true`; a busy coordinator receives completion at the next safe tool boundary, while an idle coordinator starts a normal turn without a racy manual idle check. Script-authored `notify`/`followup` actor messages still follow their declared outbox delivery policy. Terminal notifications include recipe-level named `artifacts` when declared. The generic runner also emits compact `command.done` actor messages for completed leaf commands; recipe authors declare that capability in `mailbox.emits` rather than configuring a separate delivery policy. Failures and in-flight parallel branch completions can bubble according to outbox policy, while successful final leaf completions stay diagnostic to avoid flooding long sequential pipelines. Intentional `control.kill` and recipe-local stop commands stay out of coordinator context because the initiating message already returns synchronously or is handled by actor-local policy. If a notification asks for direction, answer with `message` rather than starting a polling loop. Use explicit `inspect` only when a delivered notification requests inspection, a real decision depends on state, or a suspected stuck run needs diagnosis — never merely because a timeout elapsed.
|
|
211
220
|
|
|
212
|
-
Ambient status indicators may refresh while work is active, but coordinator attention is driven from run-state changes rather than a coordinator agent loop. This lets the coordinator continue other work after `spawn`; the run signals back through lifecycle state, results, and actor messages. The ambient triangle count represents active async work units: each running async run contributes at least one triangle, and a run with multiple active parallel command/subagent branches contributes the reported active branch count. If a coordinator starts one parent run with four active parallel branches, four triangles are shown; if the same coordinator starts five independent single-branch runs, five triangles are shown.
|
|
221
|
+
Ambient status indicators may refresh while work is active, but coordinator attention is driven from run-state changes rather than a coordinator agent loop. This lets the coordinator continue other work after `spawn`; the run signals back through lifecycle state, results, and actor messages. An owned terminal run without `terminal-handled.json` remains retry-eligible during same-runtime and extension/session replacement reconciliation; the marker is written only after successful steering delivery, and initial reconciliation does not replay historical outbox traffic. This is an at-least-once contract: a process crash after send but before marker persistence can produce a duplicate notification, while a failed send remains durably retryable. The ambient triangle count represents active async work units: each running async run contributes at least one triangle, and a run with multiple active parallel command/subagent branches contributes the reported active branch count. If a coordinator starts one parent run with four active parallel branches, four triangles are shown; if the same coordinator starts five independent single-branch runs, five triangles are shown.
|
|
213
222
|
|
|
214
223
|
## Run Actor Messages
|
|
215
224
|
|
|
@@ -57,6 +57,7 @@ Common object fields:
|
|
|
57
57
|
- `defaults`: Placeholder default values by name.
|
|
58
58
|
- `timeout`: Optional execution timeout in milliseconds. Omit it, or set `0`, to leave the command unbounded. Set an explicit positive timeout when a tool must fail closed instead of waiting indefinitely. Numeric control fields may be literal numbers or placeholders such as `"{timeout_ms}"`.
|
|
59
59
|
- `delay`: Optional wait in milliseconds before starting this node. Default is no delay. It may be a literal number or placeholder.
|
|
60
|
+
- `accept_output`: Optional fail-closed semantic output contract. `review_evidence` requires successful stdout to begin with `ACTOR_REVIEW_RESULT`; missing markers become code-65 failures while rejected stdout remains visible in branch diagnostics.
|
|
60
61
|
- `output`: Optional result selector. Default is `"stdout"`; runtime values such as `"ogg"` are valid.
|
|
61
62
|
- `retry`: Optional max attempts including the first. Default is `1`.
|
|
62
63
|
- `failure`: Optional failure propagation scope: `continue`, `branch`, or `root`. Default is `continue`.
|
|
@@ -185,7 +186,8 @@ Composition rules:
|
|
|
185
186
|
- Top-level `args` and `defaults` apply to every leaf unless the leaf defines private values
|
|
186
187
|
- Leaf `args` replace inherited `args`; leaf `defaults` merge over inherited defaults; `timeout` and `output` are not inherited into leaves
|
|
187
188
|
- Timeout is disabled by default; configure a positive `timeout` for bounded commands that should fail closed
|
|
188
|
-
-
|
|
189
|
+
- Child stdout and stderr are captured as raw bytes with independent bounded in-memory tails; streams that exceed the capture limit spill completely to byte-exact diagnostic files and return byte counts, truncation flags, and spill paths. Text tails align their start to a UTF-8 code-point boundary so chunking or truncation does not corrupt valid multibyte output. Async runs persist complete streams even below the capture limit under command- and retry-specific run-state paths
|
|
190
|
+
- Each sequence leaf receives the previous leaf's complete stdout on stdin by default, reading the byte-complete spill when the model-facing capture is truncated; the final leaf stdout remains the bounded default composition result. Parallel joins likewise use complete branch spills for downstream stdin while branch details and returned output stay bounded. If a declared spill is unavailable, the pipeline fails closed instead of forwarding a partial tail
|
|
189
191
|
- Skipped nodes preserve current stdin/stdout flow and do not execute commands
|
|
190
192
|
- Each parallel child receives the same stdin, and child stdout values are joined in stable array order before flowing to the next sequence leaf
|
|
191
193
|
- Parallel branch joins include branch label and status, and tool details include branch metadata plus coverage summary
|
|
@@ -401,7 +403,7 @@ string → leaf command
|
|
|
401
403
|
string[] → sequential composition
|
|
402
404
|
{ template } → leaf command object
|
|
403
405
|
{ parallel, template } → sequence or parallel subtree
|
|
404
|
-
{ parallel, concurrency, min_successful, when, args, defaults, delay, retry, failure, recover, output, template } → full node
|
|
406
|
+
{ parallel, concurrency, min_successful, when, args, defaults, delay, retry, failure, recover, accept_output, output, template } → full node
|
|
405
407
|
```
|
|
406
408
|
|
|
407
409
|
Start with a string. Add composition when needed. Add `parallel: true` when independent work can run concurrently. Add `when` when a node is conditional. Add delay when launch pacing matters. Add retry when flaky. Add `failure` when propagation scope matters. Add `recover` when a retried node needs cleanup before another attempt. Same contract, growing capability, no dead weight.
|
package/docs/recipe-library.md
CHANGED
|
@@ -30,6 +30,7 @@ Core subagent recipes:
|
|
|
30
30
|
- `recipes/subagent-tools.json`: Start a subagent with an explicit tool allowlist.
|
|
31
31
|
- `recipes/subagents-prompts.json`: Run prompt fanout with one imported subagent component.
|
|
32
32
|
- `recipes/subagent-preflight.json`: Tiny model/thinking/tool-policy smoke check before expensive fanout; failures surface `ACTOR_PREFLIGHT_FAILED` with stage, selected policy, provider error class, prompt file, and override args.
|
|
33
|
+
- Packaged reviewer, verifier, merger, judge, and normalizer stages use `accept_output: review_evidence` and require `ACTOR_REVIEW_RESULT` as the exact first non-whitespace output line. Marker prefixes, format acknowledgements, and input requests therefore remain rejected branch diagnostics rather than usable quorum evidence.
|
|
33
34
|
- `recipes/subagent-review.json`: Evidence-grounded review lens.
|
|
34
35
|
- `recipes/subagent-critic.json`: Assumption and failure-mode critique.
|
|
35
36
|
- `recipes/subagent-plan.json`: Bounded plan slices and validation gates.
|