@llblab/pi-actors 0.39.0 → 0.40.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +10 -3
- package/BACKLOG.md +4 -11
- package/CHANGELOG.md +51 -0
- package/README.md +5 -5
- package/dist/index.js +6 -4
- package/dist/lib/async-runs.d.ts +4 -0
- package/dist/lib/async-runs.js +112 -17
- package/dist/lib/command-templates.d.ts +9 -0
- package/dist/lib/command-templates.js +92 -11
- package/dist/lib/config.js +0 -5
- package/dist/lib/execution.d.ts +31 -0
- package/dist/lib/execution.js +145 -12
- package/dist/lib/file-state.d.ts +1 -0
- package/dist/lib/file-state.js +91 -3
- package/dist/lib/observability.js +5 -2
- package/dist/lib/prompts.d.ts +1 -2
- package/dist/lib/prompts.js +3 -4
- package/dist/lib/recipes-context.js +17 -9
- package/dist/lib/recipes-discovery.js +11 -5
- package/dist/lib/recipes-references.d.ts +1 -1
- package/dist/lib/recipes-references.js +4 -5
- package/dist/lib/recipes-usage.d.ts +2 -0
- package/dist/lib/recipes-usage.js +35 -21
- package/dist/lib/registry.d.ts +0 -2
- package/dist/lib/registry.js +33 -10
- package/dist/lib/runs-ownership.d.ts +7 -0
- package/dist/lib/runs-ownership.js +82 -0
- package/dist/lib/runs-process.d.ts +17 -2
- package/dist/lib/runs-process.js +99 -11
- package/dist/lib/runs-retention.d.ts +3 -0
- package/dist/lib/runs-retention.js +18 -3
- package/dist/lib/runs-start.d.ts +2 -2
- package/dist/lib/runs-start.js +51 -17
- package/dist/lib/runs-status.d.ts +1 -1
- package/dist/lib/runs-status.js +8 -6
- package/dist/lib/runtime.js +69 -13
- package/dist/lib/tools-inspect.d.ts +2 -0
- package/dist/lib/tools-inspect.js +39 -2
- package/dist/lib/tools-register.js +0 -1
- package/dist/lib/tools-spawn.js +3 -2
- package/dist/lib/tools.d.ts +1 -0
- package/dist/lib/tools.js +3 -0
- package/dist/pi-actors/index.js +1 -0
- package/dist/recipes/subagent-judge.json +2 -1
- package/dist/recipes/subagent-merge.json +2 -1
- package/dist/recipes/subagent-normalize.json +2 -1
- package/dist/recipes/subagent-review-coordinator.json +1 -1
- package/dist/recipes/subagent-review.json +2 -1
- package/dist/recipes/subagent-verify.json +2 -1
- package/dist/scripts/async-runner.mjs +274 -6
- package/dist/scripts/build-dist.mjs +14 -1
- package/dist/skills/actors/SKILL.md +15 -8
- package/dist/skills/swarm/SKILL.md +4 -2
- package/docs/actor-messages.md +1 -1
- package/docs/async-runs.md +14 -5
- package/docs/command-templates.md +4 -2
- package/docs/recipe-library.md +1 -0
- package/docs/template-recipes.md +5 -7
- package/docs/tool-registry.md +4 -2
- package/index.ts +18 -7
- package/lib/async-runs.ts +138 -19
- package/lib/command-templates.ts +132 -13
- package/lib/config.ts +0 -4
- package/lib/execution.ts +198 -13
- package/lib/file-state.ts +106 -3
- package/lib/observability.ts +8 -2
- package/lib/prompts.ts +3 -5
- package/lib/recipes-context.ts +17 -9
- package/lib/recipes-discovery.ts +10 -5
- package/lib/recipes-references.ts +5 -6
- package/lib/recipes-usage.ts +36 -20
- package/lib/registry.ts +43 -13
- package/lib/runs-ownership.ts +117 -0
- package/lib/runs-process.ts +138 -16
- package/lib/runs-retention.ts +22 -2
- package/lib/runs-start.ts +89 -31
- package/lib/runs-status.ts +15 -6
- package/lib/runtime.ts +64 -12
- package/lib/tools-inspect.ts +46 -4
- package/lib/tools-register.ts +0 -3
- package/lib/tools-spawn.ts +5 -5
- package/lib/tools.ts +8 -0
- package/package.json +2 -2
- package/recipes/subagent-judge.json +2 -1
- package/recipes/subagent-merge.json +2 -1
- package/recipes/subagent-normalize.json +2 -1
- package/recipes/subagent-review-coordinator.json +1 -1
- package/recipes/subagent-review.json +2 -1
- package/recipes/subagent-verify.json +2 -1
- package/scripts/async-runner.mjs +274 -6
- package/scripts/build-dist.mjs +14 -1
- package/skills/actors/SKILL.md +15 -8
- package/skills/swarm/SKILL.md +4 -2
package/scripts/async-runner.mjs
CHANGED
|
@@ -8,8 +8,14 @@
|
|
|
8
8
|
* chasing a one-off lib entrypoint domain.
|
|
9
9
|
*/
|
|
10
10
|
|
|
11
|
-
import {
|
|
12
|
-
|
|
11
|
+
import {
|
|
12
|
+
appendFileSync,
|
|
13
|
+
existsSync,
|
|
14
|
+
mkdirSync,
|
|
15
|
+
readFileSync,
|
|
16
|
+
statSync,
|
|
17
|
+
} from "node:fs";
|
|
18
|
+
import { dirname, join, relative } from "node:path";
|
|
13
19
|
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
14
20
|
|
|
15
21
|
function packageRoot() {
|
|
@@ -30,7 +36,8 @@ const { appendRecipeContextToPiArgs, materializePiPrintPromptArg } =
|
|
|
30
36
|
const { buildReviewPreflightDiagnostic, formatReviewPreflightDiagnostic } =
|
|
31
37
|
await importRuntimeModule("preflight-diagnostics");
|
|
32
38
|
const { execCommandTemplate } = await importRuntimeModule("command-templates");
|
|
33
|
-
const { executeRegisteredTool } =
|
|
39
|
+
const { applyOutputAcceptancePolicy, executeRegisteredTool } =
|
|
40
|
+
await importRuntimeModule("execution");
|
|
34
41
|
const { writeJsonAtomic } = await importRuntimeModule("file-state");
|
|
35
42
|
|
|
36
43
|
function quoteCommandDetailPart(value) {
|
|
@@ -43,6 +50,17 @@ function formatCommandDetail(command, args) {
|
|
|
43
50
|
return [command, ...args].map(quoteCommandDetailPart).join(" ");
|
|
44
51
|
}
|
|
45
52
|
|
|
53
|
+
function captureDetails(result) {
|
|
54
|
+
return {
|
|
55
|
+
...(typeof result.stdoutBytes === "number" ? { stdout_bytes: result.stdoutBytes } : {}),
|
|
56
|
+
...(typeof result.stderrBytes === "number" ? { stderr_bytes: result.stderrBytes } : {}),
|
|
57
|
+
...(result.stdoutFile ? { stdout_file: result.stdoutFile } : {}),
|
|
58
|
+
...(result.stderrFile ? { stderr_file: result.stderrFile } : {}),
|
|
59
|
+
...(result.stdoutTruncated ? { stdout_truncated: true } : {}),
|
|
60
|
+
...(result.stderrTruncated ? { stderr_truncated: true } : {}),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
46
64
|
function summarizeCommandDetail(commandDetail) {
|
|
47
65
|
return commandDetail.length > 160
|
|
48
66
|
? `${commandDetail.slice(0, 157)}...`
|
|
@@ -59,6 +77,7 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
59
77
|
const outboxPath = join(stateDir, "outbox.jsonl");
|
|
60
78
|
const stdoutPath = join(stateDir, "stdout.log");
|
|
61
79
|
const stderrPath = join(stateDir, "stderr.log");
|
|
80
|
+
const evidencePath = join(stateDir, "review-evidence.json");
|
|
62
81
|
const meta = JSON.parse(readFileSync(runPath, "utf8"));
|
|
63
82
|
|
|
64
83
|
function event(name, data = {}) {
|
|
@@ -87,7 +106,171 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
87
106
|
let activeSubagents = 0;
|
|
88
107
|
let completedSubagents = 0;
|
|
89
108
|
let promptCounter = 0;
|
|
109
|
+
let captureCounter = 0;
|
|
90
110
|
const subagentFailures = [];
|
|
111
|
+
const evidenceRecords = [];
|
|
112
|
+
const stageOccurrences = new Map();
|
|
113
|
+
let reportEvidence;
|
|
114
|
+
|
|
115
|
+
function writeEvidenceManifest(status) {
|
|
116
|
+
writeJsonAtomic(evidencePath, {
|
|
117
|
+
version: 1,
|
|
118
|
+
run: meta.run,
|
|
119
|
+
status,
|
|
120
|
+
...(meta.model_policy ? { model_policy: meta.model_policy } : {}),
|
|
121
|
+
commands: [...evidenceRecords].sort((left, right) =>
|
|
122
|
+
left.id.localeCompare(right.id)
|
|
123
|
+
),
|
|
124
|
+
...(reportEvidence ? { report_evidence: reportEvidence } : {}),
|
|
125
|
+
updated_at: new Date().toISOString(),
|
|
126
|
+
});
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function commandEvidenceStartRecord({
|
|
130
|
+
commandDetail,
|
|
131
|
+
commandId,
|
|
132
|
+
materialized,
|
|
133
|
+
options,
|
|
134
|
+
stage,
|
|
135
|
+
occurrence,
|
|
136
|
+
}) {
|
|
137
|
+
const recipeContext = options?.actorRecipeContext;
|
|
138
|
+
return {
|
|
139
|
+
id: commandId,
|
|
140
|
+
stage,
|
|
141
|
+
occurrence,
|
|
142
|
+
status: "running",
|
|
143
|
+
started_at: new Date().toISOString(),
|
|
144
|
+
...(options?.evidenceContext?.label
|
|
145
|
+
? { label: options.evidenceContext.label }
|
|
146
|
+
: {}),
|
|
147
|
+
...(options?.evidenceContext?.repeatIndex !== undefined
|
|
148
|
+
? { branch_index: options.evidenceContext.repeatIndex }
|
|
149
|
+
: {}),
|
|
150
|
+
...(recipeContext ? { recipe_context: recipeContext } : {}),
|
|
151
|
+
command: commandDetail,
|
|
152
|
+
...(materialized.promptFile
|
|
153
|
+
? { prompt_file: relative(stateDir, materialized.promptFile) }
|
|
154
|
+
: {}),
|
|
155
|
+
...(materialized.promptBytes
|
|
156
|
+
? { prompt_bytes: materialized.promptBytes }
|
|
157
|
+
: {}),
|
|
158
|
+
attempts: [],
|
|
159
|
+
semantic_acceptance:
|
|
160
|
+
options?.evidenceContext?.acceptOutput === "review_evidence" ||
|
|
161
|
+
stage === "preflight"
|
|
162
|
+
? "pending"
|
|
163
|
+
: "not_required",
|
|
164
|
+
};
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function commandEvidenceRecord({
|
|
168
|
+
captureDir,
|
|
169
|
+
commandDetail,
|
|
170
|
+
commandId,
|
|
171
|
+
materialized,
|
|
172
|
+
options,
|
|
173
|
+
result,
|
|
174
|
+
rawExitCode,
|
|
175
|
+
stage,
|
|
176
|
+
occurrence,
|
|
177
|
+
startedAt,
|
|
178
|
+
}) {
|
|
179
|
+
const recipeContext = options?.actorRecipeContext;
|
|
180
|
+
const attempts = [];
|
|
181
|
+
const maxAttempts = Math.max(1, options?.retry || 1);
|
|
182
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
|
|
183
|
+
const attemptDir = join(
|
|
184
|
+
captureDir,
|
|
185
|
+
`attempt-${String(attempt).padStart(3, "0")}`,
|
|
186
|
+
);
|
|
187
|
+
const stdoutFile = join(attemptDir, "stdout.log");
|
|
188
|
+
const stderrFile = join(attemptDir, "stderr.log");
|
|
189
|
+
if (!existsSync(stdoutFile) && !existsSync(stderrFile)) continue;
|
|
190
|
+
attempts.push({
|
|
191
|
+
attempt,
|
|
192
|
+
stdout: {
|
|
193
|
+
path: relative(stateDir, stdoutFile),
|
|
194
|
+
bytes: existsSync(stdoutFile) ? statSync(stdoutFile).size : 0,
|
|
195
|
+
},
|
|
196
|
+
stderr: {
|
|
197
|
+
path: relative(stateDir, stderrFile),
|
|
198
|
+
bytes: existsSync(stderrFile) ? statSync(stderrFile).size : 0,
|
|
199
|
+
},
|
|
200
|
+
});
|
|
201
|
+
}
|
|
202
|
+
const expectedReviewMarker =
|
|
203
|
+
options?.evidenceContext?.acceptOutput === "review_evidence";
|
|
204
|
+
const expectedPreflightMarker = stage === "preflight";
|
|
205
|
+
const semanticStdout = result.stdoutTruncated && result.stdoutFile && existsSync(result.stdoutFile)
|
|
206
|
+
? readFileSync(result.stdoutFile, "utf8")
|
|
207
|
+
: result.stdout;
|
|
208
|
+
const firstNonWhitespaceLine = semanticStdout
|
|
209
|
+
.split(/\r?\n/)
|
|
210
|
+
.find((line) => line.trim().length > 0);
|
|
211
|
+
const markerAccepted = expectedReviewMarker
|
|
212
|
+
? firstNonWhitespaceLine?.trim() === "ACTOR_REVIEW_RESULT"
|
|
213
|
+
: expectedPreflightMarker
|
|
214
|
+
? result.stdout.trimStart().startsWith("ACTOR_PREFLIGHT_OK")
|
|
215
|
+
: undefined;
|
|
216
|
+
return {
|
|
217
|
+
id: commandId,
|
|
218
|
+
stage,
|
|
219
|
+
occurrence,
|
|
220
|
+
status: result.code === 0 ? "done" : "failed",
|
|
221
|
+
started_at: startedAt,
|
|
222
|
+
completed_at: new Date().toISOString(),
|
|
223
|
+
...(options?.evidenceContext?.label
|
|
224
|
+
? { label: options.evidenceContext.label }
|
|
225
|
+
: {}),
|
|
226
|
+
...(options?.evidenceContext?.repeatIndex !== undefined
|
|
227
|
+
? { branch_index: options.evidenceContext.repeatIndex }
|
|
228
|
+
: {}),
|
|
229
|
+
...(recipeContext ? { recipe_context: recipeContext } : {}),
|
|
230
|
+
command: commandDetail,
|
|
231
|
+
...(materialized.promptFile
|
|
232
|
+
? { prompt_file: relative(stateDir, materialized.promptFile) }
|
|
233
|
+
: {}),
|
|
234
|
+
...(materialized.promptBytes
|
|
235
|
+
? { prompt_bytes: materialized.promptBytes }
|
|
236
|
+
: {}),
|
|
237
|
+
attempts,
|
|
238
|
+
exit_code: rawExitCode ?? result.code,
|
|
239
|
+
effective_exit_code: result.code,
|
|
240
|
+
killed: result.killed,
|
|
241
|
+
stdout_bytes: result.stdoutBytes ?? Buffer.byteLength(result.stdout),
|
|
242
|
+
stderr_bytes: result.stderrBytes ?? Buffer.byteLength(result.stderr),
|
|
243
|
+
stdout_truncated: result.stdoutTruncated === true,
|
|
244
|
+
stderr_truncated: result.stderrTruncated === true,
|
|
245
|
+
semantic_acceptance:
|
|
246
|
+
markerAccepted === undefined
|
|
247
|
+
? "not_required"
|
|
248
|
+
: markerAccepted && result.code === 0
|
|
249
|
+
? "accepted"
|
|
250
|
+
: "rejected",
|
|
251
|
+
};
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
function auditReviewReport(text) {
|
|
255
|
+
const required = evidenceRecords
|
|
256
|
+
.filter((record) =>
|
|
257
|
+
["reviewer", "verifier", "merger", "judge"].includes(record.stage)
|
|
258
|
+
)
|
|
259
|
+
.map((record) => `review-evidence.json#${record.id}`);
|
|
260
|
+
if (required.length === 0) return undefined;
|
|
261
|
+
const cited = [...new Set(
|
|
262
|
+
text.match(/review-evidence\.json#command-\d{3}/g) || [],
|
|
263
|
+
)].sort();
|
|
264
|
+
const missing = required.filter((reference) => !cited.includes(reference));
|
|
265
|
+
const claimsComplete = /\bStatus\b[\s:*#_-]{0,40}\bcomplete\b/i.test(text);
|
|
266
|
+
return {
|
|
267
|
+
required,
|
|
268
|
+
cited,
|
|
269
|
+
missing,
|
|
270
|
+
claims_complete: claimsComplete,
|
|
271
|
+
complete_allowed: missing.length === 0,
|
|
272
|
+
};
|
|
273
|
+
}
|
|
91
274
|
|
|
92
275
|
function getCommandDoneDelivery(result) {
|
|
93
276
|
return result.code !== 0 || activeSubagents > 0 ? "followup" : "log";
|
|
@@ -131,15 +314,64 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
131
314
|
);
|
|
132
315
|
const execArgs = materialized.args;
|
|
133
316
|
const commandDetail = formatCommandDetail(command, execArgs);
|
|
317
|
+
captureCounter += 1;
|
|
318
|
+
const commandId = `command-${String(captureCounter).padStart(3, "0")}`;
|
|
319
|
+
const stage =
|
|
320
|
+
options?.actorRecipeContext?.alias ||
|
|
321
|
+
options?.actorRecipeContext?.name ||
|
|
322
|
+
"command";
|
|
323
|
+
const occurrence = (stageOccurrences.get(stage) || 0) + 1;
|
|
324
|
+
stageOccurrences.set(stage, occurrence);
|
|
325
|
+
if (
|
|
326
|
+
materialized.promptFile &&
|
|
327
|
+
["verifier", "merger", "judge", "normalizer"].includes(stage)
|
|
328
|
+
) {
|
|
329
|
+
const references = evidenceRecords
|
|
330
|
+
.filter((record) =>
|
|
331
|
+
["reviewer", "verifier", "merger", "judge"].includes(record.stage)
|
|
332
|
+
)
|
|
333
|
+
.map((record) => `ACTOR_EVIDENCE_REF: review-evidence.json#${record.id}`);
|
|
334
|
+
if (references.length > 0) {
|
|
335
|
+
appendFileSync(
|
|
336
|
+
materialized.promptFile,
|
|
337
|
+
`\n\nRetained actor evidence references available for citation:\n${references.join("\n")}\n`,
|
|
338
|
+
);
|
|
339
|
+
materialized.promptBytes = statSync(materialized.promptFile).size;
|
|
340
|
+
}
|
|
341
|
+
}
|
|
134
342
|
activeSubagents += 1;
|
|
343
|
+
const evidenceIndex = evidenceRecords.length;
|
|
344
|
+
const startedEvidence = commandEvidenceStartRecord({
|
|
345
|
+
commandDetail,
|
|
346
|
+
commandId,
|
|
347
|
+
materialized,
|
|
348
|
+
options,
|
|
349
|
+
stage,
|
|
350
|
+
occurrence,
|
|
351
|
+
});
|
|
352
|
+
evidenceRecords.push(startedEvidence);
|
|
353
|
+
writeEvidenceManifest("running");
|
|
135
354
|
event("command.start", {
|
|
136
355
|
activeSubagents,
|
|
356
|
+
command_id: commandId,
|
|
137
357
|
command: commandDetail,
|
|
138
358
|
...(materialized.promptFile ? { prompt_file: materialized.promptFile } : {}),
|
|
139
359
|
...(materialized.promptBytes ? { prompt_bytes: materialized.promptBytes } : {}),
|
|
140
360
|
});
|
|
141
361
|
progressRunning();
|
|
142
|
-
|
|
362
|
+
const captureDir = join(stateDir, "captures", commandId);
|
|
363
|
+
const rawResult = await execCommandTemplate(command, execArgs, {
|
|
364
|
+
...options,
|
|
365
|
+
captureDir,
|
|
366
|
+
});
|
|
367
|
+
let result = await applyOutputAcceptancePolicy(
|
|
368
|
+
rawResult,
|
|
369
|
+
options?.evidenceContext?.acceptOutput,
|
|
370
|
+
);
|
|
371
|
+
result = {
|
|
372
|
+
...result,
|
|
373
|
+
evidenceRef: `review-evidence.json#${commandId}`,
|
|
374
|
+
};
|
|
143
375
|
const preflightDiagnostic = result.code !== 0
|
|
144
376
|
? buildReviewPreflightDiagnostic({
|
|
145
377
|
args: execArgs,
|
|
@@ -160,6 +392,19 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
160
392
|
].filter(Boolean).join("\n"),
|
|
161
393
|
};
|
|
162
394
|
}
|
|
395
|
+
evidenceRecords[evidenceIndex] = commandEvidenceRecord({
|
|
396
|
+
captureDir,
|
|
397
|
+
commandDetail,
|
|
398
|
+
commandId,
|
|
399
|
+
materialized,
|
|
400
|
+
options,
|
|
401
|
+
result,
|
|
402
|
+
rawExitCode: rawResult.code,
|
|
403
|
+
stage,
|
|
404
|
+
occurrence,
|
|
405
|
+
startedAt: startedEvidence.started_at,
|
|
406
|
+
});
|
|
407
|
+
writeEvidenceManifest("running");
|
|
163
408
|
activeSubagents = Math.max(0, activeSubagents - 1);
|
|
164
409
|
completedSubagents += 1;
|
|
165
410
|
if (result.code !== 0) {
|
|
@@ -167,15 +412,18 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
167
412
|
code: result.code,
|
|
168
413
|
command: commandDetail,
|
|
169
414
|
killed: result.killed,
|
|
415
|
+
...captureDetails(result),
|
|
170
416
|
...(materialized.promptFile ? { prompt_file: materialized.promptFile } : {}),
|
|
171
417
|
...(preflightDiagnostic ? { preflight: preflightDiagnostic } : {}),
|
|
172
418
|
});
|
|
173
419
|
}
|
|
174
420
|
event("command.done", {
|
|
175
421
|
activeSubagents,
|
|
422
|
+
command_id: commandId,
|
|
176
423
|
code: result.code,
|
|
177
424
|
command: commandDetail,
|
|
178
425
|
killed: result.killed,
|
|
426
|
+
...captureDetails(result),
|
|
179
427
|
...(materialized.promptFile ? { prompt_file: materialized.promptFile } : {}),
|
|
180
428
|
...(materialized.promptBytes ? { prompt_bytes: materialized.promptBytes } : {}),
|
|
181
429
|
...(preflightDiagnostic ? { preflight: preflightDiagnostic } : {}),
|
|
@@ -186,10 +434,12 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
186
434
|
{
|
|
187
435
|
activeSubagents,
|
|
188
436
|
...(meta.artifacts ? { artifacts: meta.artifacts } : {}),
|
|
189
|
-
run_files: [stdoutPath, stderrPath, resultPath, eventsPath, outboxPath],
|
|
437
|
+
run_files: [stdoutPath, stderrPath, resultPath, eventsPath, outboxPath, evidencePath],
|
|
438
|
+
command_id: commandId,
|
|
190
439
|
code: result.code,
|
|
191
440
|
command: commandDetail,
|
|
192
441
|
killed: result.killed,
|
|
442
|
+
...captureDetails(result),
|
|
193
443
|
...(materialized.promptFile ? { prompt_file: materialized.promptFile } : {}),
|
|
194
444
|
...(materialized.promptBytes ? { prompt_bytes: materialized.promptBytes } : {}),
|
|
195
445
|
...(preflightDiagnostic ? { preflight: preflightDiagnostic } : {}),
|
|
@@ -218,6 +468,20 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
218
468
|
);
|
|
219
469
|
const text = result.content?.[0]?.text || "";
|
|
220
470
|
appendFileSync(stdoutPath, text);
|
|
471
|
+
reportEvidence = auditReviewReport(text);
|
|
472
|
+
if (
|
|
473
|
+
reportEvidence?.claims_complete &&
|
|
474
|
+
reportEvidence.complete_allowed !== true
|
|
475
|
+
) {
|
|
476
|
+
const error = new Error(
|
|
477
|
+
`review report evidence incomplete: missing ${reportEvidence.missing.join(", ")}`,
|
|
478
|
+
);
|
|
479
|
+
error.details = {
|
|
480
|
+
code: 65,
|
|
481
|
+
failureReason: "incomplete review report evidence",
|
|
482
|
+
};
|
|
483
|
+
throw error;
|
|
484
|
+
}
|
|
221
485
|
writeJsonAtomic(resultPath, {
|
|
222
486
|
code: result.details.code,
|
|
223
487
|
command: result.details.command,
|
|
@@ -226,6 +490,7 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
226
490
|
truncated: result.details.truncated,
|
|
227
491
|
completedAt: new Date().toISOString(),
|
|
228
492
|
});
|
|
493
|
+
writeEvidenceManifest("done");
|
|
229
494
|
progress("done", {
|
|
230
495
|
completed: 1,
|
|
231
496
|
failures: result.details.nonCriticalFailures || [],
|
|
@@ -245,11 +510,14 @@ export async function runAsyncRunner(stateDir = process.argv[2]) {
|
|
|
245
510
|
...(details?.softQuorum ? { soft_quorum: details.softQuorum } : {}),
|
|
246
511
|
completedAt: new Date().toISOString(),
|
|
247
512
|
});
|
|
513
|
+
writeEvidenceManifest("failed");
|
|
248
514
|
progress("failed", {
|
|
249
515
|
completed: 0,
|
|
250
516
|
failures: Array.isArray(details?.branches) && details.branches.length > 0
|
|
251
517
|
? details.branches
|
|
252
|
-
:
|
|
518
|
+
: subagentFailures.length > 0
|
|
519
|
+
? subagentFailures
|
|
520
|
+
: [{ message }],
|
|
253
521
|
...(details?.failureReason ? { failureReason: details.failureReason } : {}),
|
|
254
522
|
});
|
|
255
523
|
event("run.failed", {
|
package/scripts/build-dist.mjs
CHANGED
|
@@ -9,7 +9,13 @@
|
|
|
9
9
|
*/
|
|
10
10
|
|
|
11
11
|
import { spawnSync } from "node:child_process";
|
|
12
|
-
import {
|
|
12
|
+
import {
|
|
13
|
+
cpSync,
|
|
14
|
+
mkdirSync,
|
|
15
|
+
readdirSync,
|
|
16
|
+
rmSync,
|
|
17
|
+
writeFileSync,
|
|
18
|
+
} from "node:fs";
|
|
13
19
|
import { join } from "node:path";
|
|
14
20
|
|
|
15
21
|
function run(command, args) {
|
|
@@ -22,6 +28,13 @@ mkdirSync("dist", { recursive: true });
|
|
|
22
28
|
|
|
23
29
|
run("tsc", ["-p", "tsconfig.build.json"]);
|
|
24
30
|
|
|
31
|
+
mkdirSync(join("dist", "pi-actors"), { recursive: true });
|
|
32
|
+
writeFileSync(
|
|
33
|
+
join("dist", "pi-actors", "index.js"),
|
|
34
|
+
'export { default } from "../index.js";\n',
|
|
35
|
+
"utf8",
|
|
36
|
+
);
|
|
37
|
+
|
|
25
38
|
for (const dir of ["scripts", "recipes", "fixtures", "skills"]) {
|
|
26
39
|
cpSync(dir, join("dist", dir), { recursive: true });
|
|
27
40
|
}
|
package/skills/actors/SKILL.md
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: actors
|
|
3
|
-
description: Required practical guide for non-trivial pi-actors use. Read before using or changing spawn, message, inspect, actor runs, tools, recipes, command templates, async lifecycle, mailboxes, artifacts, and local orchestration mechanics.
|
|
3
|
+
description: Required practical guide for non-trivial pi-actors use, including parallel actor launches, subagent fanout, and autonomous coordinator workflows. Read before using or changing spawn, message, inspect, actor runs, tools, recipes, command templates, async lifecycle, mailboxes, artifacts, and local orchestration mechanics.
|
|
4
4
|
metadata:
|
|
5
|
-
version: 0.
|
|
5
|
+
version: 0.40.1
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# Actors (pi-actors)
|
|
@@ -53,6 +53,8 @@ Actor-mode trigger: if work may outlive this turn, needs steering/follow-up/arti
|
|
|
53
53
|
|
|
54
54
|
Use for long work, background services, subagents, fanout, pipelines, and reusable recipes.
|
|
55
55
|
|
|
56
|
+
Before a parallel launch, the coordinator should verify five things once: each actor owns a disjoint mutation scope, every run has a stable id, each durable result has an artifact path, every command template respects shell-free argv execution, and completion can return through follow-up delivery without polling. For multiple delegated implementation or review actors, also load the bundled Swarm skill before choosing decomposition, locks, or quorum shape.
|
|
57
|
+
|
|
56
58
|
```json
|
|
57
59
|
{
|
|
58
60
|
"as": "run:repo-health",
|
|
@@ -64,10 +66,12 @@ Use for long work, background services, subagents, fanout, pipelines, and reusab
|
|
|
64
66
|
|
|
65
67
|
Rules:
|
|
66
68
|
|
|
69
|
+
- Command-template strings execute directly without a shell. Operators such as `&&`, `||`, pipes, redirects, and `cd` remain literal argv unless an explicit trusted shell is the executable. Prefer absolute paths or template arrays for sequencing; put non-trivial shell behavior in a reviewed script.
|
|
67
70
|
- Use `file`/`recipe` for saved recipes; bare names resolve under `~/.pi/agent/recipes`.
|
|
68
71
|
- Use inline `template` for one-off experiments; promote useful repeats to recipes.
|
|
69
72
|
- When a successful actor follow-up suggests persistence, decide whether the pattern deserves durable tool memory; call `register_tool` yourself only when the evidence is strong, and ask before writing the user recipe root.
|
|
70
73
|
- Use stable `as` names when you will inspect or message the actor later.
|
|
74
|
+
- Public run state is runtime-owned; do not pass custom `state_dir` paths. This keeps `run:<id>` addressability and retention on one boundary.
|
|
71
75
|
- `async: true` on the recipe is the detached run switch.
|
|
72
76
|
|
|
73
77
|
### `message` — send a typed envelope
|
|
@@ -90,6 +94,7 @@ Envelope fields:
|
|
|
90
94
|
- Group posts to `room:<run>` require `from` from the same run (`run:<run>` or `branch:<run>/<branch>`).
|
|
91
95
|
- Runtime termination message: `control.kill` is the only documented actor message that kills a run. `control.stop` and `control.cancel` are actor-local mailbox vocabulary only when a recipe declares and handles them. Terminal retention messages: `control.archive`, `control.prune`.
|
|
92
96
|
- Long-lived child processes should remain in the run-owned process group unless the recipe implements an explicit daemon termination bridge; do not leave unowned detached services behind `control.kill`.
|
|
97
|
+
- Run controls revalidate a persisted cross-platform process identity proof before delivery or signaling. Treat `dead pid`, `owner mismatch`, and `unsupported proof` as distinct fail-closed states; never bypass them with direct pid signals.
|
|
93
98
|
|
|
94
99
|
Check `inspect view=mailbox` before domain-specific messages.
|
|
95
100
|
|
|
@@ -116,8 +121,8 @@ Views:
|
|
|
116
121
|
- Advanced `contacts`: roster-derived direct-message targets without full roster metadata.
|
|
117
122
|
- Advanced `previews`: TUI-ready bounded group-message previews with timestamp/from/to/type/summary/body_preview.
|
|
118
123
|
- `mailbox`: declared accepts/emits contract for runs; queued direct branch inbox messages for `branch:<run>/<branch>` with `id`, status, route/type, and queue/handling timestamps.
|
|
119
|
-
- `files`: run state
|
|
120
|
-
- `artifacts`: declared artifact paths/status.
|
|
124
|
+
- `files`: run state file summary plus a `lines`-bounded `review-evidence.json` manifest when present, including total/truncated command counts and stage capture paths.
|
|
125
|
+
- `artifacts`: declared artifact paths/status plus the same bounded owned review-evidence manifest when present.
|
|
121
126
|
- `recipes` target: registry summary for active, shadowed, invalid, disabled, and diagnostic recipe entries.
|
|
122
127
|
|
|
123
128
|
Actor inspector commands:
|
|
@@ -128,7 +133,7 @@ Actor inspector commands:
|
|
|
128
133
|
|
|
129
134
|
The table is compact and optimistic by default: bounded body previews, capped noisy room rows, branch-local inbox previews, stable event ids in selected-message details, and an inline roster summary in the form `name/role` that wraps only when needed. Use `unread` for queued branch inbox work and `branch <name>` / `current-branch <name>` for one branch's room/direct/inbox traffic. Rows with `metadata.requires_response=true` show a `!` attention marker. `/actors-inspect <number>` marks that row read for the current session filter. Active roster members use the target color; members that sent `actor.leave` stay visible as inactive/muted participants from the current run. Actor display names come from `actor.join` bodies (`display`) or branch addresses, keeping debugger output plain and name-driven.
|
|
130
135
|
|
|
131
|
-
Let terminal notifications arrive;
|
|
136
|
+
Let terminal notifications arrive. They queue through Pi's follow-up delivery mode, so a busy coordinator finishes its current work before receiving concurrently completed actor results; the host's `followUpMode` controls whether queued results arrive together or one at a time. When a deferred actor result gates the next step, wait for that terminal follow-up instead of scheduling continuation loops, repeatedly inspecting, or mutating the actor's reviewed scope. Idle coordinators still start a normal turn through `triggerTurn: true`. Inspect early only for an operator request, a meaningful actor event, or diagnosis of an overdue or stuck run.
|
|
132
137
|
|
|
133
138
|
## Runtime Communication Rules
|
|
134
139
|
|
|
@@ -154,11 +159,13 @@ Controls:
|
|
|
154
159
|
- `args`, `defaults`: public placeholder declarations and defaults.
|
|
155
160
|
- `parallel: true`: fanout child nodes.
|
|
156
161
|
- `when`: conditional execution.
|
|
162
|
+
- `accept_output: review_evidence`: fail closed unless the exact first non-whitespace stdout line is `ACTOR_REVIEW_RESULT`; marker prefixes are rejected and rejected stdout remains diagnostic evidence.
|
|
157
163
|
- `timeout`, `delay`, `retry`: timing and retry controls; string placeholders are allowed where supported.
|
|
158
164
|
- `failure`: `continue`, `branch`, or `root` propagation.
|
|
159
165
|
- `recover`: cleanup between retry attempts.
|
|
160
166
|
- `repeat`: repeated node expansion.
|
|
161
167
|
- `output`: output behavior selection.
|
|
168
|
+
- Command stdout/stderr use bounded tails plus complete spill files; tool/run diagnostics expose byte counts, truncation, and spill paths, while pipelines fail with `incomplete pipeline stdin` rather than consuming a partial tail.
|
|
162
169
|
|
|
163
170
|
Placeholders:
|
|
164
171
|
|
|
@@ -199,7 +206,7 @@ Rules:
|
|
|
199
206
|
7. Declare `mailbox` for actors that accept or emit meaningful messages.
|
|
200
207
|
8. Declare `artifacts` for durable outputs the coordinator should inspect.
|
|
201
208
|
9. File-backed recipe identity comes from the filename basename; legacy top-level `name` fields are ignored by loaders.
|
|
202
|
-
10. File-backed async recipes pass child `pi -p` actors a bounded JSONL recipe context bundle by default: raw entry/import recipe records, derived `name`, import path/alias, and `"you_are_here": true` on the launching recipe node. The runner
|
|
209
|
+
10. File-backed async recipes pass child `pi -p` actors a bounded JSONL recipe context bundle by default: raw entry/import recipe records, derived `name`, import path/alias, and `"you_are_here": true` on the launching recipe node. The runner collapses all natural-language positional fragments into one prompt under `prompts/command-NNN.md`, keeps intentional `@file` attachments separate, and invokes Pi with one authoritative prompt-file arg so large prompts and recipe context stay inspectable and argv-safe. Set `"actor_context": false` or `"off"` to suppress recipe context for minimal prompts.
|
|
203
210
|
11. Keep packaged recipes generic: no machine-local paths, no private companion identities, no project-specific defaults unless the recipe is explicitly project-specific.
|
|
204
211
|
12. Do not ship concrete model-version defaults in packaged recipes. For review-oriented subagent/lens recipes, default model/thinking args through `{current_model}` and `{current_thinking}` so they inherit the selected Pi session policy; keep `model`, `models`, `thinking`, and stage-specific model args explicit so callers can override policy at launch.
|
|
205
212
|
|
|
@@ -219,7 +226,7 @@ Muscle-memory lens: pi-actors has two durable executable-memory layers.
|
|
|
219
226
|
|
|
220
227
|
Agents grow active memory by calling `register_tool` or by deliberate recipe-file edits. They grow draft memory by trying ad hoc actors successfully. Treat both as executable habits: drafts are the workbench/proving ground; root recipes are promoted muscle memory.
|
|
221
228
|
|
|
222
|
-
Usage lens: user
|
|
229
|
+
Usage lens: user recipe launches update extension-maintained `.usage/<recipe-filename>.json` sidecars with fields such as `usage.calls` and `usage.last_called`; authored recipe files are not rewritten for telemetry. Discovery merges the sidecar into inspection. Agents should not hand-edit counters as part of normal recipe maintenance. Treat usage as evidence for usefulness analysis: heavily used recipes are good candidates for promotion, documentation, or stronger tests; unused recipes are cleanup candidates. Do not use failure counts as a primary usefulness signal because failures may reflect bad caller judgment rather than bad recipes. Do not delete or demote solely from counters without operator approval.
|
|
223
230
|
|
|
224
231
|
Promotion lens: successful transient/ad hoc actor runs are evidence, not commands. Inline spawns leave draft recipes as replayable evidence, not active tools. If a draft is repeatable, parameterized, safe enough, and likely useful later, the agent may promote it by moving/copying it into `~/.pi/agent/recipes` or by calling `register_tool` with a concise name, typed args/defaults, and a reviewed template or recipe path. Do not auto-register every success; do not promote temp paths, secrets, one-off prompts, or project-private assumptions without normalization and approval.
|
|
225
232
|
|
|
@@ -284,7 +291,7 @@ The user recipe root is the default tool set by location. It accepts canonical J
|
|
|
284
291
|
|
|
285
292
|
Use packaged recipes by name with `spawn file=<name>` for async actors, or register/call them as tools when repeated use deserves a stable shortcut.
|
|
286
293
|
|
|
287
|
-
Packaged review recipes are directly spawnable. Use `spawn file="pipeline-review-readiness" values={...}` for readiness review or `spawn file="subagent-review" values={...}` for one reviewer; pass model/thinking/tool policy through values, then inspect the run. Review coordinators preflight stage models before fanout; `ACTOR_PREFLIGHT_FAILED` diagnostics identify the failed stage, selected policy, provider error class, prompt file, and override args. Quorum-aware review fanout exposes `subagent_ttl_ms`, `reviewer_concurrency`, `min_successful_reviewers`, and `merge_policy`; partial reviewer evidence is preserved and marked `complete`, `degraded`, or `insufficient_data`. Run status/progress exposes `model_policy` so inherited vs explicit model/thinking choices remain visible. Do not recreate their script commands, call packaged scripts directly, or create wrapper recipes just to launch the maintained recipe.
|
|
294
|
+
Packaged review recipes are directly spawnable. Use `spawn file="pipeline-review-readiness" values={...}` for readiness review or `spawn file="subagent-review" values={...}` for one reviewer; pass model/thinking/tool policy through values, then inspect the run. Review coordinators preflight stage models before fanout; `ACTOR_PREFLIGHT_FAILED` diagnostics identify the failed stage, selected policy, provider error class, prompt file, and override args. Review stages require the `ACTOR_REVIEW_RESULT` evidence marker, so format acknowledgements and input requests fail closed before satisfying quorum or flowing downstream; rejected stdout remains in branch diagnostics. Quorum-aware review fanout exposes `subagent_ttl_ms`, `reviewer_concurrency`, `min_successful_reviewers`, and `merge_policy`; partial reviewer evidence is preserved and marked `complete`, `degraded`, or `insufficient_data`. Run status/progress exposes `model_policy` so inherited vs explicit model/thinking choices remain visible. Do not recreate their script commands, call packaged scripts directly, or create wrapper recipes just to launch the maintained recipe.
|
|
288
295
|
|
|
289
296
|
- [`pipeline-room-swarm`](../../recipes/pipeline-room-swarm.json): room-visible swarm coordination with roles, rounds, optional locker, artifact synthesis, and `subagent_ttl_ms` for hard participant budgets.
|
|
290
297
|
- [`pipeline-repo-health`](../../recipes/pipeline-repo-health.json): git/doc/validation evidence → normalized repository health report.
|
package/skills/swarm/SKILL.md
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: swarm
|
|
3
|
-
description: Subagent orchestration with scoped locks and quorum consensus. Use for
|
|
3
|
+
description: Subagent and actor orchestration with scoped locks, fanout, and quorum consensus. Use before launching multiple parallel actors or subagents for independent implementation, artifact generation, review, delegated audit, coordinated execution, or any workflow that needs autonomous coordinator decomposition and integration.
|
|
4
4
|
metadata:
|
|
5
|
-
version: 0.
|
|
5
|
+
version: 0.40.1
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# Swarm
|
|
@@ -13,6 +13,8 @@ Subagent orchestration: delegated review, quorum consensus, scoped locks, clean-
|
|
|
13
13
|
|
|
14
14
|
Run subagents safely and predictably through reusable orchestration contracts.
|
|
15
15
|
|
|
16
|
+
Activation rule: load this skill before launching multiple independent actors or subagents, even when the work is creative artifact generation rather than code review. The coordinator owns decomposition, disjoint scopes, launch correctness, result integration, and final validation; participants may choose local content or implementation details independently inside their assigned boundaries.
|
|
17
|
+
|
|
16
18
|
Swarm is independent. It must not require concrete sibling skill names, private repositories, local model aliases, or a specific tool registry layout. Local agents may bind the contracts to their own tools, command templates, model names, and review protocols.
|
|
17
19
|
|
|
18
20
|
Maintain this skill as a living orchestration standard. When real swarm work exposes better decomposition shapes, lock etiquette, quorum rules, checkpoint semantics, failure modes, or trade-offs, fold those lessons back here as durable guidance instead of leaving them only in one-off transcripts or backlog notes.
|