pi-crew 0.9.65 → 0.9.67
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +79 -0
- package/README.md +2 -3
- package/agents/executor.md +1 -1
- package/agents/test-engineer.md +1 -1
- package/agents/verifier.md +1 -1
- package/dist/index.mjs +4526 -4100
- package/package.json +7 -4
- package/skills/real-test-pi-crew/SKILL.md +10 -9
- package/src/config/config.ts +19 -3
- package/src/config/role-tools.ts +2 -2
- package/src/config/types.ts +2 -0
- package/src/extension/knowledge-injection.ts +17 -0
- package/src/extension/pi-api.ts +0 -16
- package/src/extension/team-tool/api/agent-control.ts +358 -0
- package/src/extension/team-tool/api/handler-context.ts +57 -0
- package/src/extension/team-tool/api/heartbeat.ts +75 -0
- package/src/extension/team-tool/api/mailbox.ts +242 -0
- package/src/extension/team-tool/api/plan-approval.ts +190 -0
- package/src/extension/team-tool/api/read.ts +443 -0
- package/src/extension/team-tool/api/task-claims.ts +207 -0
- package/src/extension/team-tool/api.ts +56 -1301
- package/src/extension/team-tool/cancel.ts +49 -4
- package/src/extension/team-tool/dispatch/manage.ts +7 -4
- package/src/extension/team-tool/explain.ts +3 -1
- package/src/extension/team-tool/goal-wrap.ts +2 -2
- package/src/extension/team-tool/goal.ts +4 -4
- package/src/extension/team-tool/lifecycle-actions.ts +9 -6
- package/src/extension/team-tool/parallel-dispatch.ts +2 -2
- package/src/extension/team-tool/respond.ts +11 -3
- package/src/extension/team-tool/run-intent.ts +357 -0
- package/src/extension/team-tool/run.ts +8 -285
- package/src/extension/team-tool/status.ts +56 -21
- package/src/extension/team-tool-types.ts +2 -0
- package/src/extension/team-tool.ts +10 -8
- package/src/prompt/scratchpad-lifecycle.ts +80 -5
- package/src/runtime/child-pi/child-pi-spawn.ts +13 -7
- package/src/runtime/child-pi/child-pi.ts +22 -144
- package/src/runtime/child-pi/mock-fixtures.ts +171 -0
- package/src/runtime/crew-agent-records.ts +18 -1
- package/src/runtime/output/output-validator.ts +34 -6
- package/src/runtime/scheduling/scheduler.ts +67 -19
- package/src/runtime/scratchpad/README.md +6 -0
- package/src/runtime/scratchpad/guest.ts +54 -5
- package/src/runtime/scratchpad/snapshot-hmac.ts +7 -1
- package/src/runtime/supervisor-contact.ts +0 -16
- package/src/runtime/task-runner/child-executor.ts +1 -0
- package/src/runtime/task-runner/state-helpers.ts +9 -1
- package/src/runtime/team-runner.ts +39 -2
- package/src/state/contracts.ts +3 -0
- package/src/state/event-log/event-log.ts +19 -27
- package/src/state/gitignore-manager.ts +61 -7
- package/src/utils/glob-match.ts +29 -0
- package/types/dwf.d.ts +1 -1
- package/src/types/new-api-types.ts +0 -35
|
@@ -32,6 +32,7 @@ import * as path from "node:path";
|
|
|
32
32
|
import { type Static, Type } from "@sinclair/typebox";
|
|
33
33
|
import { defineTool, type ExtensionAPI, type ToolDefinition } from "../extension/pi-api.ts";
|
|
34
34
|
import { EngineManager, type ExecuteResult } from "../runtime/scratchpad/engine.ts";
|
|
35
|
+
import { appendEventFireAndForget } from "../state/event-log/event-log.ts";
|
|
35
36
|
// D5/MAJOR-S1: PI_CREW_PARENT_PID + PI_CREW_GUEST build the guest's zombie-
|
|
36
37
|
// backstop env (PI_CREW_KIND_ENV is already exported above).
|
|
37
38
|
import { type ArtifactWriteOptions, writeArtifact } from "../state/stores/artifact-store.ts";
|
|
@@ -56,8 +57,43 @@ export const PI_CREW_SCRATCHPAD_RESTORE_ENV = "PI_CREW_SCRATCHPAD_RESTORE";
|
|
|
56
57
|
// an integrity/authn control (NIT-CA-1).
|
|
57
58
|
export const PI_CREW_SCRATCHPAD_RESTORE_MTIME_ENV = "PI_CREW_SCRATCHPAD_RESTORE_MTIME";
|
|
58
59
|
|
|
60
|
+
// I5 (plan): scratchpad adoption/value metric. The events path + run id are
|
|
61
|
+
// threaded from the team-runner via child-pi-spawn so the worker (where the
|
|
62
|
+
// execute handler runs) can append fire-and-forget metric events. Absent in
|
|
63
|
+
// non-team contexts (e.g. direct tests) → emission is silently skipped.
|
|
64
|
+
export const PI_CREW_EVENTS_PATH_ENV = "PI_CREW_EVENTS_PATH";
|
|
65
|
+
export const PI_CREW_RUN_ID_ENV = "PI_CREW_BROKER_RUN_ID";
|
|
66
|
+
|
|
59
67
|
/** Per-cell wall-clock bound (D9/Q2): the ONLY default anti-hang limit. */
|
|
60
68
|
export const EXECUTE_CELL_TIMEOUT_MS = 120_000;
|
|
69
|
+
|
|
70
|
+
// I5: fire-and-forget metric emission. Events are written ONLY when the worker
|
|
71
|
+
// was spawned by a team-runner that threads PI_CREW_EVENTS_PATH; in any other
|
|
72
|
+
// context (tests, standalone) the path is absent and we no-op. Fire-and-forget
|
|
73
|
+
// is mandatory — the cell hot path must never block on the event log (H1).
|
|
74
|
+
function emitScratchpadMetric(
|
|
75
|
+
type: "scratchpad.cell" | "scratchpad.restored",
|
|
76
|
+
env: NodeJS.ProcessEnv,
|
|
77
|
+
data: Record<string, unknown>,
|
|
78
|
+
appendEvent: typeof appendEventFireAndForget = appendEventFireAndForget,
|
|
79
|
+
): void {
|
|
80
|
+
const eventsPath = env[PI_CREW_EVENTS_PATH_ENV];
|
|
81
|
+
const runId = env[PI_CREW_RUN_ID_ENV];
|
|
82
|
+
if (!eventsPath || !runId) return;
|
|
83
|
+
// Fire-and-forget semantics (H1): a throwing writer (sync or async) must
|
|
84
|
+
// never break the cell. The real appendEventFireAndForget already catches
|
|
85
|
+
// async; this guard also absorbs a sync throw from a bad writer.
|
|
86
|
+
try {
|
|
87
|
+
appendEvent(eventsPath, {
|
|
88
|
+
type,
|
|
89
|
+
runId,
|
|
90
|
+
taskId: env[PI_CREW_TASK_ID_ENV],
|
|
91
|
+
data,
|
|
92
|
+
});
|
|
93
|
+
} catch {
|
|
94
|
+
// Metric is best-effort — drop the event, never the cell.
|
|
95
|
+
}
|
|
96
|
+
}
|
|
61
97
|
/** Debounce window for the post-cell snapshot (D5/F8). */
|
|
62
98
|
export const SNAPSHOT_DEBOUNCE_MS = 1500;
|
|
63
99
|
/** SEC-10: cap a single cell so a giant payload cannot OOM the esbuild
|
|
@@ -92,11 +128,12 @@ export interface ExecuteDetails {
|
|
|
92
128
|
export const SCRATCHPAD_DOCTRINE: string[] = [
|
|
93
129
|
"State compounds: variables persist across scratchpad calls in the task's persistent namespace. Don't re-derive what a previous cell already computed.",
|
|
94
130
|
"Write small cells and run many: the cell's result is the value of its final (trailing) expression.",
|
|
95
|
-
"The runtime is Node.js —
|
|
131
|
+
"The runtime is Node.js — shell commands go through the built-in sh(cmd, args[]) helper (it refuses null/empty arguments, so a missing variable can never leak into the command). Never interpolate variables into raw child_process/exec strings — use sh().",
|
|
96
132
|
"Writes are surgical; reads are full: read all the data you need, write the minimum.",
|
|
97
133
|
"Non-serializable variables (functions/classes) are reported in the snapshot's failed list — do not rely on them across calls.",
|
|
98
|
-
"
|
|
99
|
-
|
|
134
|
+
"A message starting [scratchpad] means the namespace was restored or reset — re-verify variables before using them, especially inside shell commands.",
|
|
135
|
+
// LAZY: the doctrine example string mentions await import('node:fs') as example TEXT — not a real dynamic import.
|
|
136
|
+
"Prefer scratchpad over bash when a later step reuses an earlier step's data. Example — cell 1: const raw = await import('node:fs').then(m => m.readFileSync('out.json','utf8')); const failures = JSON.parse(raw).tests.filter(t => !t.ok); failures.length — cell 2: failures.slice(0,3).map(t => t.name) // no re-read, no re-parse. For a single one-shot command, bash is cheaper — use it. Keep namespace values small — do not park large parsed objects across cells (they are V8-serialized into every snapshot).",
|
|
100
137
|
];
|
|
101
138
|
|
|
102
139
|
// ── singleton engine + debounce timer (per worker process) ─────────────────
|
|
@@ -285,6 +322,9 @@ export interface ScratchpadSnapshotDeps {
|
|
|
285
322
|
/** Injected for tests; production defaults to the real artifact-store writer.
|
|
286
323
|
* Return type is loose (unknown) — the flush ignores the descriptor. */
|
|
287
324
|
writeArtifact?: ScratchpadWriteArtifact;
|
|
325
|
+
/** I5: injected for tests (failing-writer path); production defaults to the
|
|
326
|
+
* real fire-and-forget event appender. Never awaited on the cell path. */
|
|
327
|
+
appendEvent?: typeof appendEventFireAndForget;
|
|
288
328
|
env?: NodeJS.ProcessEnv;
|
|
289
329
|
logInternalError?: typeof logInternalError;
|
|
290
330
|
}
|
|
@@ -435,10 +475,10 @@ export function createExecuteTool(engine: EngineManager, deps: Partial<Scratchpa
|
|
|
435
475
|
name: "scratchpad",
|
|
436
476
|
label: "Execute JavaScript",
|
|
437
477
|
description:
|
|
438
|
-
"
|
|
478
|
+
"Run JavaScript in the task's persistent namespace. State (variables assigned in earlier cells) persists across calls. The cell's result is the value of its final expression.",
|
|
439
479
|
parameters: ExecuteParams,
|
|
440
480
|
renderShell: "default",
|
|
441
|
-
promptSnippet: "scratchpad(code) —
|
|
481
|
+
promptSnippet: "scratchpad(code) — run JS in the task's persistent namespace",
|
|
442
482
|
promptGuidelines: SCRATCHPAD_DOCTRINE,
|
|
443
483
|
async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
|
|
444
484
|
const env = deps.env ?? process.env;
|
|
@@ -470,9 +510,30 @@ export function createExecuteTool(engine: EngineManager, deps: Partial<Scratchpa
|
|
|
470
510
|
restoreNotice = r
|
|
471
511
|
? `[scratchpad] restored ${r.restored.length} vars from attempt-${pending.attempt}; restored: [${truncateNameList(r.restored)}]; failed: [${truncateNameList(r.failed.map((f) => f.name))}]`
|
|
472
512
|
: "[scratchpad] snapshot restore: no state to restore (fail-open)";
|
|
513
|
+
// I5: metric — restoredCount/failedCount only (names already in the
|
|
514
|
+
// model-facing notice; never embed paths). Fire-and-forget.
|
|
515
|
+
emitScratchpadMetric(
|
|
516
|
+
"scratchpad.restored",
|
|
517
|
+
env,
|
|
518
|
+
{
|
|
519
|
+
status: r ? (r.restored.length > 0 ? "restored" : "empty") : "rejected",
|
|
520
|
+
attempt: pending.attempt,
|
|
521
|
+
restoredCount: r?.restored.length ?? 0,
|
|
522
|
+
failedCount: r?.failed.length ?? 0,
|
|
523
|
+
},
|
|
524
|
+
deps.appendEvent,
|
|
525
|
+
);
|
|
473
526
|
} catch (error) {
|
|
474
527
|
logInternalError("scratchpad.restore", error);
|
|
475
528
|
restoreNotice = "[scratchpad] snapshot restore failed; continuing with empty namespace";
|
|
529
|
+
// I5: a restore that THREW is still a restore attempt — record it so a
|
|
530
|
+
// corrupt/unreadable snapshot is visible to the metric, not silent.
|
|
531
|
+
emitScratchpadMetric(
|
|
532
|
+
"scratchpad.restored",
|
|
533
|
+
env,
|
|
534
|
+
{ status: "failed", attempt: pending.attempt, restoredCount: 0, failedCount: 0 },
|
|
535
|
+
deps.appendEvent,
|
|
536
|
+
);
|
|
476
537
|
}
|
|
477
538
|
}
|
|
478
539
|
// 3. Ping-before-execute (S-2/N2-2): skip on idle so the first cell of a
|
|
@@ -523,6 +584,20 @@ export function createExecuteTool(engine: EngineManager, deps: Partial<Scratchpa
|
|
|
523
584
|
if (result.status === "ok") {
|
|
524
585
|
scheduleScratchpadSnapshot({ ...deps, engine });
|
|
525
586
|
}
|
|
587
|
+
// I5: metric — exactly one scratchpad.cell per cell, fire-and-forget.
|
|
588
|
+
// resultBytes approximates the rendered output size (best-effort, never
|
|
589
|
+
// blocks the cell path).
|
|
590
|
+
emitScratchpadMetric(
|
|
591
|
+
"scratchpad.cell",
|
|
592
|
+
env,
|
|
593
|
+
{
|
|
594
|
+
status: result.status,
|
|
595
|
+
durationMs: result.durationMs,
|
|
596
|
+
codeLength: params.code.length,
|
|
597
|
+
resultBytes: result.result?.length ?? 0,
|
|
598
|
+
},
|
|
599
|
+
deps.appendEvent,
|
|
600
|
+
);
|
|
526
601
|
const text = restoreNotice ? `${restoreNotice}\n\n${renderExecuteResult(result)}` : renderExecuteResult(result);
|
|
527
602
|
return {
|
|
528
603
|
content: [{ type: "text", text }],
|
|
@@ -259,6 +259,12 @@ export function prepareSpawnContext(
|
|
|
259
259
|
});
|
|
260
260
|
// Pass steering file path to child for real-time steer injection
|
|
261
261
|
if (input.steeringFile) built.env.PI_CREW_STEERING_FILE = input.steeringFile;
|
|
262
|
+
// I5: run/task identity is threaded ALWAYS (not broker-gated) so the worker's
|
|
263
|
+
// scratchpad metric events (emitScratchpadMetric) can write runId/taskId even
|
|
264
|
+
// when the inter-pi broker is disabled. Control-namespace keys — pass
|
|
265
|
+
// assertOnlyControlEnvKeys.
|
|
266
|
+
if (input.runId) built.env.PI_CREW_BROKER_RUN_ID = input.runId;
|
|
267
|
+
if (input.agentId) built.env.PI_CREW_BROKER_TASK_ID = input.agentId;
|
|
262
268
|
// Phase 0 inter-pi broker: inject socket path + token (control-namespace keys,
|
|
263
269
|
// safe under assertOnlyControlEnvKeys). Only when the parent broker issued
|
|
264
270
|
// credentials for this run — i.e. the broker is enabled AND this run is
|
|
@@ -267,12 +273,7 @@ export function prepareSpawnContext(
|
|
|
267
273
|
if (input.brokerSpawn?.socketPath && input.brokerSpawn.token) {
|
|
268
274
|
built.env.PI_CREW_BROKER_SOCKET = input.brokerSpawn.socketPath;
|
|
269
275
|
built.env.PI_CREW_BROKER_TOKEN = input.brokerSpawn.token;
|
|
270
|
-
//
|
|
271
|
-
// (the token is validated against the runId; the taskId binds the
|
|
272
|
-
// connection for message routing). Both are control-namespace keys so
|
|
273
|
-
// they pass assertOnlyControlEnvKeys. agentId is the per-task id.
|
|
274
|
-
if (input.runId) built.env.PI_CREW_BROKER_RUN_ID = input.runId;
|
|
275
|
-
if (input.agentId) built.env.PI_CREW_BROKER_TASK_ID = input.agentId;
|
|
276
|
+
// runId/taskId are threaded unconditionally above (I5); nothing to add here.
|
|
276
277
|
}
|
|
277
278
|
// Phase 1 scratchpad: opt in the persistent Bun-free JS evaluator (execute
|
|
278
279
|
// tool) for this worker. Gated by role/agent (S-6 read-only roles never;
|
|
@@ -289,7 +290,12 @@ export function prepareSpawnContext(
|
|
|
289
290
|
// artifactsRoot after F4/S-1).
|
|
290
291
|
built.env.PI_CREW_ARTIFACTS_ROOT = input.artifactsRoot;
|
|
291
292
|
}
|
|
292
|
-
//
|
|
293
|
+
// I5: thread the run's events path so the worker's scratchpad execute
|
|
294
|
+
// handler can append fire-and-forget metric events (scratchpad.cell /
|
|
295
|
+
// scratchpad.restored). Optional — absent in non-team contexts.
|
|
296
|
+
if (input.eventsPath) {
|
|
297
|
+
built.env.PI_CREW_EVENTS_PATH = input.eventsPath;
|
|
298
|
+
} // F4/S-1: the RAW (unredacted) snapshot must NEVER land in artifactsRoot —
|
|
293
299
|
// point it at a temp dir; the worker reads it then writeArtifact()
|
|
294
300
|
// (redact+atomic) is the ONLY writer into artifactsRoot.
|
|
295
301
|
// R3-1: built.tempDir is only created by buildPiWorkerArgs when the agent
|
|
@@ -1,11 +1,8 @@
|
|
|
1
1
|
import { spawn } from "node:child_process";
|
|
2
2
|
import * as fs from "node:fs";
|
|
3
|
-
import * as os from "node:os";
|
|
4
|
-
import * as path from "node:path";
|
|
5
3
|
import type { AgentConfig } from "../../agents/agent-config.ts";
|
|
6
4
|
import { DEFAULT_CHILD_PI } from "../../config/defaults.ts";
|
|
7
5
|
import { registerChildProcess, unregisterChildProcess } from "../../extension/crew-cleanup.ts";
|
|
8
|
-
import { atomicWriteFile } from "../../state/atomic-write.ts";
|
|
9
6
|
import type { WorkerExitStatus } from "../../state/types.ts";
|
|
10
7
|
import { logInternalError } from "../../utils/internal-error.ts";
|
|
11
8
|
import { redactSecretString } from "../../utils/redaction.ts";
|
|
@@ -17,6 +14,7 @@ import { buildFinalChildPiSpawnOptions, prepareSpawnContext } from "./child-pi-s
|
|
|
17
14
|
import { ChildPiSteeringController } from "./child-pi-steering.ts";
|
|
18
15
|
// Internal helpers for active-child bookkeeping (extracted to child-pi-kill.ts).
|
|
19
16
|
import { ChildPiLineObserver } from "./child-pi-streams.ts";
|
|
17
|
+
import { runMockChildPi } from "./mock-fixtures.ts";
|
|
20
18
|
|
|
21
19
|
// ── Re-exports from child-pi-kill.ts (H-7 decomposition step 2) ──
|
|
22
20
|
// killProcessTree is internal (not previously exported) — keep that invariant.
|
|
@@ -147,6 +145,10 @@ export interface ChildPiRunInput {
|
|
|
147
145
|
thinkingOverride?: string;
|
|
148
146
|
/** Root directory for artifacts (used to validate transcriptPath). */
|
|
149
147
|
artifactsRoot?: string;
|
|
148
|
+
/** I5: run events JSONL path — threaded to the worker so its scratchpad
|
|
149
|
+
* execute handler can append fire-and-forget metric events. Optional;
|
|
150
|
+
* absent in non-team contexts (emission silently skipped). */
|
|
151
|
+
eventsPath?: string;
|
|
150
152
|
/** Phase 1 scratchpad: model-fallback attempt index (0-based) for per-attempt
|
|
151
153
|
* snapshot relativePath `scratchpad/<taskId>.attempt-<attempt>.snapshot.json`
|
|
152
154
|
* (C3). Optional — worker defaults to attempt 0 when unset (e.g. custom
|
|
@@ -253,146 +255,10 @@ export async function runChildPi(input: ChildPiRunInput): Promise<ChildPiRunResu
|
|
|
253
255
|
stdout: "",
|
|
254
256
|
stderr: `pi-crew depth guard blocked child worker: depth ${depth.depth} >= max ${depth.maxDepth}`,
|
|
255
257
|
};
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
// PI_CREW_ALLOW_MOCK is NOT in the allowlist — it is only checked in the
|
|
261
|
-
// parent process scope. This means:
|
|
262
|
-
// (1) If an attacker sets PI_CREW_ALLOW_MOCK in the parent's environment,
|
|
263
|
-
// it will NOT be passed to child processes (safe).
|
|
264
|
-
// (2) Mock mode activation in the child always fails the PI_CREW_ALLOW_MOCK
|
|
265
|
-
// check, so mock mode can only be triggered from the parent process.
|
|
266
|
-
// This asymmetry is intentional: PI_CREW_ALLOW_MOCK must be set in the Pi root
|
|
267
|
-
// process (the entry point that spawns children), not inherited from a parent.
|
|
268
|
-
// Setup hooks cannot inject PI_CREW_ALLOW_MOCK into the parent's env.
|
|
269
|
-
const allowMock = process.env.PI_CREW_ALLOW_MOCK === "1" || process.env.PI_CREW_ALLOW_MOCK === "true";
|
|
270
|
-
if (!allowMock) {
|
|
271
|
-
return {
|
|
272
|
-
exitCode: 1,
|
|
273
|
-
stdout: "",
|
|
274
|
-
stderr: "Mock mode requires PI_CREW_ALLOW_MOCK=1",
|
|
275
|
-
};
|
|
276
|
-
}
|
|
277
|
-
// SECURITY: Log mock mode activation prominently for audit trail
|
|
278
|
-
logInternalError("child-pi.mock", new Error(`Mock mode active: ${mock}`), "NOT running real agents");
|
|
279
|
-
if (mock === "success") {
|
|
280
|
-
const stdout = `[MOCK] Success for ${input.agent.name}\n`;
|
|
281
|
-
await observeStdoutChunk(input, stdout);
|
|
282
|
-
return { exitCode: 0, stdout, stderr: "" };
|
|
283
|
-
}
|
|
284
|
-
if (mock === "json-success" || mock === "adaptive-plan") {
|
|
285
|
-
const text =
|
|
286
|
-
mock === "adaptive-plan" && effectiveTask.includes("ADAPTIVE_PLAN_JSON_START")
|
|
287
|
-
? `[MOCK] Adaptive plan\nADAPTIVE_PLAN_JSON_START\n${JSON.stringify({
|
|
288
|
-
phases: [
|
|
289
|
-
{
|
|
290
|
-
name: "research",
|
|
291
|
-
tasks: [
|
|
292
|
-
{
|
|
293
|
-
role: "explorer",
|
|
294
|
-
task: "Explore adaptive target",
|
|
295
|
-
},
|
|
296
|
-
{
|
|
297
|
-
role: "analyst",
|
|
298
|
-
task: "Analyze adaptive target",
|
|
299
|
-
},
|
|
300
|
-
{
|
|
301
|
-
role: "planner",
|
|
302
|
-
task: "Plan adaptive target",
|
|
303
|
-
},
|
|
304
|
-
],
|
|
305
|
-
},
|
|
306
|
-
{
|
|
307
|
-
name: "build",
|
|
308
|
-
tasks: [
|
|
309
|
-
{
|
|
310
|
-
role: "executor",
|
|
311
|
-
task: "Implement adaptive target",
|
|
312
|
-
},
|
|
313
|
-
],
|
|
314
|
-
},
|
|
315
|
-
{
|
|
316
|
-
name: "check",
|
|
317
|
-
tasks: [
|
|
318
|
-
{
|
|
319
|
-
role: "reviewer",
|
|
320
|
-
task: "Review adaptive target",
|
|
321
|
-
},
|
|
322
|
-
{
|
|
323
|
-
role: "test-engineer",
|
|
324
|
-
task: "Test adaptive target",
|
|
325
|
-
},
|
|
326
|
-
{
|
|
327
|
-
role: "writer",
|
|
328
|
-
task: "Summarize adaptive target",
|
|
329
|
-
},
|
|
330
|
-
],
|
|
331
|
-
},
|
|
332
|
-
],
|
|
333
|
-
})}\nADAPTIVE_PLAN_JSON_END`
|
|
334
|
-
: `[MOCK] JSON success for ${input.agent.name}`;
|
|
335
|
-
const stdout = `${JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text }] } })}\n${JSON.stringify({ type: "message_end", usage: { input: 10, output: 5, cost: 0.001, turns: 1 } })}\n`;
|
|
336
|
-
await observeStdoutChunk(input, stdout);
|
|
337
|
-
return { exitCode: 0, stdout, stderr: "" };
|
|
338
|
-
}
|
|
339
|
-
if (mock === "retryable-failure")
|
|
340
|
-
return {
|
|
341
|
-
exitCode: 1,
|
|
342
|
-
stdout: "",
|
|
343
|
-
stderr: "[MOCK] rate limit: mock failure",
|
|
344
|
-
};
|
|
345
|
-
// E2E fallback-chain fixture: invocation #1 returns a SILENT retryable
|
|
346
|
-
// failure (exit code 0, no real assistant text, message_end carries a
|
|
347
|
-
// retryable-pattern errorMessage). Invocation #2+ delegates to the
|
|
348
|
-
// standard json-success shape. Counter lives in os.tmpdir() keyed by
|
|
349
|
-
// process.pid + mock name so concurrent test processes don't collide.
|
|
350
|
-
// The test cleans up the file in its finally block.
|
|
351
|
-
if (mock === "retryable-failure-then-success") {
|
|
352
|
-
const counterFile = path.join(os.tmpdir(), `pi-crew-mock-counter-${process.pid}-retryable-failure-then-success`);
|
|
353
|
-
let count = 0;
|
|
354
|
-
try {
|
|
355
|
-
const raw = fs.readFileSync(counterFile, "utf-8");
|
|
356
|
-
const parsed = Number.parseInt(raw.trim(), 10);
|
|
357
|
-
if (Number.isFinite(parsed) && parsed >= 0) count = parsed;
|
|
358
|
-
} catch {
|
|
359
|
-
// file missing or unreadable — first invocation in this process
|
|
360
|
-
}
|
|
361
|
-
count += 1;
|
|
362
|
-
try {
|
|
363
|
-
atomicWriteFile(counterFile, String(count));
|
|
364
|
-
} catch (error) {
|
|
365
|
-
logInternalError("child-pi.mock-counter-write", error as Error, `file=${counterFile}`);
|
|
366
|
-
}
|
|
367
|
-
if (count === 1) {
|
|
368
|
-
// Silent retryable failure: exit 0, no real text, message_end
|
|
369
|
-
// carries errorMessage matching `/provider[_ ]?error/i` so that
|
|
370
|
-
// `detectRetryableModelFailureFromOutput` surfaces it as an error
|
|
371
|
-
// and `isRetryableModelFailure` routes the next attempt to the
|
|
372
|
-
// next candidate model. `stopReason:"error"` (NOT "stop") so
|
|
373
|
-
// `isFinalAssistantEvent` does NOT prematurely terminate the run.
|
|
374
|
-
const failureEvent = {
|
|
375
|
-
type: "message_end",
|
|
376
|
-
message: {
|
|
377
|
-
role: "assistant",
|
|
378
|
-
content: [],
|
|
379
|
-
errorMessage: "Provider error: api_error",
|
|
380
|
-
stopReason: "error",
|
|
381
|
-
},
|
|
382
|
-
};
|
|
383
|
-
const stdout = `${JSON.stringify(failureEvent)}\n`;
|
|
384
|
-
await observeStdoutChunk(input, stdout);
|
|
385
|
-
return { exitCode: 0, stdout, stderr: "" };
|
|
386
|
-
}
|
|
387
|
-
// Subsequent invocations: delegate to json-success shape so the
|
|
388
|
-
// fallback chain's second attempt succeeds and the run completes.
|
|
389
|
-
const text = `[MOCK] JSON success for ${input.agent.name}`;
|
|
390
|
-
const stdout = `${JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text }] } })}\n${JSON.stringify({ type: "message_end", usage: { input: 10, output: 5, cost: 0.001, turns: 1 } })}\n`;
|
|
391
|
-
await observeStdoutChunk(input, stdout);
|
|
392
|
-
return { exitCode: 0, stdout, stderr: "" };
|
|
393
|
-
}
|
|
394
|
-
return { exitCode: 1, stdout: "", stderr: `[MOCK] failure: ${mock}` };
|
|
395
|
-
}
|
|
258
|
+
// H3 phase 3 (2026-08-10): mock-mode fixtures extracted to mock-fixtures.ts.
|
|
259
|
+
// Returns undefined when mock mode is NOT active → fall through to spawn.
|
|
260
|
+
const mockResult = await runMockChildPi(input, effectiveTask, observeStdoutChunk);
|
|
261
|
+
if (mockResult) return mockResult;
|
|
396
262
|
// H-7 step 6: spawn/env/args preparation extracted to child-pi-spawn.ts.
|
|
397
263
|
// prepareSpawnContext builds the worker args, attaches the steering file env,
|
|
398
264
|
// and handles the pre-spawn abort check (returns an immediate-abort result
|
|
@@ -406,7 +272,19 @@ export async function runChildPi(input: ChildPiRunInput): Promise<ChildPiRunResu
|
|
|
406
272
|
if (!brokerSpawn && brokerIssuer && input.runId) {
|
|
407
273
|
try {
|
|
408
274
|
brokerSpawn = await brokerIssuer(input.runId, input.agentId);
|
|
409
|
-
} catch {
|
|
275
|
+
} catch (error) {
|
|
276
|
+
// H8 (2026-08-10): surface the silent degradation. Previously this
|
|
277
|
+
// swallowed ALL issuer failures (token-rotation race, broker socket
|
|
278
|
+
// down, key fetch network error) with zero observability — the child
|
|
279
|
+
// spawned without broker credentials and the run just looked "slow".
|
|
280
|
+
// logInternalError writes to the internal-error channel (sampled +
|
|
281
|
+
// bounded); it does NOT propagate, so the child still runs without
|
|
282
|
+
// broker acceleration (the durable-first invariant is preserved).
|
|
283
|
+
logInternalError(
|
|
284
|
+
"child-pi.broker-issuer-failed",
|
|
285
|
+
error instanceof Error ? error : new Error(String(error)),
|
|
286
|
+
`runId=${input.runId} agentId=${input.agentId ?? "?"} — child will spawn without broker credentials`,
|
|
287
|
+
);
|
|
410
288
|
brokerSpawn = undefined;
|
|
411
289
|
}
|
|
412
290
|
}
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Mock-mode fixture runner for child Pi workers (H3 phase 3).
|
|
3
|
+
*
|
|
4
|
+
* Extracted from `runChildPi` (src/runtime/child-pi/child-pi.ts) on
|
|
5
|
+
* 2026-08-10 — the mock branch was ~140 self-contained lines embedded in the
|
|
6
|
+
* 840-line spawn orchestrator. Behaviour is byte-identical.
|
|
7
|
+
*
|
|
8
|
+
* Security model (unchanged): PI_TEAMS_MOCK_CHILD_PI is in the env allowlist
|
|
9
|
+
* (passed to children) but PI_CREW_ALLOW_MOCK is NOT — mock mode can only be
|
|
10
|
+
* activated from the parent process scope, never inherited by a child.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import * as fs from "node:fs";
|
|
14
|
+
import * as os from "node:os";
|
|
15
|
+
import * as path from "node:path";
|
|
16
|
+
import { atomicWriteFile } from "../../state/atomic-write.ts";
|
|
17
|
+
import { logInternalError } from "../../utils/internal-error.ts";
|
|
18
|
+
import type { ChildPiRunInput, ChildPiRunResult } from "./child-pi.ts";
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Run a mock child if mock mode is active (PI_TEAMS_MOCK_CHILD_PI set).
|
|
22
|
+
* Returns `undefined` when mock mode is NOT active — the caller falls through
|
|
23
|
+
* to the real spawn path.
|
|
24
|
+
*
|
|
25
|
+
* @param input The child run input (agent, env…).
|
|
26
|
+
* @param effectiveTask The task text after inherit-context prepend.
|
|
27
|
+
* @param observe Callback that feeds a stdout chunk through the line
|
|
28
|
+
* observer (owned by child-pi.ts).
|
|
29
|
+
*/
|
|
30
|
+
export async function runMockChildPi(
|
|
31
|
+
input: ChildPiRunInput,
|
|
32
|
+
effectiveTask: string,
|
|
33
|
+
observe: (input: ChildPiRunInput, text: string) => Promise<void>,
|
|
34
|
+
): Promise<ChildPiRunResult | undefined> {
|
|
35
|
+
const mock = process.env.PI_TEAMS_MOCK_CHILD_PI;
|
|
36
|
+
if (!mock) return undefined;
|
|
37
|
+
|
|
38
|
+
// SECURITY (Issue #2): see module docstring — PI_CREW_ALLOW_MOCK is only
|
|
39
|
+
// checked in the parent process scope; it is never passed to children.
|
|
40
|
+
const allowMock = process.env.PI_CREW_ALLOW_MOCK === "1" || process.env.PI_CREW_ALLOW_MOCK === "true";
|
|
41
|
+
if (!allowMock) {
|
|
42
|
+
return {
|
|
43
|
+
exitCode: 1,
|
|
44
|
+
stdout: "",
|
|
45
|
+
stderr: "Mock mode requires PI_CREW_ALLOW_MOCK=1",
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
// SECURITY: Log mock mode activation prominently for audit trail
|
|
49
|
+
logInternalError("child-pi.mock", new Error(`Mock mode active: ${mock}`), "NOT running real agents");
|
|
50
|
+
|
|
51
|
+
if (mock === "success") {
|
|
52
|
+
const stdout = `[MOCK] Success for ${input.agent.name}\n`;
|
|
53
|
+
await observe(input, stdout);
|
|
54
|
+
return { exitCode: 0, stdout, stderr: "" };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
if (mock === "json-success" || mock === "adaptive-plan") {
|
|
58
|
+
const text =
|
|
59
|
+
mock === "adaptive-plan" && effectiveTask.includes("ADAPTIVE_PLAN_JSON_START")
|
|
60
|
+
? `[MOCK] Adaptive plan\nADAPTIVE_PLAN_JSON_START\n${JSON.stringify({
|
|
61
|
+
phases: [
|
|
62
|
+
{
|
|
63
|
+
name: "research",
|
|
64
|
+
tasks: [
|
|
65
|
+
{
|
|
66
|
+
role: "explorer",
|
|
67
|
+
task: "Explore adaptive target",
|
|
68
|
+
},
|
|
69
|
+
{
|
|
70
|
+
role: "analyst",
|
|
71
|
+
task: "Analyze adaptive target",
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
role: "planner",
|
|
75
|
+
task: "Plan adaptive target",
|
|
76
|
+
},
|
|
77
|
+
],
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
name: "build",
|
|
81
|
+
tasks: [
|
|
82
|
+
{
|
|
83
|
+
role: "executor",
|
|
84
|
+
task: "Implement adaptive target",
|
|
85
|
+
},
|
|
86
|
+
],
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
name: "check",
|
|
90
|
+
tasks: [
|
|
91
|
+
{
|
|
92
|
+
role: "reviewer",
|
|
93
|
+
task: "Review adaptive target",
|
|
94
|
+
},
|
|
95
|
+
{
|
|
96
|
+
role: "test-engineer",
|
|
97
|
+
task: "Test adaptive target",
|
|
98
|
+
},
|
|
99
|
+
{
|
|
100
|
+
role: "writer",
|
|
101
|
+
task: "Summarize adaptive target",
|
|
102
|
+
},
|
|
103
|
+
],
|
|
104
|
+
},
|
|
105
|
+
],
|
|
106
|
+
})}\nADAPTIVE_PLAN_JSON_END`
|
|
107
|
+
: `[MOCK] JSON success for ${input.agent.name}`;
|
|
108
|
+
const stdout = `${JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text }] } })}\n${JSON.stringify({ type: "message_end", usage: { input: 10, output: 5, cost: 0.001, turns: 1 } })}\n`;
|
|
109
|
+
await observe(input, stdout);
|
|
110
|
+
return { exitCode: 0, stdout, stderr: "" };
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
if (mock === "retryable-failure")
|
|
114
|
+
return {
|
|
115
|
+
exitCode: 1,
|
|
116
|
+
stdout: "",
|
|
117
|
+
stderr: "[MOCK] rate limit: mock failure",
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
// E2E fallback-chain fixture: invocation #1 returns a SILENT retryable
|
|
121
|
+
// failure (exit code 0, no real assistant text, message_end carries a
|
|
122
|
+
// retryable-pattern errorMessage). Invocation #2+ delegates to the
|
|
123
|
+
// standard json-success shape. Counter lives in os.tmpdir() keyed by
|
|
124
|
+
// process.pid + mock name so concurrent test processes don't collide.
|
|
125
|
+
// The test cleans up the file in its finally block.
|
|
126
|
+
if (mock === "retryable-failure-then-success") {
|
|
127
|
+
const counterFile = path.join(os.tmpdir(), `pi-crew-mock-counter-${process.pid}-retryable-failure-then-success`);
|
|
128
|
+
let count = 0;
|
|
129
|
+
try {
|
|
130
|
+
const raw = fs.readFileSync(counterFile, "utf-8");
|
|
131
|
+
const parsed = Number.parseInt(raw.trim(), 10);
|
|
132
|
+
if (Number.isFinite(parsed) && parsed >= 0) count = parsed;
|
|
133
|
+
} catch {
|
|
134
|
+
// file missing or unreadable — first invocation in this process
|
|
135
|
+
}
|
|
136
|
+
count += 1;
|
|
137
|
+
try {
|
|
138
|
+
atomicWriteFile(counterFile, String(count));
|
|
139
|
+
} catch (error) {
|
|
140
|
+
logInternalError("child-pi.mock-counter-write", error as Error, `file=${counterFile}`);
|
|
141
|
+
}
|
|
142
|
+
if (count === 1) {
|
|
143
|
+
// Silent retryable failure: exit 0, no real text, message_end
|
|
144
|
+
// carries errorMessage matching `/provider[_ ]?error/i` so that
|
|
145
|
+
// `detectRetryableModelFailureFromOutput` surfaces it as an error
|
|
146
|
+
// and `isRetryableModelFailure` routes the next attempt to the
|
|
147
|
+
// next candidate model. `stopReason:"error"` (NOT "stop") so
|
|
148
|
+
// `isFinalAssistantEvent` does NOT prematurely terminate the run.
|
|
149
|
+
const failureEvent = {
|
|
150
|
+
type: "message_end",
|
|
151
|
+
message: {
|
|
152
|
+
role: "assistant",
|
|
153
|
+
content: [],
|
|
154
|
+
errorMessage: "Provider error: api_error",
|
|
155
|
+
stopReason: "error",
|
|
156
|
+
},
|
|
157
|
+
};
|
|
158
|
+
const stdout = `${JSON.stringify(failureEvent)}\n`;
|
|
159
|
+
await observe(input, stdout);
|
|
160
|
+
return { exitCode: 0, stdout, stderr: "" };
|
|
161
|
+
}
|
|
162
|
+
// Subsequent invocations: delegate to json-success shape so the
|
|
163
|
+
// fallback chain's second attempt succeeds and the run completes.
|
|
164
|
+
const text = `[MOCK] JSON success for ${input.agent.name}`;
|
|
165
|
+
const stdout = `${JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text }] } })}\n${JSON.stringify({ type: "message_end", usage: { input: 10, output: 5, cost: 0.001, turns: 1 } })}\n`;
|
|
166
|
+
await observe(input, stdout);
|
|
167
|
+
return { exitCode: 0, stdout, stderr: "" };
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
return { exitCode: 1, stdout: "", stderr: `[MOCK] failure: ${mock}` };
|
|
171
|
+
}
|
|
@@ -299,9 +299,26 @@ export function saveCrewAgents(manifest: TeamRunManifest, records: CrewAgentReco
|
|
|
299
299
|
withAgentsLock(manifest, () => {
|
|
300
300
|
fs.mkdirSync(manifest.stateRoot, { recursive: true });
|
|
301
301
|
const filePath = agentsPath(manifest);
|
|
302
|
+
// Index file (agents.json) is the authoritative record — always full
|
|
303
|
+
// durability.
|
|
302
304
|
atomicWriteJson(filePath, redactSecrets(records));
|
|
303
305
|
asyncAgentReaderCache.delete(filePath);
|
|
304
|
-
for (const record of records)
|
|
306
|
+
for (const record of records) {
|
|
307
|
+
// H2 (2026-08-10): per-task status.json is a DENORMALIZED read
|
|
308
|
+
// optimization for the dashboard/notifier, not an authoritative
|
|
309
|
+
// record — agents.json + events.jsonl cover crash recovery.
|
|
310
|
+
// Previously EVERY record (including running/queued progress
|
|
311
|
+
// snapshots) got a full fsync: N+1 fsyncs per saveCrewAgents call
|
|
312
|
+
// (50-task team ≈ 750ms blocking). Only TERMINAL records keep
|
|
313
|
+
// full durability (F4: notifier/dashboard must see the final
|
|
314
|
+
// state immediately); non-terminal per-task status is best-effort
|
|
315
|
+
// coalesced like the upsertCrewAgent non-terminal path.
|
|
316
|
+
if (TERMINAL_AGENT_STATUSES.has(record.status ?? "")) {
|
|
317
|
+
writeCrewAgentStatus(manifest, record);
|
|
318
|
+
} else {
|
|
319
|
+
writeCrewAgentStatusCoalesced(manifest, record);
|
|
320
|
+
}
|
|
321
|
+
}
|
|
305
322
|
});
|
|
306
323
|
}
|
|
307
324
|
|
|
@@ -9,13 +9,41 @@
|
|
|
9
9
|
* (headings, code blocks, URLs) after compression.
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
|
-
/**
|
|
12
|
+
/**
|
|
13
|
+
* Why relax: real worker LLMs emit markdown handoffs (`## Handoff`,
|
|
14
|
+
* `### Summary`, `## Follow-ups`, `- bullet`, `**bold**`) instead of the
|
|
15
|
+
* caveman formats (`file:line — text`, `PASS:`/`FAIL:`, emoji findings)
|
|
16
|
+
* these patterns originally required. Each role pattern below ORs its
|
|
17
|
+
* strict contract with a markdown-structured alternation so structured
|
|
18
|
+
* handoffs validate while empty/garbage output still fails.
|
|
19
|
+
*
|
|
20
|
+
* What changed: added the MARKDOWN_STRUCTURED alternation to every role.
|
|
21
|
+
* What is preserved: the strict patterns still match verbatim, and the
|
|
22
|
+
* structural-preservation checks (code blocks, URLs, headings) in
|
|
23
|
+
* validateWorkerOutput are UNCHANGED.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
/** Accepts atx headings (`## X`), bold (`**x**`), bullets (`- x` / `* x`), and numbered lists (`1. x`) */
|
|
27
|
+
const MARKDOWN_STRUCTURED = /^(?:#{1,6}\s|\*\*|[-*]\s|\d+\.\s)/m;
|
|
28
|
+
|
|
29
|
+
/** Strict per-role contract patterns (kept verbatim; `.source` is embedded in the alternations below) */
|
|
30
|
+
const STRICT_ROLE_PATTERNS: Record<string, RegExp> = {
|
|
31
|
+
explorer: /^(\S+:\d+|Defs:|Refs:|Callers:|Tests:|Sites:|No match\.|totals:)/m,
|
|
32
|
+
executor: /^(\S+:\d+(-\d+)? — .{1,80}\.|verified:|too-big\.|needs-confirm\.|ambiguous\.|regressed\.)/m,
|
|
33
|
+
reviewer: /^([^:\s]+:\d+:\s+\p{Emoji_Presentation}|No issues\.|totals:)/mu,
|
|
34
|
+
"security-reviewer": /^([^:\s]+:\d+:\s+\p{Emoji_Presentation}|No issues\.|totals:)/mu,
|
|
35
|
+
verifier: /^(PASS:|FAIL:)/m,
|
|
36
|
+
};
|
|
37
|
+
|
|
38
|
+
/** Role-specific output format patterns — constructed fresh per call to avoid /g lastIndex leak.
|
|
39
|
+
* Each factory: strict contract OR markdown-structured alternation. */
|
|
13
40
|
const ROLE_PATTERN_DEFS: Record<string, () => RegExp> = {
|
|
14
|
-
explorer: () =>
|
|
15
|
-
executor: () =>
|
|
16
|
-
reviewer: () =>
|
|
17
|
-
"security-reviewer": () =>
|
|
18
|
-
|
|
41
|
+
explorer: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.explorer.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"),
|
|
42
|
+
executor: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.executor.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"),
|
|
43
|
+
reviewer: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.reviewer.source})|(?:${MARKDOWN_STRUCTURED.source})`, "mu"),
|
|
44
|
+
"security-reviewer": () =>
|
|
45
|
+
new RegExp(`(?:${STRICT_ROLE_PATTERNS["security-reviewer"].source})|(?:${MARKDOWN_STRUCTURED.source})`, "mu"),
|
|
46
|
+
verifier: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.verifier.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"),
|
|
19
47
|
};
|
|
20
48
|
|
|
21
49
|
/** Fresh RegExp factories for structural preservation checks (avoids /g lastIndex leak) */
|