pi-crew 0.9.65 → 0.9.67

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/CHANGELOG.md +79 -0
  2. package/README.md +2 -3
  3. package/agents/executor.md +1 -1
  4. package/agents/test-engineer.md +1 -1
  5. package/agents/verifier.md +1 -1
  6. package/dist/index.mjs +4526 -4100
  7. package/package.json +7 -4
  8. package/skills/real-test-pi-crew/SKILL.md +10 -9
  9. package/src/config/config.ts +19 -3
  10. package/src/config/role-tools.ts +2 -2
  11. package/src/config/types.ts +2 -0
  12. package/src/extension/knowledge-injection.ts +17 -0
  13. package/src/extension/pi-api.ts +0 -16
  14. package/src/extension/team-tool/api/agent-control.ts +358 -0
  15. package/src/extension/team-tool/api/handler-context.ts +57 -0
  16. package/src/extension/team-tool/api/heartbeat.ts +75 -0
  17. package/src/extension/team-tool/api/mailbox.ts +242 -0
  18. package/src/extension/team-tool/api/plan-approval.ts +190 -0
  19. package/src/extension/team-tool/api/read.ts +443 -0
  20. package/src/extension/team-tool/api/task-claims.ts +207 -0
  21. package/src/extension/team-tool/api.ts +56 -1301
  22. package/src/extension/team-tool/cancel.ts +49 -4
  23. package/src/extension/team-tool/dispatch/manage.ts +7 -4
  24. package/src/extension/team-tool/explain.ts +3 -1
  25. package/src/extension/team-tool/goal-wrap.ts +2 -2
  26. package/src/extension/team-tool/goal.ts +4 -4
  27. package/src/extension/team-tool/lifecycle-actions.ts +9 -6
  28. package/src/extension/team-tool/parallel-dispatch.ts +2 -2
  29. package/src/extension/team-tool/respond.ts +11 -3
  30. package/src/extension/team-tool/run-intent.ts +357 -0
  31. package/src/extension/team-tool/run.ts +8 -285
  32. package/src/extension/team-tool/status.ts +56 -21
  33. package/src/extension/team-tool-types.ts +2 -0
  34. package/src/extension/team-tool.ts +10 -8
  35. package/src/prompt/scratchpad-lifecycle.ts +80 -5
  36. package/src/runtime/child-pi/child-pi-spawn.ts +13 -7
  37. package/src/runtime/child-pi/child-pi.ts +22 -144
  38. package/src/runtime/child-pi/mock-fixtures.ts +171 -0
  39. package/src/runtime/crew-agent-records.ts +18 -1
  40. package/src/runtime/output/output-validator.ts +34 -6
  41. package/src/runtime/scheduling/scheduler.ts +67 -19
  42. package/src/runtime/scratchpad/README.md +6 -0
  43. package/src/runtime/scratchpad/guest.ts +54 -5
  44. package/src/runtime/scratchpad/snapshot-hmac.ts +7 -1
  45. package/src/runtime/supervisor-contact.ts +0 -16
  46. package/src/runtime/task-runner/child-executor.ts +1 -0
  47. package/src/runtime/task-runner/state-helpers.ts +9 -1
  48. package/src/runtime/team-runner.ts +39 -2
  49. package/src/state/contracts.ts +3 -0
  50. package/src/state/event-log/event-log.ts +19 -27
  51. package/src/state/gitignore-manager.ts +61 -7
  52. package/src/utils/glob-match.ts +29 -0
  53. package/types/dwf.d.ts +1 -1
  54. package/src/types/new-api-types.ts +0 -35
@@ -32,6 +32,7 @@ import * as path from "node:path";
32
32
  import { type Static, Type } from "@sinclair/typebox";
33
33
  import { defineTool, type ExtensionAPI, type ToolDefinition } from "../extension/pi-api.ts";
34
34
  import { EngineManager, type ExecuteResult } from "../runtime/scratchpad/engine.ts";
35
+ import { appendEventFireAndForget } from "../state/event-log/event-log.ts";
35
36
  // D5/MAJOR-S1: PI_CREW_PARENT_PID + PI_CREW_GUEST build the guest's zombie-
36
37
  // backstop env (PI_CREW_KIND_ENV is already exported above).
37
38
  import { type ArtifactWriteOptions, writeArtifact } from "../state/stores/artifact-store.ts";
@@ -56,8 +57,43 @@ export const PI_CREW_SCRATCHPAD_RESTORE_ENV = "PI_CREW_SCRATCHPAD_RESTORE";
56
57
  // an integrity/authn control (NIT-CA-1).
57
58
  export const PI_CREW_SCRATCHPAD_RESTORE_MTIME_ENV = "PI_CREW_SCRATCHPAD_RESTORE_MTIME";
58
59
 
60
+ // I5 (plan): scratchpad adoption/value metric. The events path + run id are
61
+ // threaded from the team-runner via child-pi-spawn so the worker (where the
62
+ // execute handler runs) can append fire-and-forget metric events. Absent in
63
+ // non-team contexts (e.g. direct tests) → emission is silently skipped.
64
+ export const PI_CREW_EVENTS_PATH_ENV = "PI_CREW_EVENTS_PATH";
65
+ export const PI_CREW_RUN_ID_ENV = "PI_CREW_BROKER_RUN_ID";
66
+
59
67
  /** Per-cell wall-clock bound (D9/Q2): the ONLY default anti-hang limit. */
60
68
  export const EXECUTE_CELL_TIMEOUT_MS = 120_000;
69
+
70
+ // I5: fire-and-forget metric emission. Events are written ONLY when the worker
71
+ // was spawned by a team-runner that threads PI_CREW_EVENTS_PATH; in any other
72
+ // context (tests, standalone) the path is absent and we no-op. Fire-and-forget
73
+ // is mandatory — the cell hot path must never block on the event log (H1).
74
+ function emitScratchpadMetric(
75
+ type: "scratchpad.cell" | "scratchpad.restored",
76
+ env: NodeJS.ProcessEnv,
77
+ data: Record<string, unknown>,
78
+ appendEvent: typeof appendEventFireAndForget = appendEventFireAndForget,
79
+ ): void {
80
+ const eventsPath = env[PI_CREW_EVENTS_PATH_ENV];
81
+ const runId = env[PI_CREW_RUN_ID_ENV];
82
+ if (!eventsPath || !runId) return;
83
+ // Fire-and-forget semantics (H1): a throwing writer (sync or async) must
84
+ // never break the cell. The real appendEventFireAndForget already catches
85
+ // async; this guard also absorbs a sync throw from a bad writer.
86
+ try {
87
+ appendEvent(eventsPath, {
88
+ type,
89
+ runId,
90
+ taskId: env[PI_CREW_TASK_ID_ENV],
91
+ data,
92
+ });
93
+ } catch {
94
+ // Metric is best-effort — drop the event, never the cell.
95
+ }
96
+ }
61
97
  /** Debounce window for the post-cell snapshot (D5/F8). */
62
98
  export const SNAPSHOT_DEBOUNCE_MS = 1500;
63
99
  /** SEC-10: cap a single cell so a giant payload cannot OOM the esbuild
@@ -92,11 +128,12 @@ export interface ExecuteDetails {
92
128
  export const SCRATCHPAD_DOCTRINE: string[] = [
93
129
  "State compounds: variables persist across scratchpad calls in the task's persistent namespace. Don't re-derive what a previous cell already computed.",
94
130
  "Write small cells and run many: the cell's result is the value of its final (trailing) expression.",
95
- "The runtime is Node.js — use child_process for shell commands; there is no Bun.",
131
+ "The runtime is Node.js — shell commands go through the built-in sh(cmd, args[]) helper (it refuses null/empty arguments, so a missing variable can never leak into the command). Never interpolate variables into raw child_process/exec strings — use sh().",
96
132
  "Writes are surgical; reads are full: read all the data you need, write the minimum.",
97
133
  "Non-serializable variables (functions/classes) are reported in the snapshot's failed list — do not rely on them across calls.",
98
- "If you see <rlm_engine_reset> or a snapshot-restore notice, re-verify variables before use.",
99
- "Tool calls inside scratchpad are await expressions — await the result before the next cell.",
134
+ "A message starting [scratchpad] means the namespace was restored or reset — re-verify variables before using them, especially inside shell commands.",
135
+ // LAZY: the doctrine example string mentions await import('node:fs') as example TEXT — not a real dynamic import.
136
+ "Prefer scratchpad over bash when a later step reuses an earlier step's data. Example — cell 1: const raw = await import('node:fs').then(m => m.readFileSync('out.json','utf8')); const failures = JSON.parse(raw).tests.filter(t => !t.ok); failures.length — cell 2: failures.slice(0,3).map(t => t.name) // no re-read, no re-parse. For a single one-shot command, bash is cheaper — use it. Keep namespace values small — do not park large parsed objects across cells (they are V8-serialized into every snapshot).",
100
137
  ];
101
138
 
102
139
  // ── singleton engine + debounce timer (per worker process) ─────────────────
@@ -285,6 +322,9 @@ export interface ScratchpadSnapshotDeps {
285
322
  /** Injected for tests; production defaults to the real artifact-store writer.
286
323
  * Return type is loose (unknown) — the flush ignores the descriptor. */
287
324
  writeArtifact?: ScratchpadWriteArtifact;
325
+ /** I5: injected for tests (failing-writer path); production defaults to the
326
+ * real fire-and-forget event appender. Never awaited on the cell path. */
327
+ appendEvent?: typeof appendEventFireAndForget;
288
328
  env?: NodeJS.ProcessEnv;
289
329
  logInternalError?: typeof logInternalError;
290
330
  }
@@ -435,10 +475,10 @@ export function createExecuteTool(engine: EngineManager, deps: Partial<Scratchpa
435
475
  name: "scratchpad",
436
476
  label: "Execute JavaScript",
437
477
  description:
438
- "Chạy JavaScript trong namespace bền vững của task. State (biến gán ở cell trước) tồn tại qua các lần gọi. Kết quả cell = giá trị biểu thức cuối.",
478
+ "Run JavaScript in the task's persistent namespace. State (variables assigned in earlier cells) persists across calls. The cell's result is the value of its final expression.",
439
479
  parameters: ExecuteParams,
440
480
  renderShell: "default",
441
- promptSnippet: "scratchpad(code) — chạy JS trong namespace bền vững của task",
481
+ promptSnippet: "scratchpad(code) — run JS in the task's persistent namespace",
442
482
  promptGuidelines: SCRATCHPAD_DOCTRINE,
443
483
  async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
444
484
  const env = deps.env ?? process.env;
@@ -470,9 +510,30 @@ export function createExecuteTool(engine: EngineManager, deps: Partial<Scratchpa
470
510
  restoreNotice = r
471
511
  ? `[scratchpad] restored ${r.restored.length} vars from attempt-${pending.attempt}; restored: [${truncateNameList(r.restored)}]; failed: [${truncateNameList(r.failed.map((f) => f.name))}]`
472
512
  : "[scratchpad] snapshot restore: no state to restore (fail-open)";
513
+ // I5: metric — restoredCount/failedCount only (names already in the
514
+ // model-facing notice; never embed paths). Fire-and-forget.
515
+ emitScratchpadMetric(
516
+ "scratchpad.restored",
517
+ env,
518
+ {
519
+ status: r ? (r.restored.length > 0 ? "restored" : "empty") : "rejected",
520
+ attempt: pending.attempt,
521
+ restoredCount: r?.restored.length ?? 0,
522
+ failedCount: r?.failed.length ?? 0,
523
+ },
524
+ deps.appendEvent,
525
+ );
473
526
  } catch (error) {
474
527
  logInternalError("scratchpad.restore", error);
475
528
  restoreNotice = "[scratchpad] snapshot restore failed; continuing with empty namespace";
529
+ // I5: a restore that THREW is still a restore attempt — record it so a
530
+ // corrupt/unreadable snapshot is visible to the metric, not silent.
531
+ emitScratchpadMetric(
532
+ "scratchpad.restored",
533
+ env,
534
+ { status: "failed", attempt: pending.attempt, restoredCount: 0, failedCount: 0 },
535
+ deps.appendEvent,
536
+ );
476
537
  }
477
538
  }
478
539
  // 3. Ping-before-execute (S-2/N2-2): skip on idle so the first cell of a
@@ -523,6 +584,20 @@ export function createExecuteTool(engine: EngineManager, deps: Partial<Scratchpa
523
584
  if (result.status === "ok") {
524
585
  scheduleScratchpadSnapshot({ ...deps, engine });
525
586
  }
587
+ // I5: metric — exactly one scratchpad.cell per cell, fire-and-forget.
588
+ // resultBytes approximates the rendered output size (best-effort, never
589
+ // blocks the cell path).
590
+ emitScratchpadMetric(
591
+ "scratchpad.cell",
592
+ env,
593
+ {
594
+ status: result.status,
595
+ durationMs: result.durationMs,
596
+ codeLength: params.code.length,
597
+ resultBytes: result.result?.length ?? 0,
598
+ },
599
+ deps.appendEvent,
600
+ );
526
601
  const text = restoreNotice ? `${restoreNotice}\n\n${renderExecuteResult(result)}` : renderExecuteResult(result);
527
602
  return {
528
603
  content: [{ type: "text", text }],
@@ -259,6 +259,12 @@ export function prepareSpawnContext(
259
259
  });
260
260
  // Pass steering file path to child for real-time steer injection
261
261
  if (input.steeringFile) built.env.PI_CREW_STEERING_FILE = input.steeringFile;
262
+ // I5: run/task identity is threaded ALWAYS (not broker-gated) so the worker's
263
+ // scratchpad metric events (emitScratchpadMetric) can write runId/taskId even
264
+ // when the inter-pi broker is disabled. Control-namespace keys — pass
265
+ // assertOnlyControlEnvKeys.
266
+ if (input.runId) built.env.PI_CREW_BROKER_RUN_ID = input.runId;
267
+ if (input.agentId) built.env.PI_CREW_BROKER_TASK_ID = input.agentId;
262
268
  // Phase 0 inter-pi broker: inject socket path + token (control-namespace keys,
263
269
  // safe under assertOnlyControlEnvKeys). Only when the parent broker issued
264
270
  // credentials for this run — i.e. the broker is enabled AND this run is
@@ -267,12 +273,7 @@ export function prepareSpawnContext(
267
273
  if (input.brokerSpawn?.socketPath && input.brokerSpawn.token) {
268
274
  built.env.PI_CREW_BROKER_SOCKET = input.brokerSpawn.socketPath;
269
275
  built.env.PI_CREW_BROKER_TOKEN = input.brokerSpawn.token;
270
- // The child needs its own runId + taskId to complete the broker `hello`
271
- // (the token is validated against the runId; the taskId binds the
272
- // connection for message routing). Both are control-namespace keys so
273
- // they pass assertOnlyControlEnvKeys. agentId is the per-task id.
274
- if (input.runId) built.env.PI_CREW_BROKER_RUN_ID = input.runId;
275
- if (input.agentId) built.env.PI_CREW_BROKER_TASK_ID = input.agentId;
276
+ // runId/taskId are threaded unconditionally above (I5); nothing to add here.
276
277
  }
277
278
  // Phase 1 scratchpad: opt in the persistent Bun-free JS evaluator (execute
278
279
  // tool) for this worker. Gated by role/agent (S-6 read-only roles never;
@@ -289,7 +290,12 @@ export function prepareSpawnContext(
289
290
  // artifactsRoot after F4/S-1).
290
291
  built.env.PI_CREW_ARTIFACTS_ROOT = input.artifactsRoot;
291
292
  }
292
- // F4/S-1: the RAW (unredacted) snapshot must NEVER land in artifactsRoot —
293
+ // I5: thread the run's events path so the worker's scratchpad execute
294
+ // handler can append fire-and-forget metric events (scratchpad.cell /
295
+ // scratchpad.restored). Optional — absent in non-team contexts.
296
+ if (input.eventsPath) {
297
+ built.env.PI_CREW_EVENTS_PATH = input.eventsPath;
298
+ } // F4/S-1: the RAW (unredacted) snapshot must NEVER land in artifactsRoot —
293
299
  // point it at a temp dir; the worker reads it then writeArtifact()
294
300
  // (redact+atomic) is the ONLY writer into artifactsRoot.
295
301
  // R3-1: built.tempDir is only created by buildPiWorkerArgs when the agent
@@ -1,11 +1,8 @@
1
1
  import { spawn } from "node:child_process";
2
2
  import * as fs from "node:fs";
3
- import * as os from "node:os";
4
- import * as path from "node:path";
5
3
  import type { AgentConfig } from "../../agents/agent-config.ts";
6
4
  import { DEFAULT_CHILD_PI } from "../../config/defaults.ts";
7
5
  import { registerChildProcess, unregisterChildProcess } from "../../extension/crew-cleanup.ts";
8
- import { atomicWriteFile } from "../../state/atomic-write.ts";
9
6
  import type { WorkerExitStatus } from "../../state/types.ts";
10
7
  import { logInternalError } from "../../utils/internal-error.ts";
11
8
  import { redactSecretString } from "../../utils/redaction.ts";
@@ -17,6 +14,7 @@ import { buildFinalChildPiSpawnOptions, prepareSpawnContext } from "./child-pi-s
17
14
  import { ChildPiSteeringController } from "./child-pi-steering.ts";
18
15
  // Internal helpers for active-child bookkeeping (extracted to child-pi-kill.ts).
19
16
  import { ChildPiLineObserver } from "./child-pi-streams.ts";
17
+ import { runMockChildPi } from "./mock-fixtures.ts";
20
18
 
21
19
  // ── Re-exports from child-pi-kill.ts (H-7 decomposition step 2) ──
22
20
  // killProcessTree is internal (not previously exported) — keep that invariant.
@@ -147,6 +145,10 @@ export interface ChildPiRunInput {
147
145
  thinkingOverride?: string;
148
146
  /** Root directory for artifacts (used to validate transcriptPath). */
149
147
  artifactsRoot?: string;
148
+ /** I5: run events JSONL path — threaded to the worker so its scratchpad
149
+ * execute handler can append fire-and-forget metric events. Optional;
150
+ * absent in non-team contexts (emission silently skipped). */
151
+ eventsPath?: string;
150
152
  /** Phase 1 scratchpad: model-fallback attempt index (0-based) for per-attempt
151
153
  * snapshot relativePath `scratchpad/<taskId>.attempt-<attempt>.snapshot.json`
152
154
  * (C3). Optional — worker defaults to attempt 0 when unset (e.g. custom
@@ -253,146 +255,10 @@ export async function runChildPi(input: ChildPiRunInput): Promise<ChildPiRunResu
253
255
  stdout: "",
254
256
  stderr: `pi-crew depth guard blocked child worker: depth ${depth.depth} >= max ${depth.maxDepth}`,
255
257
  };
256
- const mock = process.env.PI_TEAMS_MOCK_CHILD_PI;
257
- if (mock) {
258
- // SECURITY (Issue #2): Mock mode security model is intentionally asymmetric.
259
- // PI_TEAMS_MOCK_CHILD_PI is in the allowlist (passed to children) but
260
- // PI_CREW_ALLOW_MOCK is NOT in the allowlist — it is only checked in the
261
- // parent process scope. This means:
262
- // (1) If an attacker sets PI_CREW_ALLOW_MOCK in the parent's environment,
263
- // it will NOT be passed to child processes (safe).
264
- // (2) Mock mode activation in the child always fails the PI_CREW_ALLOW_MOCK
265
- // check, so mock mode can only be triggered from the parent process.
266
- // This asymmetry is intentional: PI_CREW_ALLOW_MOCK must be set in the Pi root
267
- // process (the entry point that spawns children), not inherited from a parent.
268
- // Setup hooks cannot inject PI_CREW_ALLOW_MOCK into the parent's env.
269
- const allowMock = process.env.PI_CREW_ALLOW_MOCK === "1" || process.env.PI_CREW_ALLOW_MOCK === "true";
270
- if (!allowMock) {
271
- return {
272
- exitCode: 1,
273
- stdout: "",
274
- stderr: "Mock mode requires PI_CREW_ALLOW_MOCK=1",
275
- };
276
- }
277
- // SECURITY: Log mock mode activation prominently for audit trail
278
- logInternalError("child-pi.mock", new Error(`Mock mode active: ${mock}`), "NOT running real agents");
279
- if (mock === "success") {
280
- const stdout = `[MOCK] Success for ${input.agent.name}\n`;
281
- await observeStdoutChunk(input, stdout);
282
- return { exitCode: 0, stdout, stderr: "" };
283
- }
284
- if (mock === "json-success" || mock === "adaptive-plan") {
285
- const text =
286
- mock === "adaptive-plan" && effectiveTask.includes("ADAPTIVE_PLAN_JSON_START")
287
- ? `[MOCK] Adaptive plan\nADAPTIVE_PLAN_JSON_START\n${JSON.stringify({
288
- phases: [
289
- {
290
- name: "research",
291
- tasks: [
292
- {
293
- role: "explorer",
294
- task: "Explore adaptive target",
295
- },
296
- {
297
- role: "analyst",
298
- task: "Analyze adaptive target",
299
- },
300
- {
301
- role: "planner",
302
- task: "Plan adaptive target",
303
- },
304
- ],
305
- },
306
- {
307
- name: "build",
308
- tasks: [
309
- {
310
- role: "executor",
311
- task: "Implement adaptive target",
312
- },
313
- ],
314
- },
315
- {
316
- name: "check",
317
- tasks: [
318
- {
319
- role: "reviewer",
320
- task: "Review adaptive target",
321
- },
322
- {
323
- role: "test-engineer",
324
- task: "Test adaptive target",
325
- },
326
- {
327
- role: "writer",
328
- task: "Summarize adaptive target",
329
- },
330
- ],
331
- },
332
- ],
333
- })}\nADAPTIVE_PLAN_JSON_END`
334
- : `[MOCK] JSON success for ${input.agent.name}`;
335
- const stdout = `${JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text }] } })}\n${JSON.stringify({ type: "message_end", usage: { input: 10, output: 5, cost: 0.001, turns: 1 } })}\n`;
336
- await observeStdoutChunk(input, stdout);
337
- return { exitCode: 0, stdout, stderr: "" };
338
- }
339
- if (mock === "retryable-failure")
340
- return {
341
- exitCode: 1,
342
- stdout: "",
343
- stderr: "[MOCK] rate limit: mock failure",
344
- };
345
- // E2E fallback-chain fixture: invocation #1 returns a SILENT retryable
346
- // failure (exit code 0, no real assistant text, message_end carries a
347
- // retryable-pattern errorMessage). Invocation #2+ delegates to the
348
- // standard json-success shape. Counter lives in os.tmpdir() keyed by
349
- // process.pid + mock name so concurrent test processes don't collide.
350
- // The test cleans up the file in its finally block.
351
- if (mock === "retryable-failure-then-success") {
352
- const counterFile = path.join(os.tmpdir(), `pi-crew-mock-counter-${process.pid}-retryable-failure-then-success`);
353
- let count = 0;
354
- try {
355
- const raw = fs.readFileSync(counterFile, "utf-8");
356
- const parsed = Number.parseInt(raw.trim(), 10);
357
- if (Number.isFinite(parsed) && parsed >= 0) count = parsed;
358
- } catch {
359
- // file missing or unreadable — first invocation in this process
360
- }
361
- count += 1;
362
- try {
363
- atomicWriteFile(counterFile, String(count));
364
- } catch (error) {
365
- logInternalError("child-pi.mock-counter-write", error as Error, `file=${counterFile}`);
366
- }
367
- if (count === 1) {
368
- // Silent retryable failure: exit 0, no real text, message_end
369
- // carries errorMessage matching `/provider[_ ]?error/i` so that
370
- // `detectRetryableModelFailureFromOutput` surfaces it as an error
371
- // and `isRetryableModelFailure` routes the next attempt to the
372
- // next candidate model. `stopReason:"error"` (NOT "stop") so
373
- // `isFinalAssistantEvent` does NOT prematurely terminate the run.
374
- const failureEvent = {
375
- type: "message_end",
376
- message: {
377
- role: "assistant",
378
- content: [],
379
- errorMessage: "Provider error: api_error",
380
- stopReason: "error",
381
- },
382
- };
383
- const stdout = `${JSON.stringify(failureEvent)}\n`;
384
- await observeStdoutChunk(input, stdout);
385
- return { exitCode: 0, stdout, stderr: "" };
386
- }
387
- // Subsequent invocations: delegate to json-success shape so the
388
- // fallback chain's second attempt succeeds and the run completes.
389
- const text = `[MOCK] JSON success for ${input.agent.name}`;
390
- const stdout = `${JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text }] } })}\n${JSON.stringify({ type: "message_end", usage: { input: 10, output: 5, cost: 0.001, turns: 1 } })}\n`;
391
- await observeStdoutChunk(input, stdout);
392
- return { exitCode: 0, stdout, stderr: "" };
393
- }
394
- return { exitCode: 1, stdout: "", stderr: `[MOCK] failure: ${mock}` };
395
- }
258
+ // H3 phase 3 (2026-08-10): mock-mode fixtures extracted to mock-fixtures.ts.
259
+ // Returns undefined when mock mode is NOT active → fall through to spawn.
260
+ const mockResult = await runMockChildPi(input, effectiveTask, observeStdoutChunk);
261
+ if (mockResult) return mockResult;
396
262
  // H-7 step 6: spawn/env/args preparation extracted to child-pi-spawn.ts.
397
263
  // prepareSpawnContext builds the worker args, attaches the steering file env,
398
264
  // and handles the pre-spawn abort check (returns an immediate-abort result
@@ -406,7 +272,19 @@ export async function runChildPi(input: ChildPiRunInput): Promise<ChildPiRunResu
406
272
  if (!brokerSpawn && brokerIssuer && input.runId) {
407
273
  try {
408
274
  brokerSpawn = await brokerIssuer(input.runId, input.agentId);
409
- } catch {
275
+ } catch (error) {
276
+ // H8 (2026-08-10): surface the silent degradation. Previously this
277
+ // swallowed ALL issuer failures (token-rotation race, broker socket
278
+ // down, key fetch network error) with zero observability — the child
279
+ // spawned without broker credentials and the run just looked "slow".
280
+ // logInternalError writes to the internal-error channel (sampled +
281
+ // bounded); it does NOT propagate, so the child still runs without
282
+ // broker acceleration (the durable-first invariant is preserved).
283
+ logInternalError(
284
+ "child-pi.broker-issuer-failed",
285
+ error instanceof Error ? error : new Error(String(error)),
286
+ `runId=${input.runId} agentId=${input.agentId ?? "?"} — child will spawn without broker credentials`,
287
+ );
410
288
  brokerSpawn = undefined;
411
289
  }
412
290
  }
@@ -0,0 +1,171 @@
1
+ /**
2
+ * Mock-mode fixture runner for child Pi workers (H3 phase 3).
3
+ *
4
+ * Extracted from `runChildPi` (src/runtime/child-pi/child-pi.ts) on
5
+ * 2026-08-10 — the mock branch was ~140 self-contained lines embedded in the
6
+ * 840-line spawn orchestrator. Behaviour is byte-identical.
7
+ *
8
+ * Security model (unchanged): PI_TEAMS_MOCK_CHILD_PI is in the env allowlist
9
+ * (passed to children) but PI_CREW_ALLOW_MOCK is NOT — mock mode can only be
10
+ * activated from the parent process scope, never inherited by a child.
11
+ */
12
+
13
+ import * as fs from "node:fs";
14
+ import * as os from "node:os";
15
+ import * as path from "node:path";
16
+ import { atomicWriteFile } from "../../state/atomic-write.ts";
17
+ import { logInternalError } from "../../utils/internal-error.ts";
18
+ import type { ChildPiRunInput, ChildPiRunResult } from "./child-pi.ts";
19
+
20
+ /**
21
+ * Run a mock child if mock mode is active (PI_TEAMS_MOCK_CHILD_PI set).
22
+ * Returns `undefined` when mock mode is NOT active — the caller falls through
23
+ * to the real spawn path.
24
+ *
25
+ * @param input The child run input (agent, env…).
26
+ * @param effectiveTask The task text after inherit-context prepend.
27
+ * @param observe Callback that feeds a stdout chunk through the line
28
+ * observer (owned by child-pi.ts).
29
+ */
30
+ export async function runMockChildPi(
31
+ input: ChildPiRunInput,
32
+ effectiveTask: string,
33
+ observe: (input: ChildPiRunInput, text: string) => Promise<void>,
34
+ ): Promise<ChildPiRunResult | undefined> {
35
+ const mock = process.env.PI_TEAMS_MOCK_CHILD_PI;
36
+ if (!mock) return undefined;
37
+
38
+ // SECURITY (Issue #2): see module docstring — PI_CREW_ALLOW_MOCK is only
39
+ // checked in the parent process scope; it is never passed to children.
40
+ const allowMock = process.env.PI_CREW_ALLOW_MOCK === "1" || process.env.PI_CREW_ALLOW_MOCK === "true";
41
+ if (!allowMock) {
42
+ return {
43
+ exitCode: 1,
44
+ stdout: "",
45
+ stderr: "Mock mode requires PI_CREW_ALLOW_MOCK=1",
46
+ };
47
+ }
48
+ // SECURITY: Log mock mode activation prominently for audit trail
49
+ logInternalError("child-pi.mock", new Error(`Mock mode active: ${mock}`), "NOT running real agents");
50
+
51
+ if (mock === "success") {
52
+ const stdout = `[MOCK] Success for ${input.agent.name}\n`;
53
+ await observe(input, stdout);
54
+ return { exitCode: 0, stdout, stderr: "" };
55
+ }
56
+
57
+ if (mock === "json-success" || mock === "adaptive-plan") {
58
+ const text =
59
+ mock === "adaptive-plan" && effectiveTask.includes("ADAPTIVE_PLAN_JSON_START")
60
+ ? `[MOCK] Adaptive plan\nADAPTIVE_PLAN_JSON_START\n${JSON.stringify({
61
+ phases: [
62
+ {
63
+ name: "research",
64
+ tasks: [
65
+ {
66
+ role: "explorer",
67
+ task: "Explore adaptive target",
68
+ },
69
+ {
70
+ role: "analyst",
71
+ task: "Analyze adaptive target",
72
+ },
73
+ {
74
+ role: "planner",
75
+ task: "Plan adaptive target",
76
+ },
77
+ ],
78
+ },
79
+ {
80
+ name: "build",
81
+ tasks: [
82
+ {
83
+ role: "executor",
84
+ task: "Implement adaptive target",
85
+ },
86
+ ],
87
+ },
88
+ {
89
+ name: "check",
90
+ tasks: [
91
+ {
92
+ role: "reviewer",
93
+ task: "Review adaptive target",
94
+ },
95
+ {
96
+ role: "test-engineer",
97
+ task: "Test adaptive target",
98
+ },
99
+ {
100
+ role: "writer",
101
+ task: "Summarize adaptive target",
102
+ },
103
+ ],
104
+ },
105
+ ],
106
+ })}\nADAPTIVE_PLAN_JSON_END`
107
+ : `[MOCK] JSON success for ${input.agent.name}`;
108
+ const stdout = `${JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text }] } })}\n${JSON.stringify({ type: "message_end", usage: { input: 10, output: 5, cost: 0.001, turns: 1 } })}\n`;
109
+ await observe(input, stdout);
110
+ return { exitCode: 0, stdout, stderr: "" };
111
+ }
112
+
113
+ if (mock === "retryable-failure")
114
+ return {
115
+ exitCode: 1,
116
+ stdout: "",
117
+ stderr: "[MOCK] rate limit: mock failure",
118
+ };
119
+
120
+ // E2E fallback-chain fixture: invocation #1 returns a SILENT retryable
121
+ // failure (exit code 0, no real assistant text, message_end carries a
122
+ // retryable-pattern errorMessage). Invocation #2+ delegates to the
123
+ // standard json-success shape. Counter lives in os.tmpdir() keyed by
124
+ // process.pid + mock name so concurrent test processes don't collide.
125
+ // The test cleans up the file in its finally block.
126
+ if (mock === "retryable-failure-then-success") {
127
+ const counterFile = path.join(os.tmpdir(), `pi-crew-mock-counter-${process.pid}-retryable-failure-then-success`);
128
+ let count = 0;
129
+ try {
130
+ const raw = fs.readFileSync(counterFile, "utf-8");
131
+ const parsed = Number.parseInt(raw.trim(), 10);
132
+ if (Number.isFinite(parsed) && parsed >= 0) count = parsed;
133
+ } catch {
134
+ // file missing or unreadable — first invocation in this process
135
+ }
136
+ count += 1;
137
+ try {
138
+ atomicWriteFile(counterFile, String(count));
139
+ } catch (error) {
140
+ logInternalError("child-pi.mock-counter-write", error as Error, `file=${counterFile}`);
141
+ }
142
+ if (count === 1) {
143
+ // Silent retryable failure: exit 0, no real text, message_end
144
+ // carries errorMessage matching `/provider[_ ]?error/i` so that
145
+ // `detectRetryableModelFailureFromOutput` surfaces it as an error
146
+ // and `isRetryableModelFailure` routes the next attempt to the
147
+ // next candidate model. `stopReason:"error"` (NOT "stop") so
148
+ // `isFinalAssistantEvent` does NOT prematurely terminate the run.
149
+ const failureEvent = {
150
+ type: "message_end",
151
+ message: {
152
+ role: "assistant",
153
+ content: [],
154
+ errorMessage: "Provider error: api_error",
155
+ stopReason: "error",
156
+ },
157
+ };
158
+ const stdout = `${JSON.stringify(failureEvent)}\n`;
159
+ await observe(input, stdout);
160
+ return { exitCode: 0, stdout, stderr: "" };
161
+ }
162
+ // Subsequent invocations: delegate to json-success shape so the
163
+ // fallback chain's second attempt succeeds and the run completes.
164
+ const text = `[MOCK] JSON success for ${input.agent.name}`;
165
+ const stdout = `${JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text }] } })}\n${JSON.stringify({ type: "message_end", usage: { input: 10, output: 5, cost: 0.001, turns: 1 } })}\n`;
166
+ await observe(input, stdout);
167
+ return { exitCode: 0, stdout, stderr: "" };
168
+ }
169
+
170
+ return { exitCode: 1, stdout: "", stderr: `[MOCK] failure: ${mock}` };
171
+ }
@@ -299,9 +299,26 @@ export function saveCrewAgents(manifest: TeamRunManifest, records: CrewAgentReco
299
299
  withAgentsLock(manifest, () => {
300
300
  fs.mkdirSync(manifest.stateRoot, { recursive: true });
301
301
  const filePath = agentsPath(manifest);
302
+ // Index file (agents.json) is the authoritative record — always full
303
+ // durability.
302
304
  atomicWriteJson(filePath, redactSecrets(records));
303
305
  asyncAgentReaderCache.delete(filePath);
304
- for (const record of records) writeCrewAgentStatus(manifest, record);
306
+ for (const record of records) {
307
+ // H2 (2026-08-10): per-task status.json is a DENORMALIZED read
308
+ // optimization for the dashboard/notifier, not an authoritative
309
+ // record — agents.json + events.jsonl cover crash recovery.
310
+ // Previously EVERY record (including running/queued progress
311
+ // snapshots) got a full fsync: N+1 fsyncs per saveCrewAgents call
312
+ // (50-task team ≈ 750ms blocking). Only TERMINAL records keep
313
+ // full durability (F4: notifier/dashboard must see the final
314
+ // state immediately); non-terminal per-task status is best-effort
315
+ // coalesced like the upsertCrewAgent non-terminal path.
316
+ if (TERMINAL_AGENT_STATUSES.has(record.status ?? "")) {
317
+ writeCrewAgentStatus(manifest, record);
318
+ } else {
319
+ writeCrewAgentStatusCoalesced(manifest, record);
320
+ }
321
+ }
305
322
  });
306
323
  }
307
324
 
@@ -9,13 +9,41 @@
9
9
  * (headings, code blocks, URLs) after compression.
10
10
  */
11
11
 
12
- /** Role-specific output format patterns — constructed fresh per call to avoid /g lastIndex leak */
12
+ /**
13
+ * Why relax: real worker LLMs emit markdown handoffs (`## Handoff`,
14
+ * `### Summary`, `## Follow-ups`, `- bullet`, `**bold**`) instead of the
15
+ * caveman formats (`file:line — text`, `PASS:`/`FAIL:`, emoji findings)
16
+ * these patterns originally required. Each role pattern below ORs its
17
+ * strict contract with a markdown-structured alternation so structured
18
+ * handoffs validate while empty/garbage output still fails.
19
+ *
20
+ * What changed: added the MARKDOWN_STRUCTURED alternation to every role.
21
+ * What is preserved: the strict patterns still match verbatim, and the
22
+ * structural-preservation checks (code blocks, URLs, headings) in
23
+ * validateWorkerOutput are UNCHANGED.
24
+ */
25
+
26
+ /** Accepts atx headings (`## X`), bold (`**x**`), bullets (`- x` / `* x`), and numbered lists (`1. x`) */
27
+ const MARKDOWN_STRUCTURED = /^(?:#{1,6}\s|\*\*|[-*]\s|\d+\.\s)/m;
28
+
29
+ /** Strict per-role contract patterns (kept verbatim; `.source` is embedded in the alternations below) */
30
+ const STRICT_ROLE_PATTERNS: Record<string, RegExp> = {
31
+ explorer: /^(\S+:\d+|Defs:|Refs:|Callers:|Tests:|Sites:|No match\.|totals:)/m,
32
+ executor: /^(\S+:\d+(-\d+)? — .{1,80}\.|verified:|too-big\.|needs-confirm\.|ambiguous\.|regressed\.)/m,
33
+ reviewer: /^([^:\s]+:\d+:\s+\p{Emoji_Presentation}|No issues\.|totals:)/mu,
34
+ "security-reviewer": /^([^:\s]+:\d+:\s+\p{Emoji_Presentation}|No issues\.|totals:)/mu,
35
+ verifier: /^(PASS:|FAIL:)/m,
36
+ };
37
+
38
+ /** Role-specific output format patterns — constructed fresh per call to avoid /g lastIndex leak.
39
+ * Each factory: strict contract OR markdown-structured alternation. */
13
40
  const ROLE_PATTERN_DEFS: Record<string, () => RegExp> = {
14
- explorer: () => /^(\S+:\d+|Defs:|Refs:|Callers:|Tests:|Sites:|No match\.|totals:)/m,
15
- executor: () => /^(\S+:\d+(-\d+)? — .{1,80}\.|verified:|too-big\.|needs-confirm\.|ambiguous\.|regressed\.)/m,
16
- reviewer: () => /^([^:\s]+:\d+:\s+\p{Emoji_Presentation}|No issues\.|totals:)/mu,
17
- "security-reviewer": () => /^([^:\s]+:\d+:\s+\p{Emoji_Presentation}|No issues\.|totals:)/mu,
18
- verifier: () => /^(PASS:|FAIL:)/m,
41
+ explorer: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.explorer.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"),
42
+ executor: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.executor.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"),
43
+ reviewer: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.reviewer.source})|(?:${MARKDOWN_STRUCTURED.source})`, "mu"),
44
+ "security-reviewer": () =>
45
+ new RegExp(`(?:${STRICT_ROLE_PATTERNS["security-reviewer"].source})|(?:${MARKDOWN_STRUCTURED.source})`, "mu"),
46
+ verifier: () => new RegExp(`(?:${STRICT_ROLE_PATTERNS.verifier.source})|(?:${MARKDOWN_STRUCTURED.source})`, "m"),
19
47
  };
20
48
 
21
49
  /** Fresh RegExp factories for structural preservation checks (avoids /g lastIndex leak) */