pi-crew 0.9.42 → 0.9.46

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/CHANGELOG.md +223 -0
  2. package/README.md +35 -0
  3. package/dist/build-meta.json +4778 -11087
  4. package/dist/index.mjs +54648 -65435
  5. package/dist/index.mjs.map +4 -4
  6. package/package.json +1 -1
  7. package/scripts/build-bundle.mjs +2 -1
  8. package/src/agents/agent-config.ts +1 -1
  9. package/src/agents/discover-agents.ts +1 -1
  10. package/src/config/config.ts +1 -1
  11. package/src/config/types.ts +0 -2
  12. package/src/extension/crew-vibes/config.ts +1 -1
  13. package/src/extension/crew-vibes/font-detect.ts +16 -3
  14. package/src/extension/cross-extension-rpc.ts +8 -4
  15. package/src/extension/registration/commands.ts +5 -1
  16. package/src/extension/registration/foreground-run-controller.ts +28 -0
  17. package/src/extension/registration/lifecycle-handlers.ts +34 -3
  18. package/src/extension/registration/subagent-manager-setup.ts +178 -59
  19. package/src/extension/run-import.ts +21 -1
  20. package/src/extension/team-tool/api.ts +4 -2
  21. package/src/extension/team-tool.ts +5 -0
  22. package/src/prompt/prompt-runtime.ts +15 -0
  23. package/src/runtime/async-runner.ts +9 -1
  24. package/src/runtime/child-pi-constants.ts +42 -0
  25. package/src/runtime/child-pi-kill.ts +180 -0
  26. package/src/runtime/child-pi-spawn.ts +234 -0
  27. package/src/runtime/child-pi-steering.ts +128 -0
  28. package/src/runtime/child-pi-streams.ts +296 -0
  29. package/src/runtime/child-pi-transcript.ts +169 -0
  30. package/src/runtime/child-pi.ts +75 -837
  31. package/src/runtime/compact-stages/tail-capture-stage.ts +1 -1
  32. package/src/runtime/dwf-state-store.ts +5 -0
  33. package/src/runtime/dynamic-workflow-context.ts +78 -18
  34. package/src/runtime/dynamic-workflow-runner.ts +18 -2
  35. package/src/runtime/goal-evaluator.ts +59 -27
  36. package/src/runtime/goal-loop-runner.ts +40 -11
  37. package/src/runtime/live-agent-manager.ts +18 -0
  38. package/src/runtime/manifest-cache.ts +30 -0
  39. package/src/runtime/pi-json-output.ts +2 -1
  40. package/src/runtime/pi-spawn.ts +7 -1
  41. package/src/runtime/resilient-edit.ts +16 -15
  42. package/src/runtime/role-permission.ts +27 -2
  43. package/src/runtime/run-coalesced-task-group.ts +90 -25
  44. package/src/runtime/task-packet.ts +1 -1
  45. package/src/runtime/task-runner/prompt-builder.ts +2 -2
  46. package/src/runtime/team-runner.ts +6 -1
  47. package/src/state/event-log.ts +89 -34
  48. package/src/state/locks.ts +185 -49
  49. package/src/state/mailbox.ts +165 -4
  50. package/src/state/run-metrics.ts +40 -12
  51. package/src/state/worker-atomic-writer.ts +1 -0
  52. package/src/ui/live-run-sidebar.ts +1 -9
  53. package/src/ui/run-dashboard.ts +1 -9
  54. package/src/utils/incremental-reader.ts +105 -0
  55. package/src/utils/paths.ts +1 -1
  56. package/src/utils/visual.ts +27 -91
  57. package/src/worktree/worktree-manager.ts +47 -19
  58. package/src/runtime/auto-resume.ts +0 -100
  59. package/src/runtime/notebook-helpers.ts +0 -88
  60. package/src/runtime/orphan-sentinel.ts +0 -7
@@ -59,7 +59,7 @@ export class TailCaptureStage implements ICompactStage {
59
59
  // never contains a partial multi-byte sequence.
60
60
  if (Buffer.byteLength(text, "utf-8") <= this.maxBytes) return text;
61
61
  let tail = text.slice(Math.max(0, text.length - this.maxBytes));
62
- while (Buffer.byteLength(tail, "utf-8") > this.maxBytes) tail = tail.slice(1024);
62
+ while (Buffer.byteLength(tail, "utf-8") > this.maxBytes) tail = tail.slice(0, -1);
63
63
  return this.marker ? `${this.marker}\n${tail}` : tail;
64
64
  }
65
65
  // Char cap mode.
@@ -28,6 +28,11 @@ export interface DwfCheckpointState {
28
28
  logs: string[]; // capped copy (≤1000); the events log (dwf.log) is the durable source of truth
29
29
  spent: number; // budget accumulator (round-14 P1-2)
30
30
  agentCount: number;
31
+ /** PERS-1: per-agent-call idempotency cache. Maps a deterministic call ID to the
32
+ * cached result so a DWF resume skips already-completed agent calls (avoids
33
+ * duplicating artifacts/mailbox/tokens). Uses Record (not Map) for JSON
34
+ * serialization safety. */
35
+ completedAgentCalls?: Record<string, { text: string; usage?: { input: number; output: number } }>;
31
36
  updatedAt: string;
32
37
  }
33
38
 
@@ -35,6 +35,7 @@ import { logInternalError } from "../utils/internal-error.ts";
35
35
  import { cleanupAgentWorktreeAsync, prepareAgentWorktreeAsync } from "../worktree/worktree-manager.ts";
36
36
  import { runChildPi } from "./child-pi.ts";
37
37
  import type { DwfCheckpointState } from "./dwf-state-store.ts";
38
+ import { withWorkerSlot } from "./global-worker-cap.ts";
38
39
  import { mapConcurrent } from "./parallel-utils.ts";
39
40
  import { parsePiJsonOutput } from "./pi-json-output.ts";
40
41
  import { renderPlanTemplate } from "./plan-templates.ts";
@@ -269,6 +270,11 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
269
270
  logs: opts.resumedState ? [...opts.resumedState.logs].slice(0, 1000) : [],
270
271
  args: opts.args ?? {},
271
272
  };
273
+ // PERS-1: per-agent-call idempotency cache. Hydrated from resumedState so a DWF
274
+ // resume skips already-completed agent calls (avoids duplicating artifacts/mailbox/tokens).
275
+ // Uses Record (not Map) for JSON serialization safety.
276
+ const completedAgentCalls: Record<string, { text: string; usage?: { input: number; output: number } }> =
277
+ opts.resumedState?.completedAgentCalls ?? {};
272
278
  // round-14 P1-2: frozen budget surface. The closures read wfState.spent so the
273
279
  // object stays live after Object.freeze(ctx). total is a snapshot primitive.
274
280
  const budget = Object.freeze({
@@ -290,10 +296,35 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
290
296
  // regardless of which return/throw path is taken.
291
297
  let worktreePath: string | undefined;
292
298
  let worktreeBranch: string | undefined;
299
+ // BDG-2: declared before the try so the catch block can un-reserve on failure.
300
+ const ESTIMATE = 4096;
301
+ let reserved = false;
293
302
  try {
294
- // round-14 P1-2: budget check BEFORE spawning. When the per-workflow token
295
- // budget is exhausted, reject the call without consuming a child worker.
296
- if (budget.total !== null && budget.remaining() <= 0) {
303
+ // PERS-1: per-agent-call idempotency. Compute a deterministic call ID
304
+ // from the call arguments and check if this call was already completed
305
+ // (e.g. during a previous run before a crash). If so, return the cached
306
+ // result without re-spawning the agent.
307
+ const callId = JSON.stringify({
308
+ role: call.role,
309
+ agent: call.agent,
310
+ prompt: call.prompt,
311
+ model: call.model,
312
+ schema: call.schema ? "yes" : "no",
313
+ });
314
+ const cached = completedAgentCalls[callId];
315
+ if (cached) {
316
+ return {
317
+ ok: true,
318
+ text: cached.text,
319
+ usage: cached.usage,
320
+ durationMs: 0,
321
+ };
322
+ }
323
+
324
+ // BDG-2: reserve-then-adjust budget. Before spawning, estimate the cost and
325
+ // reserve it by adding to wfState.spent. This prevents N concurrent calls from
326
+ // all seeing the same remaining budget and overspending.
327
+ if (budget.total !== null && budget.remaining() < ESTIMATE) {
297
328
  return {
298
329
  ok: false,
299
330
  text: "",
@@ -301,6 +332,10 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
301
332
  durationMs: 0,
302
333
  };
303
334
  }
335
+ // Reserve the estimate before spawning.
336
+ wfState.spent += ESTIMATE;
337
+ reserved = true;
338
+
304
339
  const agentConfig = resolveAgentForRole(call.role, {
305
340
  explicitAgent: call.agent,
306
341
  team: opts.team,
@@ -353,20 +388,25 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
353
388
  }
354
389
  }
355
390
 
356
- const childResult = await runChildPi({
357
- cwd: agentCwd,
358
- task,
359
- agent: effectiveAgent,
360
- model: call.model ?? opts.modelOverride ?? agentConfig.model,
361
- skillPaths: undefined, // skills resolved via agent config + team-role plumbing
362
- maxTurns: call.maxTurns,
363
- graceTurns: call.graceTurns,
364
- signal: opts.signal,
365
- artifactsRoot: manifest.artifactsRoot,
366
- runId: manifest.runId,
367
- role: call.role ?? call.agent,
368
- });
391
+ const childResult = await withWorkerSlot(() =>
392
+ runChildPi({
393
+ cwd: agentCwd,
394
+ task,
395
+ agent: effectiveAgent,
396
+ model: call.model ?? opts.modelOverride ?? agentConfig.model,
397
+ skillPaths: undefined, // skills resolved via agent config + team-role plumbing
398
+ maxTurns: call.maxTurns,
399
+ graceTurns: call.graceTurns,
400
+ signal: opts.signal,
401
+ artifactsRoot: manifest.artifactsRoot,
402
+ runId: manifest.runId,
403
+ role: call.role ?? call.agent,
404
+ }),
405
+ );
369
406
  if (childResult.exitCode !== 0 || childResult.error) {
407
+ // BDG-2: un-reserve the estimate on spawn failure.
408
+ wfState.spent -= ESTIMATE;
409
+ reserved = false;
370
410
  return {
371
411
  ok: false,
372
412
  text: "",
@@ -376,8 +416,10 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
376
416
  }
377
417
  const parsed = parsePiJsonOutput(childResult.stdout);
378
418
  // round-14 P1-2: accumulate this run's token usage into the workflow budget.
379
- // Covers both the success and schema-mismatch paths (both report parsed.usage).
380
- wfState.spent += (parsed.usage?.input ?? 0) + (parsed.usage?.output ?? 0);
419
+ // BDG-2: adjust the reserve — subtract the estimate, add the actual usage.
420
+ // This correctly reduces spent when actualUsage < ESTIMATE.
421
+ wfState.spent += (parsed.usage?.input ?? 0) + (parsed.usage?.output ?? 0) - ESTIMATE;
422
+ reserved = false;
381
423
  let text = parsed.finalText ?? "";
382
424
  // Round-11 test fix: parsePiJsonOutput only extracts text from pi event stream
383
425
  // ({type:"message_end", message:{role:"assistant", content:[...]}}). When the
@@ -400,6 +442,7 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
400
442
  producer: "dynamic-workflow",
401
443
  });
402
444
  if (call.schema !== undefined && !extracted.structured) {
445
+ // BDG-2: un-reserve was already done above (reserved = false after adjust).
403
446
  return {
404
447
  ok: false,
405
448
  text,
@@ -409,6 +452,11 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
409
452
  durationMs: Date.now() - started,
410
453
  };
411
454
  }
455
+ // PERS-1: cache the successful result for idempotency on resume.
456
+ completedAgentCalls[callId] = {
457
+ text,
458
+ usage: { input: parsed.usage?.input ?? 0, output: parsed.usage?.output ?? 0 },
459
+ };
412
460
  return {
413
461
  ok: true,
414
462
  text,
@@ -418,6 +466,8 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
418
466
  durationMs: Date.now() - started,
419
467
  };
420
468
  } catch (error) {
469
+ // BDG-2: un-reserve the estimate on failure (only if still reserved).
470
+ if (reserved) wfState.spent -= ESTIMATE;
421
471
  logInternalError("dynamic-workflow-context.agent", error, `runId=${manifest.runId}`);
422
472
  return {
423
473
  ok: false,
@@ -450,6 +500,7 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
450
500
  logs: wfState.logs.slice(0, 1000),
451
501
  spent: wfState.spent,
452
502
  agentCount,
503
+ completedAgentCalls,
453
504
  updatedAt: new Date().toISOString(),
454
505
  });
455
506
  } catch (checkpointError) {
@@ -709,6 +760,12 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
709
760
  get: () => agentCount,
710
761
  enumerable: false,
711
762
  });
763
+ // PERS-1: completedAgentCalls is read-only from the runner; the agent() method
764
+ // is the only writer. Exposed so getWorkflowCheckpoint() can include it.
765
+ Object.defineProperty(ctx, "__completedAgentCalls", {
766
+ get: () => completedAgentCalls,
767
+ enumerable: false,
768
+ });
712
769
  return ctx;
713
770
  }
714
771
 
@@ -757,6 +814,9 @@ export function getWorkflowCheckpoint(ctx: WorkflowCtx): DwfCheckpointState {
757
814
  logs: logs ?? [],
758
815
  spent: ctx.budget.spent(),
759
816
  agentCount: (ctx as unknown as { __agentCount?: number }).__agentCount ?? 0,
817
+ completedAgentCalls:
818
+ (ctx as unknown as { __completedAgentCalls?: Record<string, { text: string; usage?: { input: number; output: number } }> })
819
+ .__completedAgentCalls ?? {},
760
820
  updatedAt: new Date().toISOString(),
761
821
  };
762
822
  }
@@ -19,6 +19,7 @@
19
19
 
20
20
  import { readFileSync } from "node:fs";
21
21
  import { join } from "node:path";
22
+ import { transformSync } from "esbuild";
22
23
  import { writeArtifact } from "../state/artifact-store.ts";
23
24
  import { appendEvent } from "../state/event-log.ts";
24
25
  import type { TeamRunManifest, TeamTaskState } from "../state/types.ts";
@@ -113,7 +114,11 @@ async function loadWorkflowModule(scriptPath: string): Promise<DynamicWorkflowSc
113
114
  // raw .dwf.ts file. This is the same source jiti will execute.
114
115
  const scriptSource = readFileSync(scriptPath, "utf-8");
115
116
  if (isDeterminismCheckEnabled()) {
116
- assertDeterministicScript(scriptSource);
117
+ // DISC-1: acorn can only parse JavaScript, not TypeScript. Real .dwf.ts files
118
+ // contain type annotations/imports that cause a silent parse error, so the
119
+ // determinism check never runs. Transpile to JS via esbuild first.
120
+ const js = transformSync(scriptSource, { loader: "ts", format: "esm" }).code;
121
+ assertDeterministicScript(js);
117
122
  }
118
123
  // jiti is the same loader async-runner.ts uses (resolveTypeScriptLoader). We require it
119
124
  // lazily so this module stays importable in environments without jiti (type-only consumers).
@@ -166,9 +171,16 @@ export async function runDynamicWorkflow(input: RunDynamicWorkflowInput): Promis
166
171
  });
167
172
  }
168
173
 
174
+ // ERR-1: Create a timeout AbortController so we can abort spawned children on script timeout.
175
+ // Combine the input signal with the timeout controller's signal so EITHER source
176
+ // (external abort OR our timeout) propagates to runChildPi via the ctx.
177
+ // AbortSignal.any is available since Node 20.3 — the project requires Node 20+.
178
+ const timeoutController = new AbortController();
179
+ const combinedSignal = AbortSignal.any([signal, timeoutController.signal]);
180
+
169
181
  const ctx = makeWorkflowCtx(manifest, {
170
182
  concurrency: input.concurrency ?? workflow.maxConcurrency ?? 4,
171
- signal,
183
+ signal: combinedSignal,
172
184
  team: input.team,
173
185
  modelOverride: input.modelOverride,
174
186
  tokenBudget: input.tokenBudget ?? workflow.maxTokenBudget,
@@ -200,6 +212,10 @@ export async function runDynamicWorkflow(input: RunDynamicWorkflowInput): Promis
200
212
  let timeoutHandle: NodeJS.Timeout | undefined;
201
213
  const timeoutPromise = new Promise<never>((_, reject) => {
202
214
  timeoutHandle = setTimeout(() => {
215
+ // ERR-1: abort the timeout controller so the abort propagates to
216
+ // runChildPi (via the combined signal passed to makeWorkflowCtx),
217
+ // which kills spawned children instead of letting them run.
218
+ timeoutController.abort();
203
219
  reject(
204
220
  new Error(
205
221
  `Dynamic workflow script timed out after ${SCRIPT_TIMEOUT_MS}ms. The script may have spawned a child process that did not exit. Check for spawn/exec calls without proper stdio handling.`,
@@ -188,7 +188,9 @@ function tryParseDirectVerdict(stdout: string): { achieved: boolean; reason: str
188
188
  * Returns a GoalVerdict. On any failure (non-zero exit, non-JSON, invalid shape),
189
189
  * returns a `BLOCKED:`-prefixed verdict so the loop stops (§0c C6 fallback).
190
190
  */
191
- export async function evaluateGoal(input: EvaluateGoalInput): Promise<GoalVerdict> {
191
+ export async function evaluateGoal(
192
+ input: EvaluateGoalInput,
193
+ ): Promise<{ verdict: GoalVerdict; judgeUsage?: { totalTokens: number; inputTokens?: number; outputTokens?: number } }> {
192
194
  const agent = synthesizeJudgeAgentConfig();
193
195
  const task = buildJudgeTask(input);
194
196
  const evaluatedAt = new Date().toISOString();
@@ -212,12 +214,15 @@ export async function evaluateGoal(input: EvaluateGoalInput): Promise<GoalVerdic
212
214
  });
213
215
 
214
216
  if (result.exitCode !== 0 || result.error) {
215
- return blockedVerdict(
216
- input.turn,
217
- input.model,
218
- evaluatedAt,
219
- `judge spawn failed (exit=${result.exitCode}): ${result.error ?? result.stderr.slice(0, 200)}`,
220
- );
217
+ return {
218
+ verdict: blockedVerdict(
219
+ input.turn,
220
+ input.model,
221
+ evaluatedAt,
222
+ `judge spawn failed (exit=${result.exitCode}): ${result.error ?? result.stderr.slice(0, 200)}`,
223
+ ),
224
+ judgeUsage: undefined,
225
+ };
221
226
  }
222
227
 
223
228
  const parsed = parsePiJsonOutput(result.stdout);
@@ -231,16 +236,22 @@ export async function evaluateGoal(input: EvaluateGoalInput): Promise<GoalVerdic
231
236
  const direct = !finalText.trim() ? tryParseDirectVerdict(result.stdout) : undefined;
232
237
  if (direct) {
233
238
  return {
234
- turn: input.turn,
235
- achieved: direct.achieved,
236
- reason: direct.reason,
237
- evidenceRefs: direct.evidenceRefs,
238
- evaluatorModel: input.model,
239
- evaluatedAt,
239
+ verdict: {
240
+ turn: input.turn,
241
+ achieved: direct.achieved,
242
+ reason: direct.reason,
243
+ evidenceRefs: direct.evidenceRefs,
244
+ evaluatorModel: input.model,
245
+ evaluatedAt,
246
+ },
247
+ judgeUsage: undefined,
240
248
  };
241
249
  }
242
250
  if (!finalText.trim()) {
243
- return blockedVerdict(input.turn, input.model, evaluatedAt, "judge produced no output");
251
+ return {
252
+ verdict: blockedVerdict(input.turn, input.model, evaluatedAt, "judge produced no output"),
253
+ judgeUsage: undefined,
254
+ };
244
255
  }
245
256
 
246
257
  const extracted = extractStructuredResult(finalText);
@@ -252,27 +263,48 @@ export async function evaluateGoal(input: EvaluateGoalInput): Promise<GoalVerdic
252
263
  })
253
264
  : undefined;
254
265
  if (!data || typeof data.achieved !== "boolean" || typeof data.reason !== "string") {
255
- return blockedVerdict(input.turn, input.model, evaluatedAt, `judge output not valid verdict JSON: ${truncate(finalText, 200)}`);
266
+ return {
267
+ verdict: blockedVerdict(
268
+ input.turn,
269
+ input.model,
270
+ evaluatedAt,
271
+ `judge output not valid verdict JSON: ${truncate(finalText, 200)}`,
272
+ ),
273
+ judgeUsage: undefined,
274
+ };
256
275
  }
257
276
  const evidenceRefs = Array.isArray(data.evidenceRefs)
258
277
  ? data.evidenceRefs.filter((r): r is string => typeof r === "string")
259
278
  : undefined;
279
+ const judgeUsage = parsed.usage
280
+ ? {
281
+ totalTokens: (parsed.usage.input ?? 0) + (parsed.usage.output ?? 0),
282
+ inputTokens: parsed.usage.input,
283
+ outputTokens: parsed.usage.output,
284
+ }
285
+ : undefined;
260
286
  return {
261
- turn: input.turn,
262
- achieved: data.achieved,
263
- reason: data.reason,
264
- evidenceRefs,
265
- evaluatorModel: input.model,
266
- evaluatedAt,
287
+ verdict: {
288
+ turn: input.turn,
289
+ achieved: data.achieved,
290
+ reason: data.reason,
291
+ evidenceRefs,
292
+ evaluatorModel: input.model,
293
+ evaluatedAt,
294
+ },
295
+ judgeUsage,
267
296
  };
268
297
  } catch (error) {
269
298
  logInternalError("goal-evaluator.evaluateGoal", error, `turn=${input.turn}`);
270
- return blockedVerdict(
271
- input.turn,
272
- input.model,
273
- evaluatedAt,
274
- `judge threw: ${error instanceof Error ? error.message : String(error)}`,
275
- );
299
+ return {
300
+ verdict: blockedVerdict(
301
+ input.turn,
302
+ input.model,
303
+ evaluatedAt,
304
+ `judge threw: ${error instanceof Error ? error.message : String(error)}`,
305
+ ),
306
+ judgeUsage: undefined,
307
+ };
276
308
  }
277
309
  }
278
310
 
@@ -66,21 +66,29 @@ export const stubGoalEvaluator = async (
66
66
  _m?: import("../state/types.ts").TeamRunManifest,
67
67
  _t?: import("../state/types.ts").TeamTaskState[],
68
68
  _s?: AbortSignal,
69
- ): Promise<GoalVerdict> => ({
70
- turn: goal.turnsUsed,
71
- achieved: false,
72
- reason: `not-achieved: stub evaluator (P0). Turn ${goal.turnsUsed}/${goal.maxTurns} completed; P1 will judge against objective + verification.`,
73
- evaluatorModel: "stub",
74
- evaluatedAt: new Date().toISOString(),
69
+ ): Promise<GoalEvaluatorResult> => ({
70
+ verdict: {
71
+ turn: goal.turnsUsed,
72
+ achieved: false,
73
+ reason: `not-achieved: stub evaluator (P0). Turn ${goal.turnsUsed}/${goal.maxTurns} completed; P1 will judge against objective + verification.`,
74
+ evaluatorModel: "stub",
75
+ evaluatedAt: new Date().toISOString(),
76
+ },
77
+ judgeUsage: undefined,
75
78
  });
76
79
 
80
+ export type GoalEvaluatorResult = {
81
+ verdict: GoalVerdict;
82
+ judgeUsage?: { totalTokens: number; inputTokens?: number; outputTokens?: number };
83
+ };
84
+
77
85
  export type GoalEvaluatorFn = (
78
86
  goal: GoalLoopState,
79
87
  turnRunId: string,
80
88
  turnManifest: import("../state/types.ts").TeamRunManifest,
81
89
  turnTasks: import("../state/types.ts").TeamTaskState[],
82
90
  signal: AbortSignal,
83
- ) => Promise<GoalVerdict>;
91
+ ) => Promise<GoalEvaluatorResult>;
84
92
 
85
93
  /**
86
94
  * Production evaluator (P1): bundles turn evidence + calls the LLM judge.
@@ -94,7 +102,7 @@ export const realGoalEvaluator = async (
94
102
  turnManifest: import("../state/types.ts").TeamRunManifest,
95
103
  turnTasks: import("../state/types.ts").TeamTaskState[],
96
104
  signal: AbortSignal,
97
- ): Promise<GoalVerdict> => {
105
+ ): Promise<GoalEvaluatorResult> => {
98
106
  const transcriptPath = deriveTranscriptPath(turnManifest.artifactsRoot, turnTasks);
99
107
  // Fix round-7 F1: execute verification commands (if configured) so the judge has real evidence.
100
108
  // Previously bundleEvidence received `undefined` — the judge was told commands "MUST pass"
@@ -676,17 +684,38 @@ export async function runGoalLoop(input: RunGoalLoopInput): Promise<RunGoalLoopR
676
684
  logInternalError("goal-loop.saveTurnTasks", error, `turnRunId=${created.manifest.runId}`);
677
685
  }
678
686
 
687
+ // GL-1: if the turn ended in blocked/failed status, stop the loop without spawning the judge.
688
+ // Previously the loop continued to evaluateGoal on blocked/failed turns, burning maxTurns.
689
+ const turnStatus = turnResult.manifest.status;
690
+ if (turnStatus === "blocked" || turnStatus === "failed") {
691
+ goal = safeSetStatus(store, goal.goalId, "blocked", goal, eventsPath);
692
+ appendEvent(eventsPath, {
693
+ type: "goal.turn_terminal_status",
694
+ runId: manifest.runId,
695
+ data: {
696
+ goalId: goal.goalId,
697
+ turn: turnIndex,
698
+ turnRunId: created.manifest.runId,
699
+ turnStatus,
700
+ },
701
+ });
702
+ break;
703
+ }
704
+
679
705
  // ── BUDGET accumulation (§0c C2: collectRunMetrics) ──────────────────────
680
- const updatedBudget = accumulateBudget(goal, created.manifest.runId);
706
+ const turnBudget = accumulateBudget(goal, created.manifest.runId);
681
707
 
682
708
  // ── EVALUATE (P1: real LLM judge; pass turn manifest + tasks for transcript lookup) ──
683
- const verdict = await evaluator(
684
- { ...goal, budgetUsed: updatedBudget },
709
+ const { verdict, judgeUsage } = await evaluator(
710
+ { ...goal, budgetUsed: turnBudget },
685
711
  created.manifest.runId,
686
712
  turnResult.manifest,
687
713
  turnResult.tasks,
688
714
  signal,
689
715
  );
716
+ // BDG-1: accumulate judge token usage into budget alongside the turn's worker tokens.
717
+ const judgeTokens = judgeUsage?.totalTokens ?? 0;
718
+ const updatedBudget = turnBudget + judgeTokens;
690
719
  const historyEntry = {
691
720
  runId: created.manifest.runId,
692
721
  outcome: verdict.achieved ? "achieved" : "not-achieved",
@@ -268,6 +268,24 @@ export function evictStaleLiveAgentHandles(now = Date.now()): number {
268
268
  // before evicting. Only evict if the process is dead, to avoid evicting
269
269
  // slow-but-alive agents.
270
270
  const sessionPid = (handle.session as Record<string, unknown>)?.pid as number | undefined;
271
+ if (sessionPid === undefined) {
272
+ // In-process live-session agents have no pid — the Pi SDK returns
273
+ // in-process objects, not child processes. checkProcessLiveness(undefined)
274
+ // returns {alive:false} which would wrongly evict a running agent.
275
+ // Check session-level activity signals first. If the session is
276
+ // actively streaming or has pending messages, it is alive — skip.
277
+ // Otherwise, rely on the idle timeout alone: do NOT dispose just
278
+ // because pid is undefined.
279
+ const session = handle.session as Record<string, unknown>;
280
+ const isStreaming = session?.isStreaming === true;
281
+ const pendingCount = (session?.pendingMessageCount as number | undefined) ?? 0;
282
+ if (isStreaming || pendingCount > 0) {
283
+ // Session is active — skip eviction.
284
+ continue;
285
+ }
286
+ // No activity signals and no pid — do NOT evict based on PID liveness.
287
+ continue;
288
+ }
271
289
  const liveness = checkProcessLiveness(sessionPid);
272
290
  if (!liveness.alive) {
273
291
  liveAgents.delete(agentId);
@@ -164,6 +164,7 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
164
164
  manifestIndex.clear();
165
165
  }
166
166
  listCache.clear();
167
+ invalidateListActive();
167
168
  }
168
169
 
169
170
  function scheduleListRefresh(): void {
@@ -174,12 +175,17 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
174
175
  const timer = listTimer;
175
176
  listTimer = undefined;
176
177
  listCache.clear();
178
+ invalidateListActive();
177
179
  timer?.unref();
178
180
  }, ttlMs);
179
181
  // Unref immediately so the timer never blocks process exit (defense in
180
182
  // depth: the in-callback unref above may not run if shutdown happens
181
183
  // before the timer fires).
182
184
  listTimer.unref();
185
+ // FIND-03: invalidate the listActive() cache eagerly on every watcher
186
+ // tick. The TTL is the fallback for missed events; the watcher-driven
187
+ // path gives the tightest possible invalidation.
188
+ invalidateListActive();
183
189
  }
184
190
 
185
191
  function loadManifest(runId: string, rootsToCheck: string[]): CachedManifest | undefined {
@@ -266,6 +272,18 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
266
272
  return undefined;
267
273
  }
268
274
 
275
+ // FIND-03: short-TTL cache for listActive(). Mirrors the listCache pattern
276
+ // used by list(): we cache the un-capped running set and apply the caller's
277
+ // `limit` post-hoc on every return. Storing the full set (not a
278
+ // limit-sliced array) is what preserves the RT-F3 contract — callers with
279
+ // different `limit` values all see the same underlying "every running run"
280
+ // result, never a top-N createdAt-filtered view.
281
+ let listActiveCache: { result: TeamRunManifest[] | null; expiresAt: number } = { result: null, expiresAt: 0 };
282
+
283
+ function invalidateListActive(): void {
284
+ listActiveCache = { result: null, expiresAt: 0 };
285
+ }
286
+
269
287
  /**
270
288
  * RT-F3: filter to `status === "running"` BEFORE applying the limit so an
271
289
  * orphaned run that has been pushed past the top-N by recent successful
@@ -277,9 +295,19 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
277
295
  * silently drop "running" runs that fell past the top-N createdAt cutoff.
278
296
  * Still goes through parseManifestIfChanged for stat+size memoization, so
279
297
  * the per-run I/O cost is the same as list().
298
+ *
299
+ * FIND-03 perf: the full scan is memoized behind a 500ms TTL (same TTL as
300
+ * list()). fs.watch-driven scheduleListRefresh() invalidates the cache
301
+ * immediately so the next call re-scans. The cap is applied AFTER the
302
+ * cache lookup so a cached scan result can be sliced to ANY limit without
303
+ * re-scanning.
280
304
  */
281
305
  function listActive(limit: number): TeamRunManifest[] {
282
306
  const cap = Math.max(0, limit);
307
+ const now = Date.now();
308
+ if (listActiveCache.result !== null && listActiveCache.expiresAt > now) {
309
+ return listActiveCache.result.slice(0, cap);
310
+ }
283
311
  const parsedEntries = [
284
312
  ...roots.flatMap((root) => collectRoots(root)),
285
313
  ...activeRunEntries().map((entry) => ({
@@ -303,6 +331,7 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
303
331
  .filter((value): value is CachedManifest => value !== undefined)
304
332
  .map((value) => value.manifest)
305
333
  .filter((manifest) => manifest.status === "running");
334
+ listActiveCache = { result: running, expiresAt: now + ttlMs };
306
335
  return running.slice(0, cap);
307
336
  }
308
337
 
@@ -340,6 +369,7 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
340
369
  watchers = [];
341
370
  manifestIndex.clear();
342
371
  listCache.clear();
372
+ invalidateListActive();
343
373
  },
344
374
  };
345
375
  }
@@ -55,7 +55,8 @@ function extractUsage(value: unknown): ParsedPiUsage | undefined {
55
55
  turns: numberField(obj, ["turns", "turnCount", "turn_count"]),
56
56
  };
57
57
  if (Object.values(direct).some((entry) => entry !== undefined)) return direct;
58
- for (const key of ["usage", "tokenUsage", "tokens", "stats"]) {
58
+ // Pi --mode json nests usage under message for message_end / turn_end events.
59
+ for (const key of ["usage", "message", "tokenUsage", "tokens", "stats"]) {
59
60
  const nested = extractUsage(obj[key]);
60
61
  if (nested) return nested;
61
62
  }
@@ -189,7 +189,13 @@ function resolvePiCliScript(): string | undefined {
189
189
  const argv1 = process.argv[1];
190
190
  if (argv1) {
191
191
  const argvPath = path.isAbsolute(argv1) ? argv1 : path.resolve(argv1);
192
- if (isRunnableNodeScript(argvPath)) return argvPath;
192
+ // Only trust argv1 if we can confirm the current process is running from
193
+ // the pi-coding-agent package directory. Otherwise, when pi-crew is invoked
194
+ // from a standalone test script (process.argv[1] = test script path),
195
+ // resolvePiCliScript would incorrectly return the test script as the pi
196
+ // CLI. resolvePiPackageRoot walks up from argv1 looking for a package.json
197
+ // named @earendil-works/pi-coding-agent or @mariozechner/pi-coding-agent.
198
+ if (resolvePiPackageRoot() && isRunnableNodeScript(argvPath)) return argvPath;
193
199
  }
194
200
 
195
201
  // npm-global package dirs derived from `npm root -g` — placed BEFORE the