pi-crew 0.9.42 → 0.9.46
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +223 -0
- package/README.md +35 -0
- package/dist/build-meta.json +4778 -11087
- package/dist/index.mjs +54648 -65435
- package/dist/index.mjs.map +4 -4
- package/package.json +1 -1
- package/scripts/build-bundle.mjs +2 -1
- package/src/agents/agent-config.ts +1 -1
- package/src/agents/discover-agents.ts +1 -1
- package/src/config/config.ts +1 -1
- package/src/config/types.ts +0 -2
- package/src/extension/crew-vibes/config.ts +1 -1
- package/src/extension/crew-vibes/font-detect.ts +16 -3
- package/src/extension/cross-extension-rpc.ts +8 -4
- package/src/extension/registration/commands.ts +5 -1
- package/src/extension/registration/foreground-run-controller.ts +28 -0
- package/src/extension/registration/lifecycle-handlers.ts +34 -3
- package/src/extension/registration/subagent-manager-setup.ts +178 -59
- package/src/extension/run-import.ts +21 -1
- package/src/extension/team-tool/api.ts +4 -2
- package/src/extension/team-tool.ts +5 -0
- package/src/prompt/prompt-runtime.ts +15 -0
- package/src/runtime/async-runner.ts +9 -1
- package/src/runtime/child-pi-constants.ts +42 -0
- package/src/runtime/child-pi-kill.ts +180 -0
- package/src/runtime/child-pi-spawn.ts +234 -0
- package/src/runtime/child-pi-steering.ts +128 -0
- package/src/runtime/child-pi-streams.ts +296 -0
- package/src/runtime/child-pi-transcript.ts +169 -0
- package/src/runtime/child-pi.ts +75 -837
- package/src/runtime/compact-stages/tail-capture-stage.ts +1 -1
- package/src/runtime/dwf-state-store.ts +5 -0
- package/src/runtime/dynamic-workflow-context.ts +78 -18
- package/src/runtime/dynamic-workflow-runner.ts +18 -2
- package/src/runtime/goal-evaluator.ts +59 -27
- package/src/runtime/goal-loop-runner.ts +40 -11
- package/src/runtime/live-agent-manager.ts +18 -0
- package/src/runtime/manifest-cache.ts +30 -0
- package/src/runtime/pi-json-output.ts +2 -1
- package/src/runtime/pi-spawn.ts +7 -1
- package/src/runtime/resilient-edit.ts +16 -15
- package/src/runtime/role-permission.ts +27 -2
- package/src/runtime/run-coalesced-task-group.ts +90 -25
- package/src/runtime/task-packet.ts +1 -1
- package/src/runtime/task-runner/prompt-builder.ts +2 -2
- package/src/runtime/team-runner.ts +6 -1
- package/src/state/event-log.ts +89 -34
- package/src/state/locks.ts +185 -49
- package/src/state/mailbox.ts +165 -4
- package/src/state/run-metrics.ts +40 -12
- package/src/state/worker-atomic-writer.ts +1 -0
- package/src/ui/live-run-sidebar.ts +1 -9
- package/src/ui/run-dashboard.ts +1 -9
- package/src/utils/incremental-reader.ts +105 -0
- package/src/utils/paths.ts +1 -1
- package/src/utils/visual.ts +27 -91
- package/src/worktree/worktree-manager.ts +47 -19
- package/src/runtime/auto-resume.ts +0 -100
- package/src/runtime/notebook-helpers.ts +0 -88
- package/src/runtime/orphan-sentinel.ts +0 -7
|
@@ -59,7 +59,7 @@ export class TailCaptureStage implements ICompactStage {
|
|
|
59
59
|
// never contains a partial multi-byte sequence.
|
|
60
60
|
if (Buffer.byteLength(text, "utf-8") <= this.maxBytes) return text;
|
|
61
61
|
let tail = text.slice(Math.max(0, text.length - this.maxBytes));
|
|
62
|
-
while (Buffer.byteLength(tail, "utf-8") > this.maxBytes) tail = tail.slice(
|
|
62
|
+
while (Buffer.byteLength(tail, "utf-8") > this.maxBytes) tail = tail.slice(0, -1);
|
|
63
63
|
return this.marker ? `${this.marker}\n${tail}` : tail;
|
|
64
64
|
}
|
|
65
65
|
// Char cap mode.
|
|
@@ -28,6 +28,11 @@ export interface DwfCheckpointState {
|
|
|
28
28
|
logs: string[]; // capped copy (≤1000); the events log (dwf.log) is the durable source of truth
|
|
29
29
|
spent: number; // budget accumulator (round-14 P1-2)
|
|
30
30
|
agentCount: number;
|
|
31
|
+
/** PERS-1: per-agent-call idempotency cache. Maps a deterministic call ID to the
|
|
32
|
+
* cached result so a DWF resume skips already-completed agent calls (avoids
|
|
33
|
+
* duplicating artifacts/mailbox/tokens). Uses Record (not Map) for JSON
|
|
34
|
+
* serialization safety. */
|
|
35
|
+
completedAgentCalls?: Record<string, { text: string; usage?: { input: number; output: number } }>;
|
|
31
36
|
updatedAt: string;
|
|
32
37
|
}
|
|
33
38
|
|
|
@@ -35,6 +35,7 @@ import { logInternalError } from "../utils/internal-error.ts";
|
|
|
35
35
|
import { cleanupAgentWorktreeAsync, prepareAgentWorktreeAsync } from "../worktree/worktree-manager.ts";
|
|
36
36
|
import { runChildPi } from "./child-pi.ts";
|
|
37
37
|
import type { DwfCheckpointState } from "./dwf-state-store.ts";
|
|
38
|
+
import { withWorkerSlot } from "./global-worker-cap.ts";
|
|
38
39
|
import { mapConcurrent } from "./parallel-utils.ts";
|
|
39
40
|
import { parsePiJsonOutput } from "./pi-json-output.ts";
|
|
40
41
|
import { renderPlanTemplate } from "./plan-templates.ts";
|
|
@@ -269,6 +270,11 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
269
270
|
logs: opts.resumedState ? [...opts.resumedState.logs].slice(0, 1000) : [],
|
|
270
271
|
args: opts.args ?? {},
|
|
271
272
|
};
|
|
273
|
+
// PERS-1: per-agent-call idempotency cache. Hydrated from resumedState so a DWF
|
|
274
|
+
// resume skips already-completed agent calls (avoids duplicating artifacts/mailbox/tokens).
|
|
275
|
+
// Uses Record (not Map) for JSON serialization safety.
|
|
276
|
+
const completedAgentCalls: Record<string, { text: string; usage?: { input: number; output: number } }> =
|
|
277
|
+
opts.resumedState?.completedAgentCalls ?? {};
|
|
272
278
|
// round-14 P1-2: frozen budget surface. The closures read wfState.spent so the
|
|
273
279
|
// object stays live after Object.freeze(ctx). total is a snapshot primitive.
|
|
274
280
|
const budget = Object.freeze({
|
|
@@ -290,10 +296,35 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
290
296
|
// regardless of which return/throw path is taken.
|
|
291
297
|
let worktreePath: string | undefined;
|
|
292
298
|
let worktreeBranch: string | undefined;
|
|
299
|
+
// BDG-2: declared before the try so the catch block can un-reserve on failure.
|
|
300
|
+
const ESTIMATE = 4096;
|
|
301
|
+
let reserved = false;
|
|
293
302
|
try {
|
|
294
|
-
//
|
|
295
|
-
//
|
|
296
|
-
|
|
303
|
+
// PERS-1: per-agent-call idempotency. Compute a deterministic call ID
|
|
304
|
+
// from the call arguments and check if this call was already completed
|
|
305
|
+
// (e.g. during a previous run before a crash). If so, return the cached
|
|
306
|
+
// result without re-spawning the agent.
|
|
307
|
+
const callId = JSON.stringify({
|
|
308
|
+
role: call.role,
|
|
309
|
+
agent: call.agent,
|
|
310
|
+
prompt: call.prompt,
|
|
311
|
+
model: call.model,
|
|
312
|
+
schema: call.schema ? "yes" : "no",
|
|
313
|
+
});
|
|
314
|
+
const cached = completedAgentCalls[callId];
|
|
315
|
+
if (cached) {
|
|
316
|
+
return {
|
|
317
|
+
ok: true,
|
|
318
|
+
text: cached.text,
|
|
319
|
+
usage: cached.usage,
|
|
320
|
+
durationMs: 0,
|
|
321
|
+
};
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
// BDG-2: reserve-then-adjust budget. Before spawning, estimate the cost and
|
|
325
|
+
// reserve it by adding to wfState.spent. This prevents N concurrent calls from
|
|
326
|
+
// all seeing the same remaining budget and overspending.
|
|
327
|
+
if (budget.total !== null && budget.remaining() < ESTIMATE) {
|
|
297
328
|
return {
|
|
298
329
|
ok: false,
|
|
299
330
|
text: "",
|
|
@@ -301,6 +332,10 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
301
332
|
durationMs: 0,
|
|
302
333
|
};
|
|
303
334
|
}
|
|
335
|
+
// Reserve the estimate before spawning.
|
|
336
|
+
wfState.spent += ESTIMATE;
|
|
337
|
+
reserved = true;
|
|
338
|
+
|
|
304
339
|
const agentConfig = resolveAgentForRole(call.role, {
|
|
305
340
|
explicitAgent: call.agent,
|
|
306
341
|
team: opts.team,
|
|
@@ -353,20 +388,25 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
353
388
|
}
|
|
354
389
|
}
|
|
355
390
|
|
|
356
|
-
const childResult = await
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
391
|
+
const childResult = await withWorkerSlot(() =>
|
|
392
|
+
runChildPi({
|
|
393
|
+
cwd: agentCwd,
|
|
394
|
+
task,
|
|
395
|
+
agent: effectiveAgent,
|
|
396
|
+
model: call.model ?? opts.modelOverride ?? agentConfig.model,
|
|
397
|
+
skillPaths: undefined, // skills resolved via agent config + team-role plumbing
|
|
398
|
+
maxTurns: call.maxTurns,
|
|
399
|
+
graceTurns: call.graceTurns,
|
|
400
|
+
signal: opts.signal,
|
|
401
|
+
artifactsRoot: manifest.artifactsRoot,
|
|
402
|
+
runId: manifest.runId,
|
|
403
|
+
role: call.role ?? call.agent,
|
|
404
|
+
}),
|
|
405
|
+
);
|
|
369
406
|
if (childResult.exitCode !== 0 || childResult.error) {
|
|
407
|
+
// BDG-2: un-reserve the estimate on spawn failure.
|
|
408
|
+
wfState.spent -= ESTIMATE;
|
|
409
|
+
reserved = false;
|
|
370
410
|
return {
|
|
371
411
|
ok: false,
|
|
372
412
|
text: "",
|
|
@@ -376,8 +416,10 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
376
416
|
}
|
|
377
417
|
const parsed = parsePiJsonOutput(childResult.stdout);
|
|
378
418
|
// round-14 P1-2: accumulate this run's token usage into the workflow budget.
|
|
379
|
-
//
|
|
380
|
-
|
|
419
|
+
// BDG-2: adjust the reserve — subtract the estimate, add the actual usage.
|
|
420
|
+
// This correctly reduces spent when actualUsage < ESTIMATE.
|
|
421
|
+
wfState.spent += (parsed.usage?.input ?? 0) + (parsed.usage?.output ?? 0) - ESTIMATE;
|
|
422
|
+
reserved = false;
|
|
381
423
|
let text = parsed.finalText ?? "";
|
|
382
424
|
// Round-11 test fix: parsePiJsonOutput only extracts text from pi event stream
|
|
383
425
|
// ({type:"message_end", message:{role:"assistant", content:[...]}}). When the
|
|
@@ -400,6 +442,7 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
400
442
|
producer: "dynamic-workflow",
|
|
401
443
|
});
|
|
402
444
|
if (call.schema !== undefined && !extracted.structured) {
|
|
445
|
+
// BDG-2: un-reserve was already done above (reserved = false after adjust).
|
|
403
446
|
return {
|
|
404
447
|
ok: false,
|
|
405
448
|
text,
|
|
@@ -409,6 +452,11 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
409
452
|
durationMs: Date.now() - started,
|
|
410
453
|
};
|
|
411
454
|
}
|
|
455
|
+
// PERS-1: cache the successful result for idempotency on resume.
|
|
456
|
+
completedAgentCalls[callId] = {
|
|
457
|
+
text,
|
|
458
|
+
usage: { input: parsed.usage?.input ?? 0, output: parsed.usage?.output ?? 0 },
|
|
459
|
+
};
|
|
412
460
|
return {
|
|
413
461
|
ok: true,
|
|
414
462
|
text,
|
|
@@ -418,6 +466,8 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
418
466
|
durationMs: Date.now() - started,
|
|
419
467
|
};
|
|
420
468
|
} catch (error) {
|
|
469
|
+
// BDG-2: un-reserve the estimate on failure (only if still reserved).
|
|
470
|
+
if (reserved) wfState.spent -= ESTIMATE;
|
|
421
471
|
logInternalError("dynamic-workflow-context.agent", error, `runId=${manifest.runId}`);
|
|
422
472
|
return {
|
|
423
473
|
ok: false,
|
|
@@ -450,6 +500,7 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
450
500
|
logs: wfState.logs.slice(0, 1000),
|
|
451
501
|
spent: wfState.spent,
|
|
452
502
|
agentCount,
|
|
503
|
+
completedAgentCalls,
|
|
453
504
|
updatedAt: new Date().toISOString(),
|
|
454
505
|
});
|
|
455
506
|
} catch (checkpointError) {
|
|
@@ -709,6 +760,12 @@ export function makeWorkflowCtx(manifest: TeamRunManifest, opts: MakeWorkflowCtx
|
|
|
709
760
|
get: () => agentCount,
|
|
710
761
|
enumerable: false,
|
|
711
762
|
});
|
|
763
|
+
// PERS-1: completedAgentCalls is read-only from the runner; the agent() method
|
|
764
|
+
// is the only writer. Exposed so getWorkflowCheckpoint() can include it.
|
|
765
|
+
Object.defineProperty(ctx, "__completedAgentCalls", {
|
|
766
|
+
get: () => completedAgentCalls,
|
|
767
|
+
enumerable: false,
|
|
768
|
+
});
|
|
712
769
|
return ctx;
|
|
713
770
|
}
|
|
714
771
|
|
|
@@ -757,6 +814,9 @@ export function getWorkflowCheckpoint(ctx: WorkflowCtx): DwfCheckpointState {
|
|
|
757
814
|
logs: logs ?? [],
|
|
758
815
|
spent: ctx.budget.spent(),
|
|
759
816
|
agentCount: (ctx as unknown as { __agentCount?: number }).__agentCount ?? 0,
|
|
817
|
+
completedAgentCalls:
|
|
818
|
+
(ctx as unknown as { __completedAgentCalls?: Record<string, { text: string; usage?: { input: number; output: number } }> })
|
|
819
|
+
.__completedAgentCalls ?? {},
|
|
760
820
|
updatedAt: new Date().toISOString(),
|
|
761
821
|
};
|
|
762
822
|
}
|
|
@@ -19,6 +19,7 @@
|
|
|
19
19
|
|
|
20
20
|
import { readFileSync } from "node:fs";
|
|
21
21
|
import { join } from "node:path";
|
|
22
|
+
import { transformSync } from "esbuild";
|
|
22
23
|
import { writeArtifact } from "../state/artifact-store.ts";
|
|
23
24
|
import { appendEvent } from "../state/event-log.ts";
|
|
24
25
|
import type { TeamRunManifest, TeamTaskState } from "../state/types.ts";
|
|
@@ -113,7 +114,11 @@ async function loadWorkflowModule(scriptPath: string): Promise<DynamicWorkflowSc
|
|
|
113
114
|
// raw .dwf.ts file. This is the same source jiti will execute.
|
|
114
115
|
const scriptSource = readFileSync(scriptPath, "utf-8");
|
|
115
116
|
if (isDeterminismCheckEnabled()) {
|
|
116
|
-
|
|
117
|
+
// DISC-1: acorn can only parse JavaScript, not TypeScript. Real .dwf.ts files
|
|
118
|
+
// contain type annotations/imports that cause a silent parse error, so the
|
|
119
|
+
// determinism check never runs. Transpile to JS via esbuild first.
|
|
120
|
+
const js = transformSync(scriptSource, { loader: "ts", format: "esm" }).code;
|
|
121
|
+
assertDeterministicScript(js);
|
|
117
122
|
}
|
|
118
123
|
// jiti is the same loader async-runner.ts uses (resolveTypeScriptLoader). We require it
|
|
119
124
|
// lazily so this module stays importable in environments without jiti (type-only consumers).
|
|
@@ -166,9 +171,16 @@ export async function runDynamicWorkflow(input: RunDynamicWorkflowInput): Promis
|
|
|
166
171
|
});
|
|
167
172
|
}
|
|
168
173
|
|
|
174
|
+
// ERR-1: Create a timeout AbortController so we can abort spawned children on script timeout.
|
|
175
|
+
// Combine the input signal with the timeout controller's signal so EITHER source
|
|
176
|
+
// (external abort OR our timeout) propagates to runChildPi via the ctx.
|
|
177
|
+
// AbortSignal.any is available since Node 20.3 — the project requires Node 20+.
|
|
178
|
+
const timeoutController = new AbortController();
|
|
179
|
+
const combinedSignal = AbortSignal.any([signal, timeoutController.signal]);
|
|
180
|
+
|
|
169
181
|
const ctx = makeWorkflowCtx(manifest, {
|
|
170
182
|
concurrency: input.concurrency ?? workflow.maxConcurrency ?? 4,
|
|
171
|
-
signal,
|
|
183
|
+
signal: combinedSignal,
|
|
172
184
|
team: input.team,
|
|
173
185
|
modelOverride: input.modelOverride,
|
|
174
186
|
tokenBudget: input.tokenBudget ?? workflow.maxTokenBudget,
|
|
@@ -200,6 +212,10 @@ export async function runDynamicWorkflow(input: RunDynamicWorkflowInput): Promis
|
|
|
200
212
|
let timeoutHandle: NodeJS.Timeout | undefined;
|
|
201
213
|
const timeoutPromise = new Promise<never>((_, reject) => {
|
|
202
214
|
timeoutHandle = setTimeout(() => {
|
|
215
|
+
// ERR-1: abort the timeout controller so the abort propagates to
|
|
216
|
+
// runChildPi (via the combined signal passed to makeWorkflowCtx),
|
|
217
|
+
// which kills spawned children instead of letting them run.
|
|
218
|
+
timeoutController.abort();
|
|
203
219
|
reject(
|
|
204
220
|
new Error(
|
|
205
221
|
`Dynamic workflow script timed out after ${SCRIPT_TIMEOUT_MS}ms. The script may have spawned a child process that did not exit. Check for spawn/exec calls without proper stdio handling.`,
|
|
@@ -188,7 +188,9 @@ function tryParseDirectVerdict(stdout: string): { achieved: boolean; reason: str
|
|
|
188
188
|
* Returns a GoalVerdict. On any failure (non-zero exit, non-JSON, invalid shape),
|
|
189
189
|
* returns a `BLOCKED:`-prefixed verdict so the loop stops (§0c C6 fallback).
|
|
190
190
|
*/
|
|
191
|
-
export async function evaluateGoal(
|
|
191
|
+
export async function evaluateGoal(
|
|
192
|
+
input: EvaluateGoalInput,
|
|
193
|
+
): Promise<{ verdict: GoalVerdict; judgeUsage?: { totalTokens: number; inputTokens?: number; outputTokens?: number } }> {
|
|
192
194
|
const agent = synthesizeJudgeAgentConfig();
|
|
193
195
|
const task = buildJudgeTask(input);
|
|
194
196
|
const evaluatedAt = new Date().toISOString();
|
|
@@ -212,12 +214,15 @@ export async function evaluateGoal(input: EvaluateGoalInput): Promise<GoalVerdic
|
|
|
212
214
|
});
|
|
213
215
|
|
|
214
216
|
if (result.exitCode !== 0 || result.error) {
|
|
215
|
-
return
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
217
|
+
return {
|
|
218
|
+
verdict: blockedVerdict(
|
|
219
|
+
input.turn,
|
|
220
|
+
input.model,
|
|
221
|
+
evaluatedAt,
|
|
222
|
+
`judge spawn failed (exit=${result.exitCode}): ${result.error ?? result.stderr.slice(0, 200)}`,
|
|
223
|
+
),
|
|
224
|
+
judgeUsage: undefined,
|
|
225
|
+
};
|
|
221
226
|
}
|
|
222
227
|
|
|
223
228
|
const parsed = parsePiJsonOutput(result.stdout);
|
|
@@ -231,16 +236,22 @@ export async function evaluateGoal(input: EvaluateGoalInput): Promise<GoalVerdic
|
|
|
231
236
|
const direct = !finalText.trim() ? tryParseDirectVerdict(result.stdout) : undefined;
|
|
232
237
|
if (direct) {
|
|
233
238
|
return {
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
239
|
+
verdict: {
|
|
240
|
+
turn: input.turn,
|
|
241
|
+
achieved: direct.achieved,
|
|
242
|
+
reason: direct.reason,
|
|
243
|
+
evidenceRefs: direct.evidenceRefs,
|
|
244
|
+
evaluatorModel: input.model,
|
|
245
|
+
evaluatedAt,
|
|
246
|
+
},
|
|
247
|
+
judgeUsage: undefined,
|
|
240
248
|
};
|
|
241
249
|
}
|
|
242
250
|
if (!finalText.trim()) {
|
|
243
|
-
return
|
|
251
|
+
return {
|
|
252
|
+
verdict: blockedVerdict(input.turn, input.model, evaluatedAt, "judge produced no output"),
|
|
253
|
+
judgeUsage: undefined,
|
|
254
|
+
};
|
|
244
255
|
}
|
|
245
256
|
|
|
246
257
|
const extracted = extractStructuredResult(finalText);
|
|
@@ -252,27 +263,48 @@ export async function evaluateGoal(input: EvaluateGoalInput): Promise<GoalVerdic
|
|
|
252
263
|
})
|
|
253
264
|
: undefined;
|
|
254
265
|
if (!data || typeof data.achieved !== "boolean" || typeof data.reason !== "string") {
|
|
255
|
-
return
|
|
266
|
+
return {
|
|
267
|
+
verdict: blockedVerdict(
|
|
268
|
+
input.turn,
|
|
269
|
+
input.model,
|
|
270
|
+
evaluatedAt,
|
|
271
|
+
`judge output not valid verdict JSON: ${truncate(finalText, 200)}`,
|
|
272
|
+
),
|
|
273
|
+
judgeUsage: undefined,
|
|
274
|
+
};
|
|
256
275
|
}
|
|
257
276
|
const evidenceRefs = Array.isArray(data.evidenceRefs)
|
|
258
277
|
? data.evidenceRefs.filter((r): r is string => typeof r === "string")
|
|
259
278
|
: undefined;
|
|
279
|
+
const judgeUsage = parsed.usage
|
|
280
|
+
? {
|
|
281
|
+
totalTokens: (parsed.usage.input ?? 0) + (parsed.usage.output ?? 0),
|
|
282
|
+
inputTokens: parsed.usage.input,
|
|
283
|
+
outputTokens: parsed.usage.output,
|
|
284
|
+
}
|
|
285
|
+
: undefined;
|
|
260
286
|
return {
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
287
|
+
verdict: {
|
|
288
|
+
turn: input.turn,
|
|
289
|
+
achieved: data.achieved,
|
|
290
|
+
reason: data.reason,
|
|
291
|
+
evidenceRefs,
|
|
292
|
+
evaluatorModel: input.model,
|
|
293
|
+
evaluatedAt,
|
|
294
|
+
},
|
|
295
|
+
judgeUsage,
|
|
267
296
|
};
|
|
268
297
|
} catch (error) {
|
|
269
298
|
logInternalError("goal-evaluator.evaluateGoal", error, `turn=${input.turn}`);
|
|
270
|
-
return
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
299
|
+
return {
|
|
300
|
+
verdict: blockedVerdict(
|
|
301
|
+
input.turn,
|
|
302
|
+
input.model,
|
|
303
|
+
evaluatedAt,
|
|
304
|
+
`judge threw: ${error instanceof Error ? error.message : String(error)}`,
|
|
305
|
+
),
|
|
306
|
+
judgeUsage: undefined,
|
|
307
|
+
};
|
|
276
308
|
}
|
|
277
309
|
}
|
|
278
310
|
|
|
@@ -66,21 +66,29 @@ export const stubGoalEvaluator = async (
|
|
|
66
66
|
_m?: import("../state/types.ts").TeamRunManifest,
|
|
67
67
|
_t?: import("../state/types.ts").TeamTaskState[],
|
|
68
68
|
_s?: AbortSignal,
|
|
69
|
-
): Promise<
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
69
|
+
): Promise<GoalEvaluatorResult> => ({
|
|
70
|
+
verdict: {
|
|
71
|
+
turn: goal.turnsUsed,
|
|
72
|
+
achieved: false,
|
|
73
|
+
reason: `not-achieved: stub evaluator (P0). Turn ${goal.turnsUsed}/${goal.maxTurns} completed; P1 will judge against objective + verification.`,
|
|
74
|
+
evaluatorModel: "stub",
|
|
75
|
+
evaluatedAt: new Date().toISOString(),
|
|
76
|
+
},
|
|
77
|
+
judgeUsage: undefined,
|
|
75
78
|
});
|
|
76
79
|
|
|
80
|
+
export type GoalEvaluatorResult = {
|
|
81
|
+
verdict: GoalVerdict;
|
|
82
|
+
judgeUsage?: { totalTokens: number; inputTokens?: number; outputTokens?: number };
|
|
83
|
+
};
|
|
84
|
+
|
|
77
85
|
export type GoalEvaluatorFn = (
|
|
78
86
|
goal: GoalLoopState,
|
|
79
87
|
turnRunId: string,
|
|
80
88
|
turnManifest: import("../state/types.ts").TeamRunManifest,
|
|
81
89
|
turnTasks: import("../state/types.ts").TeamTaskState[],
|
|
82
90
|
signal: AbortSignal,
|
|
83
|
-
) => Promise<
|
|
91
|
+
) => Promise<GoalEvaluatorResult>;
|
|
84
92
|
|
|
85
93
|
/**
|
|
86
94
|
* Production evaluator (P1): bundles turn evidence + calls the LLM judge.
|
|
@@ -94,7 +102,7 @@ export const realGoalEvaluator = async (
|
|
|
94
102
|
turnManifest: import("../state/types.ts").TeamRunManifest,
|
|
95
103
|
turnTasks: import("../state/types.ts").TeamTaskState[],
|
|
96
104
|
signal: AbortSignal,
|
|
97
|
-
): Promise<
|
|
105
|
+
): Promise<GoalEvaluatorResult> => {
|
|
98
106
|
const transcriptPath = deriveTranscriptPath(turnManifest.artifactsRoot, turnTasks);
|
|
99
107
|
// Fix round-7 F1: execute verification commands (if configured) so the judge has real evidence.
|
|
100
108
|
// Previously bundleEvidence received `undefined` — the judge was told commands "MUST pass"
|
|
@@ -676,17 +684,38 @@ export async function runGoalLoop(input: RunGoalLoopInput): Promise<RunGoalLoopR
|
|
|
676
684
|
logInternalError("goal-loop.saveTurnTasks", error, `turnRunId=${created.manifest.runId}`);
|
|
677
685
|
}
|
|
678
686
|
|
|
687
|
+
// GL-1: if the turn ended in blocked/failed status, stop the loop without spawning the judge.
|
|
688
|
+
// Previously the loop continued to evaluateGoal on blocked/failed turns, burning maxTurns.
|
|
689
|
+
const turnStatus = turnResult.manifest.status;
|
|
690
|
+
if (turnStatus === "blocked" || turnStatus === "failed") {
|
|
691
|
+
goal = safeSetStatus(store, goal.goalId, "blocked", goal, eventsPath);
|
|
692
|
+
appendEvent(eventsPath, {
|
|
693
|
+
type: "goal.turn_terminal_status",
|
|
694
|
+
runId: manifest.runId,
|
|
695
|
+
data: {
|
|
696
|
+
goalId: goal.goalId,
|
|
697
|
+
turn: turnIndex,
|
|
698
|
+
turnRunId: created.manifest.runId,
|
|
699
|
+
turnStatus,
|
|
700
|
+
},
|
|
701
|
+
});
|
|
702
|
+
break;
|
|
703
|
+
}
|
|
704
|
+
|
|
679
705
|
// ── BUDGET accumulation (§0c C2: collectRunMetrics) ──────────────────────
|
|
680
|
-
const
|
|
706
|
+
const turnBudget = accumulateBudget(goal, created.manifest.runId);
|
|
681
707
|
|
|
682
708
|
// ── EVALUATE (P1: real LLM judge; pass turn manifest + tasks for transcript lookup) ──
|
|
683
|
-
const verdict = await evaluator(
|
|
684
|
-
{ ...goal, budgetUsed:
|
|
709
|
+
const { verdict, judgeUsage } = await evaluator(
|
|
710
|
+
{ ...goal, budgetUsed: turnBudget },
|
|
685
711
|
created.manifest.runId,
|
|
686
712
|
turnResult.manifest,
|
|
687
713
|
turnResult.tasks,
|
|
688
714
|
signal,
|
|
689
715
|
);
|
|
716
|
+
// BDG-1: accumulate judge token usage into budget alongside the turn's worker tokens.
|
|
717
|
+
const judgeTokens = judgeUsage?.totalTokens ?? 0;
|
|
718
|
+
const updatedBudget = turnBudget + judgeTokens;
|
|
690
719
|
const historyEntry = {
|
|
691
720
|
runId: created.manifest.runId,
|
|
692
721
|
outcome: verdict.achieved ? "achieved" : "not-achieved",
|
|
@@ -268,6 +268,24 @@ export function evictStaleLiveAgentHandles(now = Date.now()): number {
|
|
|
268
268
|
// before evicting. Only evict if the process is dead, to avoid evicting
|
|
269
269
|
// slow-but-alive agents.
|
|
270
270
|
const sessionPid = (handle.session as Record<string, unknown>)?.pid as number | undefined;
|
|
271
|
+
if (sessionPid === undefined) {
|
|
272
|
+
// In-process live-session agents have no pid — the Pi SDK returns
|
|
273
|
+
// in-process objects, not child processes. checkProcessLiveness(undefined)
|
|
274
|
+
// returns {alive:false} which would wrongly evict a running agent.
|
|
275
|
+
// Check session-level activity signals first. If the session is
|
|
276
|
+
// actively streaming or has pending messages, it is alive — skip.
|
|
277
|
+
// Otherwise, rely on the idle timeout alone: do NOT dispose just
|
|
278
|
+
// because pid is undefined.
|
|
279
|
+
const session = handle.session as Record<string, unknown>;
|
|
280
|
+
const isStreaming = session?.isStreaming === true;
|
|
281
|
+
const pendingCount = (session?.pendingMessageCount as number | undefined) ?? 0;
|
|
282
|
+
if (isStreaming || pendingCount > 0) {
|
|
283
|
+
// Session is active — skip eviction.
|
|
284
|
+
continue;
|
|
285
|
+
}
|
|
286
|
+
// No activity signals and no pid — do NOT evict based on PID liveness.
|
|
287
|
+
continue;
|
|
288
|
+
}
|
|
271
289
|
const liveness = checkProcessLiveness(sessionPid);
|
|
272
290
|
if (!liveness.alive) {
|
|
273
291
|
liveAgents.delete(agentId);
|
|
@@ -164,6 +164,7 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
|
|
|
164
164
|
manifestIndex.clear();
|
|
165
165
|
}
|
|
166
166
|
listCache.clear();
|
|
167
|
+
invalidateListActive();
|
|
167
168
|
}
|
|
168
169
|
|
|
169
170
|
function scheduleListRefresh(): void {
|
|
@@ -174,12 +175,17 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
|
|
|
174
175
|
const timer = listTimer;
|
|
175
176
|
listTimer = undefined;
|
|
176
177
|
listCache.clear();
|
|
178
|
+
invalidateListActive();
|
|
177
179
|
timer?.unref();
|
|
178
180
|
}, ttlMs);
|
|
179
181
|
// Unref immediately so the timer never blocks process exit (defense in
|
|
180
182
|
// depth: the in-callback unref above may not run if shutdown happens
|
|
181
183
|
// before the timer fires).
|
|
182
184
|
listTimer.unref();
|
|
185
|
+
// FIND-03: invalidate the listActive() cache eagerly on every watcher
|
|
186
|
+
// tick. The TTL is the fallback for missed events; the watcher-driven
|
|
187
|
+
// path gives the tightest possible invalidation.
|
|
188
|
+
invalidateListActive();
|
|
183
189
|
}
|
|
184
190
|
|
|
185
191
|
function loadManifest(runId: string, rootsToCheck: string[]): CachedManifest | undefined {
|
|
@@ -266,6 +272,18 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
|
|
|
266
272
|
return undefined;
|
|
267
273
|
}
|
|
268
274
|
|
|
275
|
+
// FIND-03: short-TTL cache for listActive(). Mirrors the listCache pattern
|
|
276
|
+
// used by list(): we cache the un-capped running set and apply the caller's
|
|
277
|
+
// `limit` post-hoc on every return. Storing the full set (not a
|
|
278
|
+
// limit-sliced array) is what preserves the RT-F3 contract — callers with
|
|
279
|
+
// different `limit` values all see the same underlying "every running run"
|
|
280
|
+
// result, never a top-N createdAt-filtered view.
|
|
281
|
+
let listActiveCache: { result: TeamRunManifest[] | null; expiresAt: number } = { result: null, expiresAt: 0 };
|
|
282
|
+
|
|
283
|
+
function invalidateListActive(): void {
|
|
284
|
+
listActiveCache = { result: null, expiresAt: 0 };
|
|
285
|
+
}
|
|
286
|
+
|
|
269
287
|
/**
|
|
270
288
|
* RT-F3: filter to `status === "running"` BEFORE applying the limit so an
|
|
271
289
|
* orphaned run that has been pushed past the top-N by recent successful
|
|
@@ -277,9 +295,19 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
|
|
|
277
295
|
* silently drop "running" runs that fell past the top-N createdAt cutoff.
|
|
278
296
|
* Still goes through parseManifestIfChanged for stat+size memoization, so
|
|
279
297
|
* the per-run I/O cost is the same as list().
|
|
298
|
+
*
|
|
299
|
+
* FIND-03 perf: the full scan is memoized behind a 500ms TTL (same TTL as
|
|
300
|
+
* list()). fs.watch-driven scheduleListRefresh() invalidates the cache
|
|
301
|
+
* immediately so the next call re-scans. The cap is applied AFTER the
|
|
302
|
+
* cache lookup so a cached scan result can be sliced to ANY limit without
|
|
303
|
+
* re-scanning.
|
|
280
304
|
*/
|
|
281
305
|
function listActive(limit: number): TeamRunManifest[] {
|
|
282
306
|
const cap = Math.max(0, limit);
|
|
307
|
+
const now = Date.now();
|
|
308
|
+
if (listActiveCache.result !== null && listActiveCache.expiresAt > now) {
|
|
309
|
+
return listActiveCache.result.slice(0, cap);
|
|
310
|
+
}
|
|
283
311
|
const parsedEntries = [
|
|
284
312
|
...roots.flatMap((root) => collectRoots(root)),
|
|
285
313
|
...activeRunEntries().map((entry) => ({
|
|
@@ -303,6 +331,7 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
|
|
|
303
331
|
.filter((value): value is CachedManifest => value !== undefined)
|
|
304
332
|
.map((value) => value.manifest)
|
|
305
333
|
.filter((manifest) => manifest.status === "running");
|
|
334
|
+
listActiveCache = { result: running, expiresAt: now + ttlMs };
|
|
306
335
|
return running.slice(0, cap);
|
|
307
336
|
}
|
|
308
337
|
|
|
@@ -340,6 +369,7 @@ export function createManifestCache(cwd: string, options: ManifestCacheOptions =
|
|
|
340
369
|
watchers = [];
|
|
341
370
|
manifestIndex.clear();
|
|
342
371
|
listCache.clear();
|
|
372
|
+
invalidateListActive();
|
|
343
373
|
},
|
|
344
374
|
};
|
|
345
375
|
}
|
|
@@ -55,7 +55,8 @@ function extractUsage(value: unknown): ParsedPiUsage | undefined {
|
|
|
55
55
|
turns: numberField(obj, ["turns", "turnCount", "turn_count"]),
|
|
56
56
|
};
|
|
57
57
|
if (Object.values(direct).some((entry) => entry !== undefined)) return direct;
|
|
58
|
-
|
|
58
|
+
// Pi --mode json nests usage under message for message_end / turn_end events.
|
|
59
|
+
for (const key of ["usage", "message", "tokenUsage", "tokens", "stats"]) {
|
|
59
60
|
const nested = extractUsage(obj[key]);
|
|
60
61
|
if (nested) return nested;
|
|
61
62
|
}
|
package/src/runtime/pi-spawn.ts
CHANGED
|
@@ -189,7 +189,13 @@ function resolvePiCliScript(): string | undefined {
|
|
|
189
189
|
const argv1 = process.argv[1];
|
|
190
190
|
if (argv1) {
|
|
191
191
|
const argvPath = path.isAbsolute(argv1) ? argv1 : path.resolve(argv1);
|
|
192
|
-
if
|
|
192
|
+
// Only trust argv1 if we can confirm the current process is running from
|
|
193
|
+
// the pi-coding-agent package directory. Otherwise, when pi-crew is invoked
|
|
194
|
+
// from a standalone test script (process.argv[1] = test script path),
|
|
195
|
+
// resolvePiCliScript would incorrectly return the test script as the pi
|
|
196
|
+
// CLI. resolvePiPackageRoot walks up from argv1 looking for a package.json
|
|
197
|
+
// named @earendil-works/pi-coding-agent or @mariozechner/pi-coding-agent.
|
|
198
|
+
if (resolvePiPackageRoot() && isRunnableNodeScript(argvPath)) return argvPath;
|
|
193
199
|
}
|
|
194
200
|
|
|
195
201
|
// npm-global package dirs derived from `npm root -g` — placed BEFORE the
|