pi-plans 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +5 -12
- package/README.md +5 -5
- package/agents/execution-reviewer.md +40 -0
- package/index.ts +13 -23
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +13 -7
- package/references/plan-artifact-template.md +11 -1
- package/references/state-and-config.md +3 -3
- package/scripts/validate.ts +20 -3
- package/src/auditor.ts +157 -56
- package/src/code-graph/commands.ts +6 -1
- package/src/dashboard.ts +56 -10
- package/src/exec.ts +614 -126
- package/src/plan.ts +1 -1
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +13 -3
- package/src/resume.ts +5 -1
- package/src/staleness.ts +53 -0
- package/src/state.ts +1 -0
- package/src/task-tool.ts +1 -1
- package/src/tasks.ts +39 -5
- package/src/ui-language.ts +4 -0
- package/src/workflow-state.ts +18 -5
- package/tests/auditor.test.ts +116 -17
- package/tests/dashboard.test.ts +135 -1
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +198 -44
- package/tests/extension-load.test.ts +1 -1
- package/tests/resume-lifecycle.test.ts +5 -1
- package/tests/resume.test.ts +6 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +4 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/workflow-state.test.ts +65 -0
- package/tools/execute-plan.ts +12 -5
- package/tools/plans.ts +1 -1
package/tests/exec.test.ts
CHANGED
|
@@ -11,6 +11,7 @@ import * as os from "node:os";
|
|
|
11
11
|
import * as path from "node:path";
|
|
12
12
|
import { after, before, describe, it } from "node:test";
|
|
13
13
|
import {
|
|
14
|
+
__awaitReviewRoundForTests,
|
|
14
15
|
__setAuditRunnerForTests,
|
|
15
16
|
executionContextMessage,
|
|
16
17
|
getExecution,
|
|
@@ -23,6 +24,7 @@ import {
|
|
|
23
24
|
toggleDashboardExpanded,
|
|
24
25
|
updateStatusWidget,
|
|
25
26
|
} from "../src/exec.ts";
|
|
27
|
+
import { REVIEW_MAX_ROUNDS } from "../src/auditor.ts";
|
|
26
28
|
import { setMessagingApi } from "../src/messaging.ts";
|
|
27
29
|
import { applyTaskUpdate } from "../src/task-tool.ts";
|
|
28
30
|
import { flattenTaskViews } from "../src/tasks.ts";
|
|
@@ -157,7 +159,6 @@ describe("task-tree execution core", () => {
|
|
|
157
159
|
return item.id;
|
|
158
160
|
}),
|
|
159
161
|
failed: [],
|
|
160
|
-
rolledBack: [],
|
|
161
162
|
report: "all pass",
|
|
162
163
|
}));
|
|
163
164
|
const { workdir, planPath, runId } = freshWorkdir();
|
|
@@ -186,7 +187,7 @@ describe("task-tree execution core", () => {
|
|
|
186
187
|
round += 1;
|
|
187
188
|
const failed = checklist.filter((item) => item.id === "VC-002").map((item) => item.id);
|
|
188
189
|
// Simulate the pure outcome: VC-002 fails; its covered tasks roll back.
|
|
189
|
-
return { round, passed: ["VC-001"], failed,
|
|
190
|
+
return { round, passed: ["VC-001"], failed, report: "VC-002 fails" };
|
|
190
191
|
});
|
|
191
192
|
const { workdir, planPath, runId } = freshWorkdir();
|
|
192
193
|
const { ctx } = await start(planPath, workdir);
|
|
@@ -300,7 +301,7 @@ describe("task-tree execution core", () => {
|
|
|
300
301
|
// Mirror applyAuditOutcome: passed checks are marked done.
|
|
301
302
|
const vc1 = checklist.find((item) => item.id === "VC-001");
|
|
302
303
|
if (vc1) vc1.done = true;
|
|
303
|
-
return { round: 1, passed: ["VC-001"], failed: ["VC-002"],
|
|
304
|
+
return { round: 1, passed: ["VC-001"], failed: ["VC-002"], report: "VC-002 fails: core tests missing" };
|
|
304
305
|
});
|
|
305
306
|
const { workdir, planPath, runId } = freshWorkdir();
|
|
306
307
|
const { ctx } = await start(planPath, workdir);
|
|
@@ -323,10 +324,10 @@ describe("task-tree execution core", () => {
|
|
|
323
324
|
await stopExecution(ctx, "test teardown");
|
|
324
325
|
});
|
|
325
326
|
|
|
326
|
-
it("
|
|
327
|
+
it("review round cap: every mode pauses with an in-band signal (v0.8)", async () => {
|
|
327
328
|
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
328
329
|
const failed = checklist.filter((item) => !["VC-001"].includes(item.id)).map((item) => item.id);
|
|
329
|
-
return { round: 1, passed: ["VC-001"], failed,
|
|
330
|
+
return { round: 1, passed: ["VC-001"], failed, report: "still failing" };
|
|
330
331
|
});
|
|
331
332
|
const interactive = await (async () => {
|
|
332
333
|
const { workdir, planPath, runId } = freshWorkdir();
|
|
@@ -335,36 +336,43 @@ describe("task-tree execution core", () => {
|
|
|
335
336
|
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
336
337
|
persistTaskProgress(ctx);
|
|
337
338
|
}
|
|
338
|
-
// Simulate
|
|
339
|
-
getExecution()!.audit.rounds =
|
|
339
|
+
// Simulate an exhausted budget persisted from earlier attempts.
|
|
340
|
+
getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
|
|
340
341
|
getExecution()!.audit.failed = ["VC-002"];
|
|
341
|
-
mutateCheckpoint(workdir, runId!, (cp) => applyExecutionProgress(cp, { audit: { rounds:
|
|
342
|
+
mutateCheckpoint(workdir, runId!, (cp) => applyExecutionProgress(cp, { audit: { rounds: REVIEW_MAX_ROUNDS, lastResult: "VC-002" } }));
|
|
342
343
|
const snapshot = getExecution();
|
|
343
344
|
const ctxUi = { ...makeCtx(workdir), mode: "tui" as const };
|
|
344
345
|
await restoreFromSession(ctxUi, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
346
|
+
await __awaitReviewRoundForTests(); // detached chain: pause lands inside it
|
|
345
347
|
const ex = getExecution()!;
|
|
346
|
-
//
|
|
348
|
+
// v0.8 (Q-A): interactive AND headless both pause — fail-closed, never a silent stop.
|
|
347
349
|
assert.equal(ex.stall.paused, true, "interactive cap pauses");
|
|
348
|
-
assert.match(ex.stall.pausedReason ?? "", /exhausted
|
|
350
|
+
assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
|
|
351
|
+
assert.ok(
|
|
352
|
+
ctxUi.entries.some((e) => e.customType === "pi-plans-review-paused"),
|
|
353
|
+
"the pause lands as an in-band message every mode can read",
|
|
354
|
+
);
|
|
349
355
|
await stopExecution(ctxUi, "test teardown");
|
|
350
356
|
return loadCheckpoint(workdir, runId!);
|
|
351
357
|
})();
|
|
352
358
|
assert.ok(interactive.status === "ok");
|
|
353
|
-
// Headless: no
|
|
359
|
+
// Headless: same pause, execution state kept (no more bounded stop).
|
|
354
360
|
const { workdir: wd2, planPath: pp2, runId: r2 } = freshWorkdir();
|
|
355
361
|
const { ctx: ctx3 } = await start(pp2, wd2);
|
|
356
362
|
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
357
363
|
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
358
364
|
persistTaskProgress(ctx3);
|
|
359
365
|
}
|
|
360
|
-
getExecution()!.audit.rounds =
|
|
366
|
+
getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
|
|
361
367
|
getExecution()!.audit.failed = ["VC-002"];
|
|
362
368
|
const headlessCtx = { ...makeCtx(wd2), hasUI: false } as never;
|
|
363
369
|
await restoreFromSession(headlessCtx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
364
|
-
assert.
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
assert.
|
|
370
|
+
assert.notEqual(getExecution(), null, "headless cap pauses and keeps execution state");
|
|
371
|
+
assert.equal(getExecution()!.stall.paused, true, "headless pauses too (v0.8)");
|
|
372
|
+
const paused = loadCheckpoint(wd2, r2!);
|
|
373
|
+
assert.ok(paused.status === "ok");
|
|
374
|
+
assert.match(paused.checkpoint.execution?.pausedReason ?? "", /execution review exhausted 5 rounds/);
|
|
375
|
+
await stopExecution(makeCtx(wd2), "test teardown");
|
|
368
376
|
__setAuditRunnerForTests(null);
|
|
369
377
|
});
|
|
370
378
|
|
|
@@ -383,9 +391,10 @@ describe("task-tree execution core", () => {
|
|
|
383
391
|
assert.equal(getExecution(), null, "delegate orphans never resume without a fresh handoff");
|
|
384
392
|
});
|
|
385
393
|
|
|
386
|
-
it("resuming
|
|
387
|
-
// Round 2 F-001 regression nail: cap → pause →
|
|
388
|
-
//
|
|
394
|
+
it("resuming a review-cap pause via /plans-execute grants a fresh budget and completes (v0.8)", async () => {
|
|
395
|
+
// Round 2 F-001 regression nail, v0.8 form: cap → pause → the explicit
|
|
396
|
+
// /plans-execute surface (resumeActiveExecution) resets rounds → the
|
|
397
|
+
// review runs again and can now complete the run. Ordinary input must NOT.
|
|
389
398
|
let runnerCalls = 0;
|
|
390
399
|
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
391
400
|
runnerCalls += 1;
|
|
@@ -396,7 +405,6 @@ describe("task-tree execution core", () => {
|
|
|
396
405
|
return item.id;
|
|
397
406
|
}),
|
|
398
407
|
failed: [],
|
|
399
|
-
rolledBack: [],
|
|
400
408
|
report: "all pass",
|
|
401
409
|
};
|
|
402
410
|
});
|
|
@@ -406,30 +414,40 @@ describe("task-tree execution core", () => {
|
|
|
406
414
|
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
407
415
|
persistTaskProgress(ctx);
|
|
408
416
|
}
|
|
409
|
-
// Exhaust the budget, then pause at the cap exactly like
|
|
410
|
-
getExecution()!.audit.rounds =
|
|
417
|
+
// Exhaust the budget, then pause at the cap exactly like the loop does.
|
|
418
|
+
getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
|
|
411
419
|
getExecution()!.audit.failed = ["VC-001", "VC-002"];
|
|
412
420
|
const ctxTui = { ...makeCtx(workdir), mode: "tui" as const };
|
|
413
421
|
const snapshot = getExecution();
|
|
414
422
|
await restoreFromSession(ctxTui, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
assert.
|
|
418
|
-
|
|
423
|
+
await __awaitReviewRoundForTests();
|
|
424
|
+
let ex = getExecution()!;
|
|
425
|
+
assert.equal(ex.stall.paused, true, "cap pauses");
|
|
426
|
+
assert.match(ex.stall.pausedReason ?? "", /^execution review exhausted/);
|
|
427
|
+
// Ordinary input does NOT lift a review-cap pause (CF2-004).
|
|
428
|
+
const beforeRounds = ex.audit.rounds;
|
|
429
|
+
const { resumeGoalWaitIfPaused } = await import("../src/exec.ts");
|
|
430
|
+
const viaInput = resumeGoalWaitIfPaused(ctxTui);
|
|
431
|
+
assert.equal(viaInput, false, "input never lifts a review-cap pause");
|
|
432
|
+
assert.equal(getExecution()!.stall.paused, true, "still paused after input");
|
|
433
|
+
assert.equal(getExecution()!.audit.rounds, beforeRounds, "budget survives input");
|
|
434
|
+
// The explicit surface grants the fresh budget and completes the run.
|
|
419
435
|
const { resumeActiveExecution } = await import("../src/exec.ts");
|
|
420
436
|
const resumed = resumeActiveExecution(ctxTui);
|
|
421
437
|
assert.equal(resumed, true);
|
|
422
|
-
assert.equal(getExecution()!.audit.rounds, 0, "resume grants a fresh
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
assert.equal(runnerCalls, 1, "audit re-ran after the resume (fresh budget)");
|
|
438
|
+
assert.equal(getExecution()!.audit.rounds, 0, "resume grants a fresh review budget");
|
|
439
|
+
await __awaitReviewRoundForTests(); // detached grant chain completes the run
|
|
440
|
+
assert.equal(runnerCalls, 1, "review re-ran after the resume (fresh budget)");
|
|
426
441
|
const final = loadCheckpoint(workdir, runId!);
|
|
427
442
|
assert.ok(final.status === "ok");
|
|
428
443
|
assert.equal(final.checkpoint.phase, "completed");
|
|
429
444
|
__setAuditRunnerForTests(null);
|
|
430
445
|
});
|
|
431
446
|
|
|
432
|
-
it("a null
|
|
447
|
+
it("a null review outcome stays undeterminable, self-schedules to the cap, and pauses (v0.8)", async () => {
|
|
448
|
+
// An infra failure is not evidence of bad work: it must not complete the
|
|
449
|
+
// run, roll work back, or accuse the agent — and the loop retries it
|
|
450
|
+
// internally until the budget is spent, then pauses fail-closed.
|
|
433
451
|
__setAuditRunnerForTests(async () => null);
|
|
434
452
|
const { workdir, planPath, runId } = freshWorkdir();
|
|
435
453
|
const { ctx } = await start(planPath, workdir);
|
|
@@ -440,25 +458,36 @@ describe("task-tree execution core", () => {
|
|
|
440
458
|
const snapshot = getExecution();
|
|
441
459
|
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
442
460
|
const ex = getExecution()!;
|
|
443
|
-
assert.deepEqual(ex.audit.
|
|
461
|
+
assert.deepEqual(ex.audit.undeterminable, ["VC-001", "VC-002"], "unreported checks are undeterminable");
|
|
462
|
+
assert.deepEqual(ex.audit.failed, [], "nothing was judged wrong");
|
|
444
463
|
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
445
|
-
assert.equal(ex.tasks.find((t) => t.id === id)?.status, "
|
|
464
|
+
assert.equal(ex.tasks.find((t) => t.id === id)?.status, "complete", `${id} not rolled back`);
|
|
446
465
|
}
|
|
466
|
+
assert.equal(ex.audit.rounds, REVIEW_MAX_ROUNDS, "the retry chain spent the whole budget");
|
|
467
|
+
assert.equal(ex.stall.paused, true, "the loop pauses at the cap");
|
|
468
|
+
assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
|
|
469
|
+
assert.equal(
|
|
470
|
+
ctx.entries.filter((e) => e.customType === "pi-plans-audit-undeterminable").length,
|
|
471
|
+
0,
|
|
472
|
+
"undeterminable rounds never wake the agent (self-scheduled retries)",
|
|
473
|
+
);
|
|
474
|
+
assert.ok(ctx.entries.some((e) => e.customType === "pi-plans-review-paused"), "the pause is in-band");
|
|
447
475
|
const load = loadCheckpoint(workdir, runId!);
|
|
448
476
|
assert.ok(load.status === "ok");
|
|
449
|
-
assert.equal(load.checkpoint.
|
|
450
|
-
assert.
|
|
477
|
+
assert.equal(load.checkpoint.phase, "executing", "an unreadable round must not complete the run");
|
|
478
|
+
assert.equal(load.checkpoint.execution?.audit?.rounds, REVIEW_MAX_ROUNDS);
|
|
451
479
|
__setAuditRunnerForTests(null);
|
|
452
480
|
await stopExecution(ctx, "test teardown");
|
|
453
481
|
});
|
|
454
482
|
|
|
455
|
-
it("a runner passing a strict subset
|
|
483
|
+
it("a runner passing a strict subset credits the pass and keeps the rest undeterminable", async () => {
|
|
456
484
|
// The runner reports VC-001 passed and claims zero failures — VC-002
|
|
457
|
-
// is simply missing from its report
|
|
485
|
+
// is simply missing from its report, so it must NOT complete the run,
|
|
486
|
+
// must NOT be called failed, and must NOT roll Task-3 back.
|
|
458
487
|
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
459
488
|
const vc1 = checklist.find((item) => item.id === "VC-001");
|
|
460
489
|
if (vc1) vc1.done = true;
|
|
461
|
-
return { round: 1, passed: ["VC-001"], failed: [],
|
|
490
|
+
return { round: 1, passed: ["VC-001"], failed: [], undeterminable: ["VC-002"], report: "partial report" };
|
|
462
491
|
});
|
|
463
492
|
const { workdir, planPath } = freshWorkdir();
|
|
464
493
|
const { ctx } = await start(planPath, workdir);
|
|
@@ -469,12 +498,137 @@ describe("task-tree execution core", () => {
|
|
|
469
498
|
const snapshot = getExecution();
|
|
470
499
|
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
471
500
|
const ex = getExecution()!;
|
|
472
|
-
assert.deepEqual(ex.audit.
|
|
473
|
-
assert.
|
|
501
|
+
assert.deepEqual(ex.audit.undeterminable, ["VC-002"], "the unreported check stays pending judgement");
|
|
502
|
+
assert.deepEqual(ex.audit.failed, []);
|
|
503
|
+
assert.equal(ex.tasks.find((t) => t.id === "Task-3")?.status, "complete", "partial pass does not roll back");
|
|
504
|
+
__setAuditRunnerForTests(null);
|
|
505
|
+
await stopExecution(ctx, "test teardown");
|
|
506
|
+
});
|
|
507
|
+
|
|
508
|
+
it("an emphasized pass verdict completes the run with no rollback", async () => {
|
|
509
|
+
// The production incident: the auditor wrote `verdict: **pass**` for
|
|
510
|
+
// every check and the run rolled everything back. Emphasis must parse,
|
|
511
|
+
// and a fully-passing round must complete rather than pause.
|
|
512
|
+
__setAuditRunnerForTests(async ({ checklist }) => ({
|
|
513
|
+
round: 1,
|
|
514
|
+
passed: checklist.map((item) => item.id),
|
|
515
|
+
failed: [],
|
|
516
|
+
undeterminable: [],
|
|
517
|
+
report: "- `VC-001` — verdict: **pass**\n- `VC-002` — verdict: **pass**",
|
|
518
|
+
}));
|
|
519
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
520
|
+
const { ctx } = await start(planPath, workdir);
|
|
521
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
522
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
523
|
+
persistTaskProgress(ctx);
|
|
524
|
+
}
|
|
525
|
+
const snapshot = getExecution();
|
|
526
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
527
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
528
|
+
assert.ok(load.status === "ok");
|
|
529
|
+
assert.equal(load.checkpoint.phase, "completed", "a fully-passing round completes the run");
|
|
530
|
+
assert.equal(load.checkpoint.execution?.audit?.passed, true);
|
|
531
|
+
__setAuditRunnerForTests(null);
|
|
532
|
+
await stopExecution(makeCtx(workdir), "test teardown");
|
|
533
|
+
});
|
|
534
|
+
|
|
535
|
+
it("an all-undeterminable round completes nothing, rolls nothing back, and never wakes (v0.8)", async () => {
|
|
536
|
+
// Fail-open guard: `undeterminable` is neither a pass nor a failure, so
|
|
537
|
+
// an all-undeterminable round must leave failed === [] AND must not take
|
|
538
|
+
// the completion branch. v0.8: retries self-schedule (Q-B) — zero wakes.
|
|
539
|
+
__setAuditRunnerForTests(async ({ checklist }) => ({
|
|
540
|
+
round: 1,
|
|
541
|
+
passed: [],
|
|
542
|
+
failed: [],
|
|
543
|
+
undeterminable: checklist.map((item) => item.id),
|
|
544
|
+
report: "## Findings\n\n- `F-001` — severity: high; verdict-free reviewer output\n\n## Questions\n\nNone.",
|
|
545
|
+
}));
|
|
546
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
547
|
+
const { ctx } = await start(planPath, workdir);
|
|
548
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
549
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
550
|
+
persistTaskProgress(ctx);
|
|
551
|
+
}
|
|
552
|
+
const snapshot = getExecution();
|
|
553
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
|
|
554
|
+
const ex = getExecution()!;
|
|
555
|
+
assert.deepEqual(ex.audit.failed, []);
|
|
556
|
+
assert.deepEqual(ex.audit.undeterminable.sort(), ["VC-001", "VC-002"]);
|
|
557
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
558
|
+
assert.equal(ex.tasks.find((t) => t.id === id)?.status, "complete", `${id} not rolled back`);
|
|
559
|
+
}
|
|
560
|
+
assert.equal(ex.audit.rounds, REVIEW_MAX_ROUNDS, "self-scheduled retries spent the budget");
|
|
561
|
+
assert.equal(ex.stall.paused, true, "paused at the cap");
|
|
562
|
+
const load = loadCheckpoint(workdir, runId!);
|
|
563
|
+
assert.ok(load.status === "ok");
|
|
564
|
+
assert.equal(load.checkpoint.phase, "executing", "never fail-open into completed");
|
|
565
|
+
// Undeterminable rounds never wake the agent (Q-B); only the pause is announced.
|
|
566
|
+
const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-undeterminable");
|
|
567
|
+
assert.equal(wakes.length, 0, "undeterminable rounds send no wake");
|
|
568
|
+
assert.ok(ctx.entries.some((e) => e.customType === "pi-plans-review-paused"), "the cap pause is in-band");
|
|
474
569
|
__setAuditRunnerForTests(null);
|
|
475
570
|
await stopExecution(ctx, "test teardown");
|
|
476
571
|
});
|
|
477
572
|
|
|
573
|
+
it("the audit budget is monotonic: an agent repair re-close does not refund a round", async () => {
|
|
574
|
+
// The unbounded-loop guard. The agent's repair is a forward transition
|
|
575
|
+
// (pending -> complete); if that reset the audit budget, the sequence
|
|
576
|
+
// fail -> repair -> fail would never reach the cap.
|
|
577
|
+
let call = 0;
|
|
578
|
+
__setAuditRunnerForTests(async ({ checklist, round }) => {
|
|
579
|
+
call += 1;
|
|
580
|
+
return call === 1
|
|
581
|
+
? { round, passed: ["VC-001"], failed: ["VC-002"], report: "first attempt fails" }
|
|
582
|
+
: { round, passed: ["VC-001"], failed: ["VC-002"], report: "still failing" };
|
|
583
|
+
});
|
|
584
|
+
const { workdir, planPath } = freshWorkdir();
|
|
585
|
+
const { ctx } = await start(planPath, workdir);
|
|
586
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
587
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
588
|
+
persistTaskProgress(ctx);
|
|
589
|
+
}
|
|
590
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
591
|
+
assert.equal(getExecution()!.audit.rounds, 1);
|
|
592
|
+
// The agent repairs the rolled-back work and re-closes it: a forward
|
|
593
|
+
// transition, exactly what a naive budget reset would reward.
|
|
594
|
+
for (const id of ["Task-2", "Task-3"]) {
|
|
595
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence again`);
|
|
596
|
+
persistTaskProgress(ctx);
|
|
597
|
+
}
|
|
598
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
599
|
+
assert.equal(getExecution()!.audit.rounds, 2, "the second attempt costs a second round");
|
|
600
|
+
__setAuditRunnerForTests(null);
|
|
601
|
+
await stopExecution(makeCtx(workdir), "test teardown");
|
|
602
|
+
});
|
|
603
|
+
|
|
604
|
+
it("the audit round cap test still pauses an interactive session", async () => {
|
|
605
|
+
// Unchanged intent: exhausting the budget pauses interactively. The
|
|
606
|
+
// seam moved (getExecution()!.audit.rounds), so it is pinned here.
|
|
607
|
+
__setAuditRunnerForTests(async ({ checklist }) => ({
|
|
608
|
+
round: 1,
|
|
609
|
+
passed: [],
|
|
610
|
+
failed: checklist.filter((item) => !["VC-001"].includes(item.id)).map((item) => item.id),
|
|
611
|
+
undeterminable: [],
|
|
612
|
+
report: "still failing",
|
|
613
|
+
}));
|
|
614
|
+
const { workdir, planPath } = freshWorkdir();
|
|
615
|
+
const { ctx } = await start(planPath, workdir);
|
|
616
|
+
for (const id of ["Task-1", "Task-2", "Task-3"]) {
|
|
617
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
618
|
+
persistTaskProgress(ctx);
|
|
619
|
+
}
|
|
620
|
+
getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
|
|
621
|
+
getExecution()!.audit.failed = ["VC-002"];
|
|
622
|
+
const ctxUi = { ...makeCtx(workdir), mode: "tui" as const };
|
|
623
|
+
await restoreFromSession(ctxUi, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
624
|
+
await __awaitReviewRoundForTests();
|
|
625
|
+
const ex = getExecution()!;
|
|
626
|
+
assert.equal(ex.stall.paused, true, "the review cap pauses an interactive session");
|
|
627
|
+
assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
|
|
628
|
+
__setAuditRunnerForTests(null);
|
|
629
|
+
await stopExecution(ctxUi, "test teardown");
|
|
630
|
+
});
|
|
631
|
+
|
|
478
632
|
it("toggleDashboardExpanded flips the expanded mode", () => {
|
|
479
633
|
const before = toggleEnabled();
|
|
480
634
|
toggleDashboardExpanded(makeCtx(freshWorkdir(false).workdir));
|
|
@@ -557,10 +711,10 @@ describe("v0.7.1 execution-loop fixes (event-driven)", () => {
|
|
|
557
711
|
round += 1;
|
|
558
712
|
if (round === 1) {
|
|
559
713
|
const failed = checklist.filter((i) => i.id === "VC-002").map((i) => i.id);
|
|
560
|
-
return { round, passed: ["VC-001"], failed,
|
|
714
|
+
return { round, passed: ["VC-001"], failed, report: "VC-002 fails" };
|
|
561
715
|
}
|
|
562
716
|
for (const item of checklist) item.done = true;
|
|
563
|
-
return { round, passed: checklist.map((i) => i.id), failed: [],
|
|
717
|
+
return { round, passed: checklist.map((i) => i.id), failed: [], report: "all pass" };
|
|
564
718
|
});
|
|
565
719
|
const { workdir, planPath, runId } = freshWorkdir();
|
|
566
720
|
const ctx = await startTui(planPath, workdir);
|
|
@@ -598,7 +752,7 @@ describe("v0.7.1 execution-loop fixes (event-driven)", () => {
|
|
|
598
752
|
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
599
753
|
calls += 1;
|
|
600
754
|
for (const item of checklist) item.done = true;
|
|
601
|
-
return { round: 1, passed: checklist.map((i) => i.id), failed: [],
|
|
755
|
+
return { round: 1, passed: checklist.map((i) => i.id), failed: [], report: "all pass" };
|
|
602
756
|
});
|
|
603
757
|
const { workdir, planPath } = freshWorkdir();
|
|
604
758
|
const ctx = await startTui(planPath, workdir);
|
|
@@ -630,7 +784,7 @@ describe("v0.7.1 execution-loop fixes (event-driven)", () => {
|
|
|
630
784
|
it("root cause A: an audit failure wakes the agent exactly once", async () => {
|
|
631
785
|
__setAuditRunnerForTests(async ({ checklist }) => {
|
|
632
786
|
const failed = checklist.filter((i) => i.id === "VC-002").map((i) => i.id);
|
|
633
|
-
return { round: 1, passed: ["VC-001"], failed,
|
|
787
|
+
return { round: 1, passed: ["VC-001"], failed, report: "VC-002 fails" };
|
|
634
788
|
});
|
|
635
789
|
const { workdir, planPath } = freshWorkdir();
|
|
636
790
|
const ctx = await startTui(planPath, workdir);
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
* WITHOUT typechecking — duplicate type/interface declarations pass silently.
|
|
5
5
|
* pi's extension loader parses TypeScript for real and refused the whole
|
|
6
6
|
* extension ("Identifier 'RoleConfig' has already been declared"), which made
|
|
7
|
-
* every spawned subagent (including the
|
|
7
|
+
* every spawned subagent (including the execution reviewer) fail while the
|
|
8
8
|
* whole suite stayed green.
|
|
9
9
|
*
|
|
10
10
|
* Guard: spawn `pi` from the repo root with a deliberately missing model.
|
|
@@ -259,7 +259,11 @@ describe("/resume-plans on the real Pi host", () => {
|
|
|
259
259
|
assert.equal(final.status, "ok");
|
|
260
260
|
if (final.status === "ok") {
|
|
261
261
|
assert.equal(final.checkpoint.execution?.tasks?.["Task-1"]?.status, "complete");
|
|
262
|
-
|
|
262
|
+
// Every task now carries a record: an untouched task is
|
|
263
|
+
// persisted as pending rather than omitted, so the checkpoint
|
|
264
|
+
// distinguishes "never started" from "rolled back with
|
|
265
|
+
// evidence" instead of losing the record entirely.
|
|
266
|
+
assert.equal(final.checkpoint.execution?.tasks?.["Task-2"]?.status, "pending");
|
|
263
267
|
}
|
|
264
268
|
} finally {
|
|
265
269
|
session2.session.dispose();
|
package/tests/resume.test.ts
CHANGED
|
@@ -112,6 +112,9 @@ describe("candidate discovery", () => {
|
|
|
112
112
|
it("lists resumable runs, excludes terminal ones, registry hint first", () => {
|
|
113
113
|
const workdir = setupRepo("discovery");
|
|
114
114
|
const planning = startRun(workdir, { topic: "alpha", skill: "plan-normal", requestText: "a" }).run;
|
|
115
|
+
const verifying = startRun(workdir, { topic: "zeta", skill: "plan-normal", requestText: "f" }).run;
|
|
116
|
+
setRunStatus(workdir, verifying.run_id, "executing");
|
|
117
|
+
setRunStatus(workdir, verifying.run_id, "verifying");
|
|
115
118
|
const stopped = startRun(workdir, { topic: "beta", skill: "plan-normal", requestText: "b" }).run;
|
|
116
119
|
setRunStatus(workdir, stopped.run_id, "executing");
|
|
117
120
|
setRunStatus(workdir, stopped.run_id, "stopped");
|
|
@@ -129,6 +132,9 @@ describe("candidate discovery", () => {
|
|
|
129
132
|
assert.ok(ids.includes(planning.run_id));
|
|
130
133
|
assert.ok(ids.includes(stopped.run_id));
|
|
131
134
|
assert.ok(ids.includes(doneWithReview.run_id), "done with review artifacts resumable");
|
|
135
|
+
assert.ok(ids.includes(verifying.run_id), "verifying runs are resumable");
|
|
136
|
+
const verifyingCandidate = candidates.find((candidate) => candidate.runId === verifying.run_id);
|
|
137
|
+
assert.equal(verifyingCandidate?.phaseLabel, "verifying", "verifying runs keep a distinct resume label even though checkpoint.phase stays executing");
|
|
132
138
|
assert.equal(ids.includes(abandoned.run_id), false);
|
|
133
139
|
assert.equal(ids.includes(done.run_id), false, "plain done excluded");
|
|
134
140
|
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/** Tests for the stale-extension probe shared by /plans and the completion
|
|
2
|
+
* auditor. */
|
|
3
|
+
|
|
4
|
+
import * as assert from "node:assert/strict";
|
|
5
|
+
import * as fs from "node:fs";
|
|
6
|
+
import * as os from "node:os";
|
|
7
|
+
import * as path from "node:path";
|
|
8
|
+
import { after, before, describe, it } from "node:test";
|
|
9
|
+
import { newestSourceMtime, staleReloadHint, stalenessLine } from "../src/staleness.ts";
|
|
10
|
+
|
|
11
|
+
let root: string;
|
|
12
|
+
|
|
13
|
+
before(() => {
|
|
14
|
+
root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-stale-"));
|
|
15
|
+
fs.mkdirSync(path.join(root, "src"));
|
|
16
|
+
fs.mkdirSync(path.join(root, "tools"));
|
|
17
|
+
fs.writeFileSync(path.join(root, "index.ts"), "export const x = 1;\n");
|
|
18
|
+
fs.writeFileSync(path.join(root, "src", "a.ts"), "export const a = 1;\n");
|
|
19
|
+
});
|
|
20
|
+
|
|
21
|
+
after(() => {
|
|
22
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
/** Set every scanned file's mtime to `when`. */
|
|
26
|
+
function touchAll(when: number): void {
|
|
27
|
+
for (const rel of ["index.ts", "src/a.ts", "tools"]) {
|
|
28
|
+
const full = path.join(root, rel);
|
|
29
|
+
if (fs.statSync(full).isDirectory()) {
|
|
30
|
+
fs.writeFileSync(path.join(full, "b.ts"), "export const b = 1;\n");
|
|
31
|
+
fs.utimesSync(path.join(full, "b.ts"), when / 1000, when / 1000);
|
|
32
|
+
} else {
|
|
33
|
+
fs.utimesSync(full, when / 1000, when / 1000);
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
describe("staleness probe", () => {
|
|
39
|
+
it("finds the newest .ts mtime across the root, src and tools", () => {
|
|
40
|
+
const now = Date.now();
|
|
41
|
+
touchAll(now - 60_000);
|
|
42
|
+
const newest = newestSourceMtime(root);
|
|
43
|
+
assert.ok(newest >= now - 61_000 && newest <= now + 1_000, `unexpected mtime ${newest}`);
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
it("reports the copy as current when disk is older than the load time", () => {
|
|
47
|
+
touchAll(Date.now() - 60_000);
|
|
48
|
+
const line = stalenessLine(root, new Date());
|
|
49
|
+
assert.match(line, /up to date/);
|
|
50
|
+
assert.equal(staleReloadHint(root, new Date()), null, "no /reload advice when current");
|
|
51
|
+
});
|
|
52
|
+
|
|
53
|
+
it("reports the copy as stale and advises /reload when disk is newer", () => {
|
|
54
|
+
const future = new Date(Date.now() + 10 * 60_000);
|
|
55
|
+
touchAll(future.getTime());
|
|
56
|
+
const line = stalenessLine(root, new Date(Date.now() - 60_000));
|
|
57
|
+
assert.match(line, /newer than the loaded copy/);
|
|
58
|
+
assert.match(line, /run \/reload/);
|
|
59
|
+
const hint = staleReloadHint(root, new Date(Date.now() - 60_000));
|
|
60
|
+
assert.ok(hint && hint.includes("/reload"), "the hint carries the /reload advice");
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
it("tolerates a same-second write instead of calling it stale", () => {
|
|
64
|
+
// Without the 2s tolerance, a file written moments after load reads as
|
|
65
|
+
// staleness and users chase a /reload that changes nothing.
|
|
66
|
+
touchAll(Date.now());
|
|
67
|
+
const line = stalenessLine(root, new Date());
|
|
68
|
+
assert.match(line, /up to date/);
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
it("degrades to a loaded-timestamp line when the tree cannot be walked", () => {
|
|
72
|
+
const line = stalenessLine(path.join(root, "does-not-exist"), new Date("2026-01-01T00:00:00.000Z"));
|
|
73
|
+
assert.match(line, /Extension loaded: 2026-01-01T00:00:00.000Z/);
|
|
74
|
+
assert.equal(staleReloadHint(path.join(root, "does-not-exist"), new Date()), null);
|
|
75
|
+
});
|
|
76
|
+
});
|
package/tests/state.test.ts
CHANGED
|
@@ -263,6 +263,10 @@ describe("runs", () => {
|
|
|
263
263
|
const updated = setRunStatus(workdir, run.run_id, "accepted");
|
|
264
264
|
assert.equal(updated.status, "accepted");
|
|
265
265
|
assert.equal(getRun(workdir, run.run_id)?.status, "accepted");
|
|
266
|
+
// v0.8: verifying is a valid non-terminal lifecycle status (execution-review loop).
|
|
267
|
+
const verifying = setRunStatus(workdir, run.run_id, "verifying");
|
|
268
|
+
assert.equal(verifying.status, "verifying");
|
|
269
|
+
assert.equal(getRun(workdir, run.run_id)?.status, "verifying");
|
|
266
270
|
assert.throws(() => setRunStatus(workdir, run.run_id, "bogus"), StateError);
|
|
267
271
|
});
|
|
268
272
|
|