pi-plans 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/agents/execution-reviewer.md +62 -10
- package/package.json +2 -2
- package/references/pi-planning-workflow.md +16 -8
- package/references/state-and-config.md +2 -2
- package/src/auditor.ts +170 -21
- package/src/dashboard.ts +102 -19
- package/src/exec.ts +865 -107
- package/src/refine-ui.ts +21 -5
- package/src/resume-command.ts +26 -6
- package/src/review-budget.ts +290 -0
- package/src/tasks.ts +76 -0
- package/src/ui-language.ts +38 -0
- package/src/workflow-state.ts +217 -8
- package/tests/analyze-refs.test.ts +1 -1
- package/tests/auditor.test.ts +185 -1
- package/tests/dashboard.test.ts +165 -1
- package/tests/exec-review-loop.test.ts +1031 -10
- package/tests/exec.test.ts +32 -5
- package/tests/fixtures/lattice-code-blocked/plan-v2-trimmed.md +57 -0
- package/tests/fixtures/lattice-code-blocked/state.json +158 -0
- package/tests/refine-ui.test.ts +25 -2
- package/tests/resume.test.ts +45 -1
- package/tests/review-budget.test.ts +201 -0
- package/tests/tasks.test.ts +45 -0
- package/tests/workflow-state.test.ts +227 -0
- package/tools/analyze-refs.ts +17 -6
- package/tools/execute-plan.ts +12 -5
- package/tools/refine.ts +22 -3
package/src/workflow-state.ts
CHANGED
|
@@ -29,6 +29,7 @@ import { existsSync, mkdirSync, readFileSync, renameSync, writeFileSync } from "
|
|
|
29
29
|
import * as path from "node:path";
|
|
30
30
|
import { StateError, atomicWriteJson, resolveStateRootOrNull, runGit, runDirPath, utcNow } from "./state.ts";
|
|
31
31
|
import { assertOwnership, heldOwnershipRecord } from "./run-ownership.ts";
|
|
32
|
+
import type { NoProgressState, ReviewBudget } from "./review-budget.ts";
|
|
32
33
|
|
|
33
34
|
// ---------------------------------------------------------------------------
|
|
34
35
|
// Schema
|
|
@@ -133,6 +134,21 @@ export interface ExecutionApproval {
|
|
|
133
134
|
approvedAt: string;
|
|
134
135
|
}
|
|
135
136
|
|
|
137
|
+
/** v0.9.2: the newest failed execution-review round's rollback set plus the
|
|
138
|
+
* tasks among it that are still open. Captured when the round commits (the
|
|
139
|
+
* reopen helpers mutate the tree and report only flipped nodes, so a later
|
|
140
|
+
* re-derivation cannot rebuild this set) and persisted so wakes, the pause
|
|
141
|
+
* reason, the dashboard, and resume briefs can name the blocker after a
|
|
142
|
+
* restart. `escalatedRounds` counts consecutive blocked wakes — the escalation
|
|
143
|
+
* ladder must not ride `stallRounds`, which any successful tool call resets. */
|
|
144
|
+
export interface ExecutionBlocked {
|
|
145
|
+
rolledBack: string[];
|
|
146
|
+
tasks: string[];
|
|
147
|
+
round: number;
|
|
148
|
+
escalatedRounds: number;
|
|
149
|
+
since: string;
|
|
150
|
+
}
|
|
151
|
+
|
|
136
152
|
export interface ExecutionCheckpoint {
|
|
137
153
|
approval: ExecutionApproval | null;
|
|
138
154
|
doneVcIds: string[];
|
|
@@ -156,8 +172,59 @@ export interface ExecutionCheckpoint {
|
|
|
156
172
|
* silently hand a stalled run a fresh budget; optional so checkpoints
|
|
157
173
|
* written before this field keep loading. */
|
|
158
174
|
stallRounds?: number;
|
|
175
|
+
/** v0.9.2: outstanding blocker from the newest failed review round (see
|
|
176
|
+
* `ExecutionBlocked`); absent = nothing blocks the review. Optional on
|
|
177
|
+
* read so checkpoints written before it keep loading. */
|
|
178
|
+
blocked?: ExecutionBlocked | null;
|
|
179
|
+
/** v0.9.3: the per-run execution-review budget (committed rounds, or
|
|
180
|
+
* `"unlimited"`). Absent = not decided yet (the picker runs before round
|
|
181
|
+
* 1) OR written by a pre-v0.9.3 build, in which case a checkpoint that
|
|
182
|
+
* already spent rounds keeps the legacy 5-round bound
|
|
183
|
+
* (`resolveStoredBudget` in ./review-budget.ts). */
|
|
184
|
+
reviewBudget?: ReviewBudget;
|
|
185
|
+
/** v0.9.3: the budget above was the no-UI fallback, not a user pick — the
|
|
186
|
+
* dashboard/resume brief annotate it `(default)`. Optional on read. */
|
|
187
|
+
reviewBudgetDefaulted?: boolean;
|
|
188
|
+
/** v0.9.3: committed review rounds across the WHOLE run (never reset by a
|
|
189
|
+
* grant) — bounds the unlimited budget together with `reviewCapExtension`. */
|
|
190
|
+
reviewRoundsTotal?: number;
|
|
191
|
+
/** v0.9.3: explicit `/plans-execute` grants lifted the unlimited hard cap
|
|
192
|
+
* by this many rounds (each grant that lands on `unlimited` adds
|
|
193
|
+
* `UNLIMITED_HARD_CAP`). */
|
|
194
|
+
reviewCapExtension?: number;
|
|
195
|
+
/** v0.9.3: no-progress valve state for the unlimited budget (signature of
|
|
196
|
+
* the last committed round's outcome plus its consecutive streak). */
|
|
197
|
+
reviewNoProgress?: NoProgressState | null;
|
|
198
|
+
/** v0.9.1 (F-002): set when the execution-review loop mechanically
|
|
199
|
+
* appended finding tasks to the approved plan. The checkpoint's plan
|
|
200
|
+
* identity is re-stamped at that moment so /resume-plans accepts the
|
|
201
|
+
* amended plan; the approval record keeps the ORIGINAL digest as
|
|
202
|
+
* evidence of what the user actually approved. */
|
|
203
|
+
planAmended?: { sha256: string; amendedAt: string; round: number };
|
|
159
204
|
/** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
|
|
160
|
-
audit?: {
|
|
205
|
+
audit?: {
|
|
206
|
+
rounds: number;
|
|
207
|
+
lastResult?: string;
|
|
208
|
+
passed?: boolean;
|
|
209
|
+
undeterminable?: string[];
|
|
210
|
+
/** v0.9: unresolved implementation findings from the newest committed
|
|
211
|
+
* review round (stable F-### ids). Optional so pre-v0.9 checkpoints
|
|
212
|
+
* keep loading as "no findings". */
|
|
213
|
+
findings?: ReviewFindingRecord[];
|
|
214
|
+
};
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
/** Serializable shape of one review finding (v0.9). Structurally identical to
|
|
218
|
+
* auditor.ts's ReviewFinding; declared here so the checkpoint layer does not
|
|
219
|
+
* import the auditor. */
|
|
220
|
+
export interface ReviewFindingRecord {
|
|
221
|
+
id: string;
|
|
222
|
+
severity: string;
|
|
223
|
+
taskIds: string[];
|
|
224
|
+
proposedTask?: string;
|
|
225
|
+
note: string;
|
|
226
|
+
evidence: string;
|
|
227
|
+
raw: string;
|
|
161
228
|
}
|
|
162
229
|
|
|
163
230
|
export interface OwnerInfo {
|
|
@@ -269,6 +336,13 @@ function asInt(value: unknown, label: string, min: number): number {
|
|
|
269
336
|
return value;
|
|
270
337
|
}
|
|
271
338
|
|
|
339
|
+
/** v0.9.3: a positive integer round count or the literal "unlimited". */
|
|
340
|
+
function asReviewBudget(value: unknown, label: string): ReviewBudget {
|
|
341
|
+
if (value === "unlimited") return "unlimited";
|
|
342
|
+
if (typeof value === "number" && Number.isSafeInteger(value) && value >= 1) return value;
|
|
343
|
+
throw new CheckpointValidationError(`${label}: expected a positive integer or "unlimited"`);
|
|
344
|
+
}
|
|
345
|
+
|
|
272
346
|
function asEnum<T extends string>(value: unknown, allowed: Set<T>, label: string): T {
|
|
273
347
|
const text = asString(value, label);
|
|
274
348
|
if (!allowed.has(text as T)) {
|
|
@@ -465,7 +539,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
465
539
|
const record = asRecord(value, label);
|
|
466
540
|
rejectExtraKeys(
|
|
467
541
|
record,
|
|
468
|
-
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "audit"]),
|
|
542
|
+
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "blocked", "planAmended", "audit", "reviewBudget", "reviewBudgetDefaulted", "reviewRoundsTotal", "reviewCapExtension", "reviewNoProgress"]),
|
|
469
543
|
label,
|
|
470
544
|
);
|
|
471
545
|
const execution: ExecutionCheckpoint = {
|
|
@@ -485,6 +559,17 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
485
559
|
};
|
|
486
560
|
if (record.currentI !== undefined) execution.currentI = asString(record.currentI, `${label}.currentI`);
|
|
487
561
|
if (record.pausedReason !== undefined) execution.pausedReason = asString(record.pausedReason, `${label}.pausedReason`);
|
|
562
|
+
if (record.blocked !== undefined && record.blocked !== null) {
|
|
563
|
+
const blocked = asRecord(record.blocked, `${label}.blocked`);
|
|
564
|
+
rejectExtraKeys(blocked, new Set(["rolledBack", "tasks", "round", "escalatedRounds", "since"]), `${label}.blocked`);
|
|
565
|
+
execution.blocked = {
|
|
566
|
+
rolledBack: asStringArray(blocked.rolledBack, `${label}.blocked.rolledBack`),
|
|
567
|
+
tasks: asStringArray(blocked.tasks, `${label}.blocked.tasks`),
|
|
568
|
+
round: asInt(blocked.round, `${label}.blocked.round`, 0),
|
|
569
|
+
escalatedRounds: asInt(blocked.escalatedRounds, `${label}.blocked.escalatedRounds`, 0),
|
|
570
|
+
since: asTimestamp(blocked.since, `${label}.blocked.since`),
|
|
571
|
+
};
|
|
572
|
+
}
|
|
488
573
|
if (record.reverifyAll !== undefined) execution.reverifyAll = asBool(record.reverifyAll, `${label}.reverifyAll`);
|
|
489
574
|
if (record.originWorktree !== undefined) execution.originWorktree = asString(record.originWorktree, `${label}.originWorktree`);
|
|
490
575
|
if (record.delegate !== undefined && record.delegate !== null) {
|
|
@@ -512,10 +597,43 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
512
597
|
if (record.stallRounds !== undefined && record.stallRounds !== null) {
|
|
513
598
|
execution.stallRounds = asInt(record.stallRounds, `${label}.stallRounds`, 0);
|
|
514
599
|
}
|
|
600
|
+
if (record.planAmended !== undefined && record.planAmended !== null) {
|
|
601
|
+
const rec = asRecord(record.planAmended, `${label}.planAmended`);
|
|
602
|
+
rejectExtraKeys(rec, new Set(["sha256", "amendedAt", "round"]), `${label}.planAmended`);
|
|
603
|
+
execution.planAmended = {
|
|
604
|
+
sha256: asString(rec.sha256, `${label}.planAmended.sha256`),
|
|
605
|
+
amendedAt: asString(rec.amendedAt, `${label}.planAmended.amendedAt`),
|
|
606
|
+
round: asInt(rec.round, `${label}.planAmended.round`, 0),
|
|
607
|
+
};
|
|
608
|
+
}
|
|
609
|
+
// v0.9.3 review budget. `reviewBudget` accepts a positive integer or the
|
|
610
|
+
// literal "unlimited"; the counters are non-negative integers; the
|
|
611
|
+
// no-progress record is a small fixed shape. All optional on read so
|
|
612
|
+
// pre-feature checkpoints keep loading.
|
|
613
|
+
if (record.reviewBudget !== undefined && record.reviewBudget !== null) {
|
|
614
|
+
execution.reviewBudget = asReviewBudget(record.reviewBudget, `${label}.reviewBudget`);
|
|
615
|
+
}
|
|
616
|
+
if (record.reviewBudgetDefaulted !== undefined) {
|
|
617
|
+
execution.reviewBudgetDefaulted = asBool(record.reviewBudgetDefaulted, `${label}.reviewBudgetDefaulted`);
|
|
618
|
+
}
|
|
619
|
+
if (record.reviewRoundsTotal !== undefined && record.reviewRoundsTotal !== null) {
|
|
620
|
+
execution.reviewRoundsTotal = asInt(record.reviewRoundsTotal, `${label}.reviewRoundsTotal`, 0);
|
|
621
|
+
}
|
|
622
|
+
if (record.reviewCapExtension !== undefined && record.reviewCapExtension !== null) {
|
|
623
|
+
execution.reviewCapExtension = asInt(record.reviewCapExtension, `${label}.reviewCapExtension`, 0);
|
|
624
|
+
}
|
|
625
|
+
if (record.reviewNoProgress !== undefined && record.reviewNoProgress !== null) {
|
|
626
|
+
const np = asRecord(record.reviewNoProgress, `${label}.reviewNoProgress`);
|
|
627
|
+
rejectExtraKeys(np, new Set(["key", "streak"]), `${label}.reviewNoProgress`);
|
|
628
|
+
execution.reviewNoProgress = {
|
|
629
|
+
key: asString(np.key, `${label}.reviewNoProgress.key`),
|
|
630
|
+
streak: asInt(np.streak, `${label}.reviewNoProgress.streak`, 0),
|
|
631
|
+
};
|
|
632
|
+
}
|
|
515
633
|
if (record.audit !== undefined && record.audit !== null) {
|
|
516
634
|
const audit = asRecord(record.audit, `${label}.audit`);
|
|
517
|
-
rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable"]), `${label}.audit`);
|
|
518
|
-
const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] } = {
|
|
635
|
+
rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable", "findings"]), `${label}.audit`);
|
|
636
|
+
const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[]; findings?: ReviewFindingRecord[] } = {
|
|
519
637
|
rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
|
|
520
638
|
};
|
|
521
639
|
if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
|
|
@@ -523,6 +641,24 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
523
641
|
if (audit.undeterminable !== undefined) {
|
|
524
642
|
parsed.undeterminable = asStringArray(audit.undeterminable, `${label}.audit.undeterminable`);
|
|
525
643
|
}
|
|
644
|
+
if (audit.findings !== undefined && audit.findings !== null) {
|
|
645
|
+
const arr = Array.isArray(audit.findings) ? audit.findings : null;
|
|
646
|
+
if (!arr) throw new CheckpointValidationError(`${label}.audit.findings must be an array`);
|
|
647
|
+
parsed.findings = arr.map((entry, i) => {
|
|
648
|
+
const rec = asRecord(entry, `${label}.audit.findings.${i}`);
|
|
649
|
+
rejectExtraKeys(rec, new Set(["id", "severity", "taskIds", "proposedTask", "note", "evidence", "raw"]), `${label}.audit.findings.${i}`);
|
|
650
|
+
const out: ReviewFindingRecord = {
|
|
651
|
+
id: asString(rec.id, `${label}.audit.findings.${i}.id`),
|
|
652
|
+
severity: asString(rec.severity, `${label}.audit.findings.${i}.severity`),
|
|
653
|
+
taskIds: rec.taskIds === undefined ? [] : asStringArray(rec.taskIds, `${label}.audit.findings.${i}.taskIds`),
|
|
654
|
+
note: rec.note === undefined ? "" : asString(rec.note, `${label}.audit.findings.${i}.note`),
|
|
655
|
+
evidence: rec.evidence === undefined ? "" : asString(rec.evidence, `${label}.audit.findings.${i}.evidence`),
|
|
656
|
+
raw: rec.raw === undefined ? "" : asString(rec.raw, `${label}.audit.findings.${i}.raw`),
|
|
657
|
+
};
|
|
658
|
+
if (rec.proposedTask !== undefined) out.proposedTask = asString(rec.proposedTask, `${label}.audit.findings.${i}.proposedTask`);
|
|
659
|
+
return out;
|
|
660
|
+
});
|
|
661
|
+
}
|
|
526
662
|
execution.audit = parsed;
|
|
527
663
|
}
|
|
528
664
|
return execution;
|
|
@@ -1080,10 +1216,26 @@ export interface ExecutionProgressInput {
|
|
|
1080
1216
|
delegate?: { modelSelector: string; startedAt: string } | null;
|
|
1081
1217
|
/** v0.6.1: task-tree progress snapshot (authoritative). */
|
|
1082
1218
|
tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
|
|
1083
|
-
/** v0.6.1: completion-audit bookkeeping update.
|
|
1084
|
-
|
|
1219
|
+
/** v0.6.1: completion-audit bookkeeping update. v0.9: findings rides the
|
|
1220
|
+
* same replace-semantics slot — callers that must preserve findings (e.g.
|
|
1221
|
+
* budget renewal) pass them through explicitly. */
|
|
1222
|
+
audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[]; findings?: ReviewFindingRecord[] };
|
|
1085
1223
|
/** v0.7.1: watchdog budget counter, so a restart cannot refresh it. */
|
|
1086
1224
|
stallRounds?: number;
|
|
1225
|
+
/** v0.9.2: blocker record; null clears it (delete-on-null, like
|
|
1226
|
+
* `pausedReason`/`delegate`). */
|
|
1227
|
+
blocked?: ExecutionBlocked | null;
|
|
1228
|
+
/** v0.9.3: per-run execution-review budget; null clears it (delete-on-null,
|
|
1229
|
+
* used by the stop/complete cleanup paths). */
|
|
1230
|
+
reviewBudget?: ReviewBudget | null;
|
|
1231
|
+
/** v0.9.3: whether `reviewBudget` came from the no-UI fallback. */
|
|
1232
|
+
reviewBudgetDefaulted?: boolean | null;
|
|
1233
|
+
/** v0.9.3: committed rounds across the whole run (monotonic). */
|
|
1234
|
+
reviewRoundsTotal?: number;
|
|
1235
|
+
/** v0.9.3: explicit unlimited-cap extension accumulated by grants. */
|
|
1236
|
+
reviewCapExtension?: number;
|
|
1237
|
+
/** v0.9.3: no-progress valve state; null clears it (delete-on-null). */
|
|
1238
|
+
reviewNoProgress?: NoProgressState | null;
|
|
1087
1239
|
}
|
|
1088
1240
|
|
|
1089
1241
|
export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
|
|
@@ -1104,10 +1256,37 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
|
|
|
1104
1256
|
else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
|
|
1105
1257
|
if (progress.tasks !== undefined) execution.tasks = progress.tasks;
|
|
1106
1258
|
if (progress.stallRounds !== undefined) execution.stallRounds = progress.stallRounds;
|
|
1259
|
+
if (progress.blocked === null) delete execution.blocked;
|
|
1260
|
+
else if (progress.blocked !== undefined) execution.blocked = progress.blocked;
|
|
1261
|
+
if (progress.reviewBudget === null) delete execution.reviewBudget;
|
|
1262
|
+
else if (progress.reviewBudget !== undefined) execution.reviewBudget = progress.reviewBudget;
|
|
1263
|
+
if (progress.reviewBudgetDefaulted === null) delete execution.reviewBudgetDefaulted;
|
|
1264
|
+
else if (progress.reviewBudgetDefaulted !== undefined) execution.reviewBudgetDefaulted = progress.reviewBudgetDefaulted;
|
|
1265
|
+
if (progress.reviewRoundsTotal !== undefined) execution.reviewRoundsTotal = progress.reviewRoundsTotal;
|
|
1266
|
+
if (progress.reviewCapExtension !== undefined) execution.reviewCapExtension = progress.reviewCapExtension;
|
|
1267
|
+
if (progress.reviewNoProgress === null) delete execution.reviewNoProgress;
|
|
1268
|
+
else if (progress.reviewNoProgress !== undefined) execution.reviewNoProgress = progress.reviewNoProgress;
|
|
1107
1269
|
if (progress.audit !== undefined) execution.audit = progress.audit;
|
|
1108
1270
|
return { ...cp, execution };
|
|
1109
1271
|
}
|
|
1110
1272
|
|
|
1273
|
+
/** v0.9.1 (F-002): the execution-review loop appended finding tasks to the
|
|
1274
|
+
* approved plan file; re-stamp the checkpoint's plan identity to the amended
|
|
1275
|
+
* digest so a later /resume-plans does not reject the run as plan-mismatch.
|
|
1276
|
+
* The approval record is untouched — it keeps the digest the user actually
|
|
1277
|
+
* approved, and planAmended records when and why the identity moved. */
|
|
1278
|
+
export function applyExecutionPlanAmended(cp: WorkflowCheckpoint, plan: PlanIdentity, round: number): WorkflowCheckpoint {
|
|
1279
|
+
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1280
|
+
return {
|
|
1281
|
+
...cp,
|
|
1282
|
+
plan,
|
|
1283
|
+
execution: {
|
|
1284
|
+
...cp.execution,
|
|
1285
|
+
planAmended: { sha256: plan.sha256, amendedAt: utcNow(), round },
|
|
1286
|
+
},
|
|
1287
|
+
};
|
|
1288
|
+
}
|
|
1289
|
+
|
|
1111
1290
|
/** D-011/F-001: code state changed under an unchanged plan — keep authorization, re-verify first. */
|
|
1112
1291
|
export function applyExecutionHeadChanged(cp: WorkflowCheckpoint): WorkflowCheckpoint {
|
|
1113
1292
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
@@ -1118,14 +1297,25 @@ export function applyExecutionCompleted(cp: WorkflowCheckpoint): WorkflowCheckpo
|
|
|
1118
1297
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1119
1298
|
// v0.6.1 (D-018): the post-execution amelioration loop is gone; a
|
|
1120
1299
|
// completed audit passes the run straight to the terminal phase.
|
|
1300
|
+
// v0.9.3: the review budget (and its valve state) describes a review that
|
|
1301
|
+
// no longer runs — completed runs do not carry it.
|
|
1302
|
+
const {
|
|
1303
|
+
reviewBudget: _reviewBudget,
|
|
1304
|
+
reviewBudgetDefaulted: _reviewBudgetDefaulted,
|
|
1305
|
+
reviewRoundsTotal: _reviewRoundsTotal,
|
|
1306
|
+
reviewCapExtension: _reviewCapExtension,
|
|
1307
|
+
reviewNoProgress: _reviewNoProgress,
|
|
1308
|
+
...execution
|
|
1309
|
+
} = cp.execution;
|
|
1121
1310
|
return {
|
|
1122
1311
|
...cp,
|
|
1123
1312
|
phase: "completed",
|
|
1124
1313
|
nextAction: "none",
|
|
1125
1314
|
execution: {
|
|
1126
|
-
...
|
|
1315
|
+
...execution,
|
|
1127
1316
|
pausedReason: undefined,
|
|
1128
1317
|
delegate: undefined,
|
|
1318
|
+
blocked: undefined,
|
|
1129
1319
|
audit: { ...(cp.execution.audit ?? { rounds: 0 }), passed: true },
|
|
1130
1320
|
},
|
|
1131
1321
|
};
|
|
@@ -1156,6 +1346,14 @@ export function applyMigration(
|
|
|
1156
1346
|
// migration (same rule as VC validity).
|
|
1157
1347
|
tasks: {},
|
|
1158
1348
|
audit: { rounds: 0 },
|
|
1349
|
+
// v0.9.3 (F-005): the review budget and its counters ARE carried
|
|
1350
|
+
// across a migration — they describe the user's choice and the
|
|
1351
|
+
// run's cumulative spend, not the (invalidated) code state.
|
|
1352
|
+
reviewBudget: cp.execution.reviewBudget,
|
|
1353
|
+
reviewBudgetDefaulted: cp.execution.reviewBudgetDefaulted,
|
|
1354
|
+
reviewRoundsTotal: cp.execution.reviewRoundsTotal,
|
|
1355
|
+
reviewCapExtension: cp.execution.reviewCapExtension,
|
|
1356
|
+
reviewNoProgress: cp.execution.reviewNoProgress ?? undefined,
|
|
1159
1357
|
}
|
|
1160
1358
|
: undefined,
|
|
1161
1359
|
implementationReview: cp.implementationReview
|
|
@@ -1179,7 +1377,18 @@ function migrationNextAction(cp: WorkflowCheckpoint): NextAction {
|
|
|
1179
1377
|
/** Mark a paused stop without erasing the last phase (D-008). */
|
|
1180
1378
|
export function applyExecutionStopped(cp: WorkflowCheckpoint, reason: string): WorkflowCheckpoint {
|
|
1181
1379
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1182
|
-
|
|
1380
|
+
// v0.9.3: a stop also drops the review budget and its valve state — the
|
|
1381
|
+
// next execution of this plan picks its own budget.
|
|
1382
|
+
const {
|
|
1383
|
+
delegate: _delegate,
|
|
1384
|
+
blocked: _blocked,
|
|
1385
|
+
reviewBudget: _reviewBudget,
|
|
1386
|
+
reviewBudgetDefaulted: _reviewBudgetDefaulted,
|
|
1387
|
+
reviewRoundsTotal: _reviewRoundsTotal,
|
|
1388
|
+
reviewCapExtension: _reviewCapExtension,
|
|
1389
|
+
reviewNoProgress: _reviewNoProgress,
|
|
1390
|
+
...execution
|
|
1391
|
+
} = cp.execution;
|
|
1183
1392
|
// A stop also revokes any outstanding audit-rollback authorization.
|
|
1184
1393
|
const audit = execution.audit ? { ...execution.audit } : undefined;
|
|
1185
1394
|
return { ...cp, execution: { ...execution, audit, pausedReason: reason } };
|
|
@@ -303,7 +303,7 @@ describe("analyze_refs fanout", () => {
|
|
|
303
303
|
|
|
304
304
|
it("pins the per-batch overlay lifecycle (open before spawn, close in finally, cap 3)", () => {
|
|
305
305
|
const source = fs.readFileSync(path.join(ROOT, "tools", "analyze-refs.ts"), "utf8");
|
|
306
|
-
assert.equal((source.match(/new RefineOverlayController\("refs"/g) ?? []).length, 1, "controller must be constructed per batch inside the loop");
|
|
306
|
+
assert.equal((source.match(/new RefineOverlayController\(\s*"refs"/g) ?? []).length, 1, "controller must be constructed per batch inside the loop");
|
|
307
307
|
assert.match(source, /overlay\?\.open\(refineOverlayContext\(ctx\), modelLabel\)/);
|
|
308
308
|
assert.match(source, /await overlay\?\.close\(\);/);
|
|
309
309
|
assert.match(source, /const BATCH_SIZE = 3;/);
|
package/tests/auditor.test.ts
CHANGED
|
@@ -2,6 +2,9 @@
|
|
|
2
2
|
* coverage, rollback boundaries, skipped-pass, and no-cover exclusion. */
|
|
3
3
|
|
|
4
4
|
import * as assert from "node:assert/strict";
|
|
5
|
+
import * as fs from "node:fs";
|
|
6
|
+
import * as os from "node:os";
|
|
7
|
+
import * as path from "node:path";
|
|
5
8
|
import { describe, it } from "node:test";
|
|
6
9
|
import type { CheckItem } from "../src/plan.ts";
|
|
7
10
|
import { auditRollbackSet, buildTaskView } from "../src/tasks.ts";
|
|
@@ -207,4 +210,185 @@ describe("audit rollback boundaries", () => {
|
|
|
207
210
|
assert.ok(presolved.includes("VC-003"));
|
|
208
211
|
assert.ok(!presolved.includes("VC-002"), "mixed coverage needs the auditor");
|
|
209
212
|
});
|
|
210
|
-
});
|
|
213
|
+
});
|
|
214
|
+
describe("findings parsing (v0.9)", () => {
|
|
215
|
+
const GOOD = [
|
|
216
|
+
"- `VC-001` — verdict: pass; evidence: ok",
|
|
217
|
+
"",
|
|
218
|
+
"- `F-001` — severity: high; tasks: Task-3, task-4; note: union rollback missing; evidence: src/exec.ts:1290",
|
|
219
|
+
"- `F-002` — severity: medium; tasks: none; proposed-task: cap retry backoff at 60s; note: unbounded; evidence: src/client.ts:12",
|
|
220
|
+
].join("\n");
|
|
221
|
+
|
|
222
|
+
it("parses well-formed findings with id/task normalization", () => {
|
|
223
|
+
const { findings } = parseAuditReport(GOOD, ALL_IDS);
|
|
224
|
+
assert.equal(findings.length, 2);
|
|
225
|
+
const f1 = findings[0];
|
|
226
|
+
assert.equal(f1.id, "F-001");
|
|
227
|
+
assert.equal(f1.severity, "high");
|
|
228
|
+
assert.deepEqual(f1.taskIds, ["Task-3", "Task-4"]);
|
|
229
|
+
assert.equal(f1.note, "union rollback missing");
|
|
230
|
+
assert.equal(f1.evidence, "src/exec.ts:1290");
|
|
231
|
+
const f2 = findings[1];
|
|
232
|
+
assert.equal(f2.severity, "medium");
|
|
233
|
+
assert.deepEqual(f2.taskIds, []);
|
|
234
|
+
assert.equal(f2.proposedTask, "cap retry backoff at 60s");
|
|
235
|
+
});
|
|
236
|
+
|
|
237
|
+
it("tolerates emphasis markers on severity", () => {
|
|
238
|
+
const { findings } = parseAuditReport("- `F-001` — severity: **high**; tasks: Task-1; note: n; evidence: e", ALL_IDS);
|
|
239
|
+
assert.equal(findings[0]?.severity, "high");
|
|
240
|
+
const { findings: f2 } = parseAuditReport("- `F-002` — severity: `medium`; tasks: none; note: n", ALL_IDS);
|
|
241
|
+
assert.equal(f2[0]?.severity, "medium");
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
it("degrades unreadable severity to a recorded non-blocking entry (never a rollback driver)", () => {
|
|
245
|
+
const { findings } = parseAuditReport("- `F-003` — tasks: whatever; note: no severity field", ALL_IDS);
|
|
246
|
+
assert.equal(findings[0]?.severity, "malformed");
|
|
247
|
+
assert.deepEqual(findings[0]?.taskIds, []);
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
it("degrades a missing mandatory tasks field", () => {
|
|
251
|
+
const { findings } = parseAuditReport("- `F-004` — severity: high; note: tasks field absent", ALL_IDS);
|
|
252
|
+
assert.equal(findings[0]?.severity, "malformed");
|
|
253
|
+
});
|
|
254
|
+
|
|
255
|
+
it("resolves duplicate ids to the first bullet", () => {
|
|
256
|
+
const report = [
|
|
257
|
+
"- `F-001` — severity: high; tasks: Task-1; note: first",
|
|
258
|
+
"- `F-001` — severity: low; tasks: none; note: second",
|
|
259
|
+
].join("\n");
|
|
260
|
+
const { findings } = parseAuditReport(report, ALL_IDS);
|
|
261
|
+
assert.equal(findings.length, 1);
|
|
262
|
+
assert.equal(findings[0].note, "first");
|
|
263
|
+
});
|
|
264
|
+
|
|
265
|
+
it("a finding bullet citing a VC verdict never registers that verdict", () => {
|
|
266
|
+
const report = "- `F-005` — severity: high; tasks: Task-1; note: cites VC-004 verdict: pass; evidence: z";
|
|
267
|
+
const { passed, failed } = parseAuditReport(report, ALL_IDS);
|
|
268
|
+
assert.equal(passed.length + failed.length, 0);
|
|
269
|
+
});
|
|
270
|
+
|
|
271
|
+
it("prose mentioning F-### outside a bullet is ignored", () => {
|
|
272
|
+
const { findings } = parseAuditReport("also prose mentions F-009 not a bullet\n- `F-001` — severity: low; tasks: none; note: real", ALL_IDS);
|
|
273
|
+
assert.deepEqual(findings.map((f) => f.id), ["F-001"]);
|
|
274
|
+
});
|
|
275
|
+
|
|
276
|
+
it("applyAuditOutcome carries findings through to the outcome", () => {
|
|
277
|
+
const parsed = parseAuditReport(GOOD, ["VC-001"]);
|
|
278
|
+
const outcome = applyAuditOutcome(1, parsed, GOOD);
|
|
279
|
+
assert.equal(outcome.findings?.length, 2);
|
|
280
|
+
});
|
|
281
|
+
});
|
|
282
|
+
|
|
283
|
+
describe("section-aware verdict parsing (v0.9.1 F-011)", () => {
|
|
284
|
+
it("parses heading-style sections: id heading + verdict on its own line below", () => {
|
|
285
|
+
const report = [
|
|
286
|
+
"## 1. Verification verdicts",
|
|
287
|
+
"",
|
|
288
|
+
"### VC-001",
|
|
289
|
+
"- verdict: pass; evidence: src/a.ts",
|
|
290
|
+
"",
|
|
291
|
+
"### VC-002",
|
|
292
|
+
"- verdict: **fail**; evidence: src/c.ts",
|
|
293
|
+
"",
|
|
294
|
+
"### VC-003",
|
|
295
|
+
"verdict: undeterminable",
|
|
296
|
+
].join("\n");
|
|
297
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ["VC-001", "VC-002", "VC-003"]);
|
|
298
|
+
assert.deepEqual(passed, ["VC-001"]);
|
|
299
|
+
assert.deepEqual(failed, ["VC-002"]);
|
|
300
|
+
assert.deepEqual(undeterminable, ["VC-003"]);
|
|
301
|
+
});
|
|
302
|
+
|
|
303
|
+
it("a bare verdict with no open section is ignored (never misattributed)", () => {
|
|
304
|
+
const report = "Some preamble mentioning verdict: pass with no section above\n### VC-001\n- verdict: fail";
|
|
305
|
+
const { passed, failed } = parseAuditReport(report, ["VC-001"]);
|
|
306
|
+
assert.deepEqual(passed, []);
|
|
307
|
+
assert.deepEqual(failed, ["VC-001"]);
|
|
308
|
+
});
|
|
309
|
+
|
|
310
|
+
it("an unknown section id does not capture later bare verdicts", () => {
|
|
311
|
+
const report = ["### VC-999", "- verdict: pass", "### VC-002", "- verdict: pass"].join("\n");
|
|
312
|
+
const { passed, undeterminable } = parseAuditReport(report, ["VC-002"]);
|
|
313
|
+
assert.deepEqual(passed, ["VC-002"]);
|
|
314
|
+
assert.deepEqual(undeterminable, []);
|
|
315
|
+
});
|
|
316
|
+
|
|
317
|
+
it("finding bullets never open a verdict section", () => {
|
|
318
|
+
const report = [
|
|
319
|
+
"### VC-001",
|
|
320
|
+
"- `F-005` — severity: high; tasks: Task-1; note: cites verdict: pass inside a note; evidence: e",
|
|
321
|
+
"- verdict: fail",
|
|
322
|
+
].join("\n");
|
|
323
|
+
const { passed, failed } = parseAuditReport(report, ["VC-001"]);
|
|
324
|
+
assert.deepEqual(passed, []);
|
|
325
|
+
assert.deepEqual(failed, ["VC-001"]);
|
|
326
|
+
assert.equal(parseAuditReport(report, ["VC-001"]).findings[0]?.severity, "high");
|
|
327
|
+
});
|
|
328
|
+
|
|
329
|
+
it("knownTaskIds filters finding mappings to plan tasks (F-003)", () => {
|
|
330
|
+
const report = "- `F-001` — severity: high; tasks: Task-1, VC-007, bogus; note: n; evidence: e";
|
|
331
|
+
const { findings } = parseAuditReport(report, [], new Set(["Task-1"]));
|
|
332
|
+
assert.deepEqual(findings[0]?.taskIds, ["Task-1"]);
|
|
333
|
+
});
|
|
334
|
+
});
|
|
335
|
+
|
|
336
|
+
describe("review brief dual-output contract (v0.9)", () => {
|
|
337
|
+
it("demands both sections: verdicts and findings grammar", () => {
|
|
338
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
339
|
+
assert.match(task, /1\. Verification verdicts/);
|
|
340
|
+
assert.match(task, /2\. Implementation findings/);
|
|
341
|
+
assert.match(task, /severity: high \| medium \| low/);
|
|
342
|
+
assert.match(task, /proposed-task:/);
|
|
343
|
+
});
|
|
344
|
+
|
|
345
|
+
it("lists plan tasks as the valid mapping domain", () => {
|
|
346
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
347
|
+
assert.match(task, /Plan tasks \(the only ids valid in a finding's tasks field\):/);
|
|
348
|
+
assert.match(task, /`Task-3\.1`: injection/);
|
|
349
|
+
});
|
|
350
|
+
|
|
351
|
+
it("injects prior unresolved findings with the stable-id reuse instruction", () => {
|
|
352
|
+
const prior = [{
|
|
353
|
+
id: "F-001", severity: "high" as const, taskIds: ["Task-2"], note: "still broken",
|
|
354
|
+
evidence: "src/b.ts", raw: "- `F-001` — severity: high; tasks: Task-2; note: still broken",
|
|
355
|
+
}];
|
|
356
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 3, prior);
|
|
357
|
+
assert.match(task, /reuse these exact ids while the problem persists/);
|
|
358
|
+
assert.match(task, /`F-001` — severity: high; tasks: Task-2/);
|
|
359
|
+
});
|
|
360
|
+
|
|
361
|
+
it("marks the first findings round when no prior list exists", () => {
|
|
362
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
363
|
+
assert.match(task, /first round with findings in scope/);
|
|
364
|
+
});
|
|
365
|
+
});
|
|
366
|
+
|
|
367
|
+
describe("review round report findings lines (v0.9)", () => {
|
|
368
|
+
it("emits the high-findings line and per-finding detail", async () => {
|
|
369
|
+
const { writeReviewRoundReport } = await import("../src/auditor.ts");
|
|
370
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-round-report-"));
|
|
371
|
+
try {
|
|
372
|
+
const file = writeReviewRoundReport(dir, {
|
|
373
|
+
budgetRound: 2,
|
|
374
|
+
attempt: 1,
|
|
375
|
+
outcome: "failed",
|
|
376
|
+
passed: ["VC-001"],
|
|
377
|
+
failed: [],
|
|
378
|
+
undeterminable: [],
|
|
379
|
+
findings: [
|
|
380
|
+
{ id: "F-001", severity: "high", taskIds: ["Task-2"], note: "broken", evidence: "e", raw: "raw" },
|
|
381
|
+
{ id: "F-002", severity: "medium", taskIds: [], proposedTask: "tidy up", note: "polish", evidence: "e", raw: "raw" },
|
|
382
|
+
],
|
|
383
|
+
coveredTaskIds: ["Task-1", "Task-2"],
|
|
384
|
+
report: "## Report\nbody",
|
|
385
|
+
});
|
|
386
|
+
assert.ok(file);
|
|
387
|
+
const text = fs.readFileSync(file, "utf8");
|
|
388
|
+
assert.match(text, /- high findings: F-001/);
|
|
389
|
+
assert.match(text, /F-002 \(medium; proposed: tidy up\)/);
|
|
390
|
+
} finally {
|
|
391
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
392
|
+
}
|
|
393
|
+
});
|
|
394
|
+
});
|