agentflowctl 0.13.1 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/engine.js CHANGED
@@ -3,18 +3,19 @@ import { join } from "node:path";
3
3
  import { z } from "zod";
4
4
  import { config } from "./config.js";
5
5
  import { arbitrationDecision } from "./arbitration.js";
6
- import { detectProjectDefaults, withProjectDefaults } from "./detect.js";
6
+ import { detectProjectDefaults, usesTestFramework, withProjectDefaults } from "./detect.js";
7
7
  import { escapeXml, opinion, reviewIssue } from "./feedback.js";
8
8
  import { changedFiles, commitAll, discardChanges, git, headCommit, resetTo } from "./git.js";
9
9
  import { acceptHandoff, openActions, prepareHandoff, previewHandoff, readHandoff, recoverHandoff, reviewHandoffGate, validateHandoffResponse } from "./handoff.js";
10
- import { flowDir, logDir, projectRoot, runDir, worktreeDir } from "./paths.js";
10
+ import { flowDir, logDir, planArbitrationPath, planReviewStatePath, projectRoot, runDir, worktreeDir } from "./paths.js";
11
11
  import { CMD_AGENT, nextLogFile } from "./logs.js";
12
12
  import { exec } from "./proc.js";
13
13
  import { arbiterPanel, availableAgent, fixAgent, planAgent, planFixAgent, reviewers, specAgent, taskAgents } from "./roles.js";
14
14
  import { resolveAgent, runAgent, runCommand } from "./runner.js";
15
15
  import { AcceptanceList, ArbiterResult, ConsistentReviewResult, RepoConfig, TaskList, } from "./schemas.js";
16
- import { addSubstitution, addUsage, agentRuns, saveRun } from "./store.js";
16
+ import { addRetry, addSubstitution, addUsage, agentRuns, saveRun } from "./store.js";
17
17
  import { clearModelReviewFailure, clearModelReviewStage, recordModelReviewFailure, selectModel } from "./modelSelection.js";
18
+ import { applyReviewVerdicts, dirtyGroups, dirtyTaskIds, extractPlanEvidence, groupReviewerCount, layeredReview, neighborTasks, planContentKey, planOverview, planReviewIndex, readPendingArbitration, readPlanReviewState, repliesForTasks, reviewFingerprint, roundProgress, } from "./planReview.js";
18
19
  import { orderTasks, taskAcceptance, validateTaskComplexity } from "./tasks.js";
19
20
  import { readJsonFile, renderPrompt, tail } from "./util.js";
20
21
  // ───────────────────────── 共用工具 ─────────────────────────
@@ -33,6 +34,10 @@ export class QuotaPause extends Error {
33
34
  }
34
35
  /** 這次執行中已確認額度用完的 agent;程序結束即清空,resume 時會重新嘗試 */
35
36
  const exhausted = new Set();
37
+ /** 只供測試使用:清掉這個程序內記下的額度用完 agent,模擬新啟動的程序 */
38
+ export function resetQuotaState() {
39
+ exhausted.clear();
40
+ }
36
41
  function handoffTarget(run) {
37
42
  return ["spec", "plan", "plan_review", "plan_fix"].includes(run.stage) ? "plan" : "code";
38
43
  }
@@ -61,7 +66,7 @@ async function agentStep(run, planned, step, prompt, mode) {
61
66
  info(run, `🔁 ${agent} 額度已用完,${step} 由 ${sub} 代打${note ? `(注意:${note})` : ""}`);
62
67
  agent = sub;
63
68
  }
64
- const selected = selectModel(run, cfg, agent, step, /^T-\d+-/.test(step) ? loadOrderedTasks(run)[run.taskIndex]?.complexity : undefined, mode.kind === "review" ? agent : undefined);
69
+ const selected = selectModel(run, cfg, agent, step, /^T-\d+-/.test(step) ? loadOrderedTasks(run)[run.taskIndex]?.complexity : undefined, mode.kind === "review" ? agent : undefined, mode.modelScope);
65
70
  info(run, `🤖 ${step}:${agent} 使用 ${selected.name ?? "CLI 預設(名稱未知)"}${selected.insufficient ? `(低於目標 ${selected.targetStrength})` : ""}`);
66
71
  const callKey = handoffKey(run, step, mode.slot ?? 0, agent);
67
72
  prepareHandoff(run.id, callKey, handoffTarget(run), mode.blind ?? false);
@@ -112,20 +117,32 @@ function reportMeta(run, agent, r) {
112
117
  if (r.meta.concerns)
113
118
  info(run, ` 💭 ${agent} 的疑慮:${r.meta.concerns}`);
114
119
  }
115
- function readFeedback(run) {
116
- const p = flowFile(run, "feedback.md");
120
+ /** 讀 .flow/ 下的文字檔,沒有檔就當空字串 */
121
+ function flowText(run, name) {
122
+ const p = flowFile(run, name);
117
123
  return existsSync(p) ? readFileSync(p, "utf8") : "";
118
124
  }
119
- /** 關卡未通過:寫入 feedback.md 給下一次嘗試參考,超過上限就讓整個 run 失敗 */
120
- function retry(run, key, reason, backTo) {
125
+ function readFeedback(run) {
126
+ return flowText(run, "feedback.md");
127
+ }
128
+ /** 計畫審查第幾輪仍有人要求修改時交付仲裁;固定值,不受重試上限影響 */
129
+ const PLAN_ARBITRATION_ROUND = 2;
130
+ /** 這個 run 的重試上限:`--max-attempts` 存在 run 裡,沒設定才用環境變數 */
131
+ function attemptLimit(run) {
132
+ return run.maxAttempts ?? config.maxAttempts;
133
+ }
134
+ /** 關卡未通過:寫入 feedback.md 給下一次嘗試參考,並把原因分類記進 retries.jsonl;超過上限就讓整個 run 失敗 */
135
+ function retry(run, key, reason, backTo, category) {
121
136
  const n = (run.attempts[key] ?? 0) + 1;
122
137
  const attempts = { ...run.attempts, [key]: n };
138
+ const final = n >= attemptLimit(run);
123
139
  mkdirSync(flowDir(run.id), { recursive: true });
124
140
  writeFileSync(flowFile(run, "feedback.md"), `# 前次嘗試未通過(第 ${n} 次)\n\n${reason}\n`);
125
- if (n >= config.maxAttempts) {
126
- return { ...run, attempts, stage: "failed", failedStage: backTo, failureReason: `${key} 連續失敗 ${n} 次:${tail(reason, 500)}` };
141
+ addRetry(run.id, { key, backTo, category, attempt: n, final }, run.updatedAt);
142
+ if (final) {
143
+ return { ...run, attempts, stage: "failed", failedStage: backTo, failureCategory: "retry_limit", failureReason: `${key} 連續失敗 ${n} 次:${tail(reason, 500)}` };
127
144
  }
128
- info(run, `⚠️ ${key} 未通過,重試(${n}/${config.maxAttempts})`);
145
+ info(run, `⚠️ ${key} 未通過,重試(${n}/${attemptLimit(run)})`);
129
146
  return { ...run, attempts, stage: backTo };
130
147
  }
131
148
  function succeed(run, key, next) {
@@ -147,6 +164,17 @@ export function loadRepoConfig() {
147
164
  throw new Error(r.error);
148
165
  return r.data;
149
166
  }
167
+ /** 專案有測試框架(偵測到或手動設定 test) */
168
+ function hasTestFramework() {
169
+ const root = projectRoot();
170
+ const p = join(root, "flow.config.json");
171
+ const r = existsSync(p) ? readJsonFile(p, z.unknown()) : undefined;
172
+ return usesTestFramework(r?.ok ? r.data : {}, detectProjectDefaults(root));
173
+ }
174
+ /** 這個任務要不要走紅綠燈:沒有測試框架一律不走,其餘依 planner 標記(沒標視為要走) */
175
+ function taskUsesTdd(task, framework) {
176
+ return framework && task.tdd !== false;
177
+ }
150
178
  function loadOrderedTasks(run) {
151
179
  const r = readJsonFile(flowFile(run, "tasks.ordered.json"), TaskList);
152
180
  if (!r.ok)
@@ -161,24 +189,26 @@ async function specStage(run) {
161
189
  const { r } = outcome;
162
190
  await discardChanges(worktreeDir(run.id)); // 這個階段只允許寫 .flow/
163
191
  if (!r.ok)
164
- return retry(run, "spec", `Agent 執行失敗:${r.summary}`, "spec");
192
+ return retry(run, "spec", `Agent 執行失敗:${r.summary}`, "spec", "agent_error");
165
193
  if (!existsSync(flowFile(run, "spec.md")))
166
- return retry(run, "spec", "缺少 .flow/spec.md", "spec");
194
+ return retry(run, "spec", "缺少 .flow/spec.md", "spec", "missing_artifact");
167
195
  const ac = readJsonFile(flowFile(run, "acceptance.json"), AcceptanceList);
168
196
  if (!ac.ok)
169
- return retry(run, "spec", ac.error, "spec");
197
+ return retry(run, "spec", ac.error, "spec", "format_invalid");
170
198
  const ids = ac.data.map((a) => a.id);
171
199
  if (new Set(ids).size !== ids.length)
172
- return retry(run, "spec", "驗收條件 id 有重複", "spec");
200
+ return retry(run, "spec", "驗收條件 id 有重複", "spec", "format_invalid");
173
201
  const handoffError = finishHandoff(run, outcome, "writer");
174
202
  if (handoffError)
175
- return retry(run, "spec", handoffError, "spec");
203
+ return retry(run, "spec", handoffError, "spec", "handoff_invalid");
176
204
  return succeed(run, "spec", "plan");
177
205
  }
178
206
  // ── 規格與計畫檔案:計畫審查、仲裁與計畫定案後的所有階段只能讀,不能改 ──
179
207
  const PLAN_FILES = ["spec.md", "acceptance.json", "plan.md", "tasks.json"];
180
208
  /** 計畫定案後(實作、修正、程式碼審查)另外依賴排好的任務順序,同樣不能被改 */
181
209
  const LOCKED_FILES = [...PLAN_FILES, "tasks.ordered.json"];
210
+ /** 審查、修訂與仲裁的快照另外包含審查回應;不要併進 PLAN_FILES,定案後的階段不依賴它 */
211
+ const PLAN_REPLY_FILES = [...PLAN_FILES, "plan-replies.md"];
182
212
  function snapshotPlan(run, files = PLAN_FILES) {
183
213
  return Object.fromEntries(files.map((f) => [f, existsSync(flowFile(run, f)) ? readFileSync(flowFile(run, f), "utf8") : undefined]));
184
214
  }
@@ -223,16 +253,19 @@ function acceptPlan(run, ordered) {
223
253
  function announceTasks(run, ordered) {
224
254
  const cfg = loadRepoConfig();
225
255
  info(run, `📋 共 ${ordered.length} 個任務:${ordered.map((t) => t.id).join(" → ")}`);
256
+ const framework = hasTestFramework();
226
257
  ordered.forEach((t, i) => {
227
258
  const a = taskAgents(run.cycle, i, cfg.tddSplit, run.id);
228
- info(run, ` ${t.id} 測試:${a.tests} 實作:${a.code} 審查:${a.review}`);
259
+ info(run, taskUsesTdd(t, framework)
260
+ ? ` ${t.id} 測試:${a.tests} 實作:${a.code} 審查:${a.review}`
261
+ : ` ${t.id} 略過 TDD 實作:${a.code} 審查:${a.review}`);
229
262
  });
230
263
  }
231
264
  /** 計畫定案後:預設直接開始實作;--manual-plan 時才停下來等人 */
232
265
  function planSettled(run, key) {
233
266
  const pending = openActions(readHandoff(run.id), "plan");
234
267
  if (pending.length)
235
- return retry(run, "plan-handoff", `計畫仍有未結交接事項:${pending.map((item) => item.id).join("、")}`, "plan_fix");
268
+ return retry(run, "plan-handoff", `計畫仍有未結交接事項:${pending.map((item) => item.id).join("、")}`, "plan_fix", "open_handoff");
236
269
  const next = run.autopilot ? "implement" : "awaiting_approval";
237
270
  const ordered = loadOrderedTasks(run);
238
271
  announceTasks(run, ordered);
@@ -252,59 +285,75 @@ async function planStage(run) {
252
285
  const { r, agent: actual } = outcome;
253
286
  await discardChanges(worktreeDir(run.id));
254
287
  if (!r.ok)
255
- return retry(run, "plan", `Agent 執行失敗:${r.summary}`, "plan");
288
+ return retry(run, "plan", `Agent 執行失敗:${r.summary}`, "plan", "agent_error");
256
289
  const ordered = validatePlan(run);
257
290
  if (typeof ordered === "string")
258
- return retry(run, "plan", ordered, "plan");
291
+ return retry(run, "plan", ordered, "plan", "format_invalid");
259
292
  const handoffError = finishHandoff(run, outcome, "writer");
260
293
  if (handoffError)
261
- return retry(run, "plan", handoffError, "plan");
294
+ return retry(run, "plan", handoffError, "plan", "handoff_invalid");
262
295
  acceptPlan(run, ordered);
263
296
  rmSync(flowFile(run, "plan-review-last.txt"), { force: true });
264
297
  return { ...succeed(run, "plan", "plan_review"), planWriter: actual };
265
298
  }
266
299
  async function planReviewStage(run) {
267
300
  const cfg = loadRepoConfig();
268
- const author = run.planWriter ?? planAgent(run.cycle, run.id);
269
- const round = (run.attempts["plan-review"] ?? 0) + 1;
270
- const panel = reviewers(run.cycle, author, cfg.planReviewQuorum, `${run.id}:plan-review:${round}`);
271
- const issues = [];
272
- const issueLines = []; // 只含意見內容、不含審查者名稱,用來偵測僵持
273
- let firstObjector;
274
- for (const reviewer of panel) {
275
- info(run, `🧐 計畫審查第 ${round} 輪(${reviewer},作者 ${author})`);
276
- const snap = snapshotPlan(run);
277
- rmSync(flowFile(run, "plan-review.json"), { force: true });
278
- const outcome = await agentStep(run, reviewer, "plan-review", renderPrompt("plan-review", { reviewer, author, requirement: run.requirement }), { kind: "review", slot: panel.indexOf(reviewer), reset: async () => { await discardChanges(worktreeDir(run.id)); restorePlan(run, snap); } });
279
- const { r } = outcome;
280
- await discardChanges(worktreeDir(run.id));
281
- const tampered = restorePlan(run, snap);
282
- if (tampered.length)
283
- info(run, ` ↩️ 已還原審查者修改的檔案:${tampered.join(", ")}`);
284
- if (!r.ok)
285
- return retry(recordModelReviewFailure(run, "plan-review", reviewer), "plan-review-run", `Agent 執行失敗:${r.summary}`, "plan_review");
286
- const review = readJsonFile(flowFile(run, "plan-review.json"), ConsistentReviewResult);
287
- if (!review.ok)
288
- return retry(recordModelReviewFailure(run, "plan-review", reviewer), "plan-review-run", review.error, "plan_review");
289
- const handoffError = finishHandoff(run, outcome, "reviewer", { target: "plan", verdict: review.data.verdict });
290
- if (handoffError)
291
- return retry(recordModelReviewFailure(run, "plan-review", reviewer), "plan-review-run", handoffError, "plan_review");
292
- run = clearModelReviewFailure(run, "plan-review", reviewer);
293
- // 審查紀錄移到 worktree 外面:之後的仲裁者看不到是哪一家提的意見
294
- mkdirSync(join(runDir(run.id), "reviews"), { recursive: true });
295
- renameSync(flowFile(run, "plan-review.json"), join(runDir(run.id), "reviews", `plan-review-${round}-${reviewer}.json`));
296
- if (review.data.verdict === "approve") {
297
- info(run, ` ✓ ${reviewer} 核准計畫`);
298
- continue;
299
- }
300
- info(run, ` ✗ ${reviewer} 要求修改計畫`);
301
- firstObjector ??= reviewer;
302
- const lines = review.data.items
303
- .filter((i) => i.status !== "met")
304
- .map((i) => reviewIssue(i.criterion, i.status, i.note));
305
- issueLines.push(...lines);
306
- issues.push(opinion(reviewer, lines));
301
+ const pending = pendingArbitration(run, cfg);
302
+ if (pending) {
303
+ info(run, "⚖️ 上次仲裁沒有完成,直接回到仲裁(不重跑計畫審查)");
304
+ return arbitratePlan({ ...run, planReviewer: pending.planReviewer });
305
+ }
306
+ const layered = loadLayeredPlan(run, cfg);
307
+ return layered ? planReviewLayered(run, layered, cfg) : planReviewFull(run, cfg);
308
+ }
309
+ /**
310
+ * 整份審查與分層審查共用的單次呼叫:還原審查者改過的計畫檔、驗證裁決與交接。
311
+ * 失敗時回傳已呼叫 retry 的 run、沒有 collected,呼叫端應直接回傳這個 run。
312
+ */
313
+ async function collectPlanReview(run, spec) {
314
+ const { reviewer, step } = spec;
315
+ const snap = snapshotPlan(run, PLAN_REPLY_FILES);
316
+ rmSync(flowFile(run, spec.output), { force: true });
317
+ const outcome = await agentStep(run, reviewer, step, spec.prompt, {
318
+ kind: "review", slot: spec.slot, modelScope: spec.scope,
319
+ reset: async () => { await discardChanges(worktreeDir(run.id)); restorePlan(run, snap); },
320
+ });
321
+ await discardChanges(worktreeDir(run.id));
322
+ const tampered = restorePlan(run, snap);
323
+ if (tampered.length)
324
+ info(run, ` ↩️ 已還原審查者修改的檔案:${tampered.join(", ")}`);
325
+ const stop = (category, reason) => ({
326
+ run: retry(recordModelReviewFailure(run, step, reviewer, spec.scope), "plan-review-run", reason, "plan_review", category),
327
+ });
328
+ if (!outcome.r.ok)
329
+ return stop("agent_error", `Agent 執行失敗:${outcome.r.summary}`);
330
+ const review = readJsonFile(flowFile(run, spec.output), ConsistentReviewResult);
331
+ if (!review.ok)
332
+ return stop("format_invalid", review.error);
333
+ // 群審查不帶關卡:索引要求修改並新增事項後,群的核准不算矛盾;最後由 planSettled 檢查未結事項
334
+ const handoffError = finishHandoff(run, outcome, "reviewer", spec.gated ? { target: "plan", verdict: review.data.verdict } : undefined);
335
+ if (handoffError)
336
+ return stop("handoff_invalid", handoffError);
337
+ run = clearModelReviewFailure(run, step, reviewer, spec.scope);
338
+ // 審查紀錄移到 worktree 外面:之後的仲裁者看不到是哪一家提的意見
339
+ mkdirSync(join(runDir(run.id), "reviews"), { recursive: true });
340
+ renameSync(flowFile(run, spec.output), join(runDir(run.id), "reviews", spec.archive));
341
+ if (review.data.verdict === "approve") {
342
+ info(run, ` ✓ ${reviewer} 核准${spec.subject}`);
343
+ return { run, collected: { reviewer, verdict: "approve", issueLines: [] } };
307
344
  }
345
+ info(run, ` ✗ ${reviewer} 要求修改${spec.subject}`);
346
+ const issueLines = review.data.items
347
+ .filter((i) => i.status !== "met")
348
+ .map((i) => reviewIssue(i.criterion, i.status, i.note));
349
+ return { run, collected: { reviewer, verdict: "changes_requested", issueLines } };
350
+ }
351
+ /** 一輪審查都產出合法裁決後:全部核准就定案,否則退回修訂,僵持或達輪數上限時交付仲裁 */
352
+ async function concludePlanReview(run, cfg, round, calls) {
353
+ const objections = calls.filter((call) => call.verdict !== "approve");
354
+ const issueLines = objections.flatMap((call) => call.issueLines);
355
+ const issues = objections.map((call) => opinion(call.reviewer, call.issueLines));
356
+ const firstObjector = objections[0]?.reviewer;
308
357
  run = clearModelReviewStage(run, "plan-review");
309
358
  const attempts = { ...run.attempts };
310
359
  delete attempts["plan-review-run"];
@@ -314,10 +363,11 @@ async function planReviewStage(run) {
314
363
  const report = `計畫審查要求修改:\n\n${issues.join("\n\n")}`;
315
364
  // 僵持偵測:意見和上一輪完全相同,代表修改沒有進展
316
365
  const lastPath = flowFile(run, "plan-review-last.txt");
317
- const fingerprint = [...issueLines].sort().join("\n");
366
+ const fingerprint = reviewFingerprint(issueLines);
318
367
  const stalled = existsSync(lastPath) && readFileSync(lastPath, "utf8") === fingerprint;
319
368
  writeFileSync(lastPath, fingerprint);
320
- const exhausted = round >= config.maxAttempts;
369
+ // 修訂過一次仍被要求修改就交付仲裁,不等到重試上限
370
+ const exhausted = round >= PLAN_ARBITRATION_ROUND;
321
371
  if ((stalled || exhausted) && cfg.planArbiter) {
322
372
  // 雙盲:帶有審查者名稱的 feedback.md 不留在 worktree,完整報告另存到 worktree 外
323
373
  rmSync(flowFile(run, "feedback.md"), { force: true });
@@ -326,9 +376,165 @@ async function planReviewStage(run) {
326
376
  // 給仲裁者的爭議清單不含任何模型名稱
327
377
  writeFileSync(flowFile(run, "dispute.md"), `# 尚未解決的審查意見\n\n${[...new Set(issueLines)].join("\n")}\n`);
328
378
  info(run, `⚖️ 計畫審查${stalled ? "意見沒有變化" : `已達 ${round} 輪`},交付仲裁`);
379
+ // 仲裁暫停(裁決無效或額度用完)後 resume 仍停在 plan_review,靠這份紀錄直接回到仲裁
380
+ mkdirSync(runDir(run.id), { recursive: true });
381
+ writeFileSync(planArbitrationPath(run.id), JSON.stringify({ planReviewer: firstObjector, planKey: currentPlanKey(run) }, null, 2));
329
382
  return arbitratePlan({ ...run, planReviewer: firstObjector });
330
383
  }
331
- return { ...retry(run, "plan-review", report, "plan_fix"), planReviewer: firstObjector };
384
+ return { ...retry(run, "plan-review", report, "plan_fix", "review_changes"), planReviewer: firstObjector };
385
+ }
386
+ /** 整份審查:每位審查者讀完整份規格與計畫 */
387
+ async function planReviewFull(run, cfg) {
388
+ // 整份審查不維護分層狀態;之後若又切回分層,不能沿用這之前的 approve
389
+ rmSync(planReviewStatePath(run.id), { force: true });
390
+ const author = run.planWriter ?? planAgent(run.cycle, run.id);
391
+ const round = (run.attempts["plan-review"] ?? 0) + 1;
392
+ const panel = reviewers(run.cycle, author, cfg.planReviewQuorum, `${run.id}:plan-review:${round}`);
393
+ const calls = [];
394
+ for (const [slot, reviewer] of panel.entries()) {
395
+ info(run, `🧐 計畫審查第 ${round} 輪(${reviewer},作者 ${author})`);
396
+ const passed = await collectPlanReview(run, {
397
+ reviewer, step: "plan-review", slot, gated: true, subject: "計畫",
398
+ prompt: renderPrompt("plan-review", { reviewer, author, requirement: run.requirement }),
399
+ output: "plan-review.json", archive: `plan-review-${round}-${reviewer}.json`,
400
+ });
401
+ run = passed.run;
402
+ if (!passed.collected)
403
+ return run;
404
+ calls.push(passed.collected);
405
+ }
406
+ return concludePlanReview(run, cfg, round, calls);
407
+ }
408
+ /** 本輪進度的雜湊涵蓋的檔案:任一份變了,就不能沿用先前的審查結果 */
409
+ const PLAN_KEY_FILES = ["tasks.json", "acceptance.json", "plan.md", "plan-replies.md"];
410
+ function currentPlanKey(run) {
411
+ return planContentKey(PLAN_KEY_FILES.map((name) => flowText(run, name)));
412
+ }
413
+ /**
414
+ * 上次交付仲裁卻沒有得出裁決(暫停)時的紀錄。計畫在這之間被改過、或爭議清單不見了,
415
+ * 就當成沒有待完成的仲裁並刪掉紀錄,重新審查。暫停期間關掉了 planArbiter 也一樣,
416
+ * 連同爭議清單一起刪掉。
417
+ */
418
+ function pendingArbitration(run, cfg) {
419
+ const path = planArbitrationPath(run.id);
420
+ if (!existsSync(path))
421
+ return undefined;
422
+ const pending = readPendingArbitration(readFileSync(path, "utf8"));
423
+ if (cfg.planArbiter && pending && pending.planKey === currentPlanKey(run) && existsSync(flowFile(run, "dispute.md")))
424
+ return pending;
425
+ rmSync(path, { force: true });
426
+ if (!cfg.planArbiter)
427
+ rmSync(flowFile(run, "dispute.md"), { force: true });
428
+ return undefined;
429
+ }
430
+ /** 符合分層條件時讀出分層審查需要的資料;不符合時回傳 undefined,已達門檻的會印出原因 */
431
+ function loadLayeredPlan(run, cfg) {
432
+ const tasks = readJsonFile(flowFile(run, "tasks.json"), TaskList);
433
+ const acceptance = readJsonFile(flowFile(run, "acceptance.json"), AcceptanceList);
434
+ if (!tasks.ok || !acceptance.ok)
435
+ return undefined;
436
+ const planMd = flowText(run, "plan.md");
437
+ const decision = layeredReview(tasks.data, planMd, cfg.planReviewLayers);
438
+ if (!decision.layered) {
439
+ if (decision.reason)
440
+ info(run, `📋 這次計畫審查讀整份計畫:${decision.reason}`);
441
+ return undefined;
442
+ }
443
+ const statePath = planReviewStatePath(run.id);
444
+ const state = existsSync(statePath) ? readPlanReviewState(readFileSync(statePath, "utf8")) : undefined;
445
+ return {
446
+ tasks: tasks.data, acceptance: acceptance.data, planMd, state,
447
+ dirty: dirtyGroups(decision.groups, dirtyTaskIds(state?.reviewed, tasks.data, acceptance.data, planMd)),
448
+ };
449
+ }
450
+ /** 分層審查:每輪一次索引審查,再只審有變動的任務群 */
451
+ async function planReviewLayered(run, layered, cfg) {
452
+ const author = run.planWriter ?? planAgent(run.cycle, run.id);
453
+ const round = (run.attempts["plan-review"] ?? 0) + 1;
454
+ const panel = reviewers(run.cycle, author, cfg.planReviewQuorum, `${run.id}:plan-review:${round}`);
455
+ const replies = flowText(run, "plan-replies.md");
456
+ const statePath = planReviewStatePath(run.id);
457
+ const reviewed = layered.state?.reviewed;
458
+ const planKey = currentPlanKey(run);
459
+ // 同一輪因某一呼叫失敗而重跑時,沿用已成功的呼叫:不重複付費,已套用的交接也不會被重新提出
460
+ const progress = roundProgress(layered.state, round, planKey) ?? { round, planKey, calls: [] };
461
+ const saveState = (state) => {
462
+ mkdirSync(runDir(run.id), { recursive: true });
463
+ writeFileSync(statePath, JSON.stringify(state, null, 2));
464
+ };
465
+ const done = new Map(progress.calls.map((call) => [call.key, call]));
466
+ const calls = [];
467
+ // 沿用的呼叫也佔一格,重跑時每個呼叫的 slot 才不會變
468
+ let nextSlot = 0;
469
+ /** 執行或沿用一次呼叫;失敗時回傳 false,run 已是 retry 後的狀態 */
470
+ const runCall = async (key, taskIds, label, spec) => {
471
+ const slot = nextSlot++;
472
+ const reused = done.get(key);
473
+ if (reused) {
474
+ info(run, ` ↪ 沿用本輪已完成的${label}(${reused.reviewer})`);
475
+ calls.push(reused);
476
+ return true;
477
+ }
478
+ info(run, `🧐 ${label}第 ${round} 輪(${spec.reviewer},作者 ${author})`);
479
+ // run 是外層參數,刻意在閉包裡更新:後續呼叫與最後的彙總都要看到 retry、clearModelReviewFailure 之後的 run
480
+ const passed = await collectPlanReview(run, { ...spec, slot });
481
+ run = passed.run;
482
+ if (!passed.collected)
483
+ return false;
484
+ // 真的執行並成功就是有進展:同一輪不同呼叫輪流失敗時,不會累計到重試上限而讓 run 失敗。
485
+ // 進度寫在 round 裡、不會重跑,所以一輪最多失敗「呼叫數 × maxAttempts」次。
486
+ // 整份審查每次重跑整輪,不能這樣歸零,否則同一位審查者反覆失敗會無限重試。
487
+ const attempts = { ...run.attempts };
488
+ delete attempts["plan-review-run"];
489
+ run = { ...run, attempts };
490
+ const call = { key, ...passed.collected, ...(taskIds ? { taskIds } : {}) };
491
+ progress.calls.push(call);
492
+ calls.push(call);
493
+ saveState({ version: 1, ...(reviewed ? { reviewed } : {}), round: progress });
494
+ return true;
495
+ };
496
+ for (const reviewer of panel) {
497
+ const ok = await runCall(`index:${reviewer}`, undefined, "計畫索引審查", {
498
+ reviewer, step: "plan-review", gated: true, subject: "計畫索引",
499
+ prompt: renderPrompt("plan-review-index", {
500
+ reviewer, author, requirement: run.requirement,
501
+ index: planReviewIndex(layered.tasks),
502
+ acceptance: JSON.stringify(layered.acceptance, null, 2),
503
+ overview: planOverview(layered.planMd),
504
+ replies,
505
+ }),
506
+ output: "plan-review.json", archive: `plan-review-${round}-${reviewer}.json`,
507
+ });
508
+ if (!ok)
509
+ return run;
510
+ }
511
+ for (const group of layered.dirty) {
512
+ const groupTasks = layered.tasks.filter((task) => group.taskIds.includes(task.id));
513
+ const acceptanceIds = new Set(groupTasks.flatMap((task) => task.acceptance));
514
+ const neighbors = neighborTasks(layered.tasks, group.taskIds).map(({ id, title, description, dependsOn }) => ({ id, title, description, dependsOn }));
515
+ const groupPanel = reviewers(run.cycle, author, groupReviewerCount(groupTasks, cfg.planReviewQuorum), `${run.id}:plan-group:${group.id}:${round}`);
516
+ for (const reviewer of groupPanel) {
517
+ const ok = await runCall(`group:${group.id}:${group.taskIds.join(",")}:${reviewer}`, group.taskIds, `計畫群 ${group.id} 審查`, {
518
+ reviewer, step: "plan-review-group", gated: false, subject: `任務群 ${group.id}`, scope: group.id,
519
+ prompt: renderPrompt("plan-review-group", {
520
+ reviewer, author, groupId: group.id,
521
+ files: group.files.join("、") || "(這群的描述沒有點名檔案)",
522
+ tasks: JSON.stringify(groupTasks, null, 2),
523
+ neighbors: neighbors.length ? JSON.stringify(neighbors, null, 2) : "(沒有跨群的直接相依)",
524
+ acceptance: JSON.stringify(layered.acceptance.filter((item) => acceptanceIds.has(item.id)), null, 2),
525
+ evidence: extractPlanEvidence(layered.planMd, group.taskIds),
526
+ replies: repliesForTasks(replies, group.taskIds),
527
+ }),
528
+ output: "plan-review-group.json", archive: `plan-review-${round}-${group.id}-${reviewer}.json`,
529
+ });
530
+ if (!ok)
531
+ return run;
532
+ }
533
+ }
534
+ // 只放這一輪真的審過的群(含沿用的),沒審到的任務才留得住前次 verdict
535
+ const groupVerdicts = calls.flatMap((call) => call.taskIds ? [{ taskIds: call.taskIds, verdict: call.verdict }] : []);
536
+ saveState({ version: 1, reviewed: applyReviewVerdicts(reviewed, layered.tasks, layered.acceptance, layered.planMd, groupVerdicts) });
537
+ return concludePlanReview(run, cfg, round, calls);
332
538
  }
333
539
  async function planFixStage(run) {
334
540
  const cfg = loadRepoConfig();
@@ -340,24 +546,35 @@ async function planFixStage(run) {
340
546
  });
341
547
  info(run, `✏️ 依 ${run.planReviewer ?? "審查者"} 的意見修改計畫(${agent})`);
342
548
  const feedback = readFeedback(run);
343
- const snap = snapshotPlan(run);
344
- const outcome = await agentStep(run, agent, "plan-fix", renderPrompt("plan-fix", { requirement: run.requirement, testPattern: cfg.testPattern }), { kind: "write", reset: async () => { await discardChanges(worktreeDir(run.id)); restorePlan(run, snap); } });
549
+ const snap = snapshotPlan(run, PLAN_REPLY_FILES);
550
+ // 回應每輪整份覆寫:先刪掉上一輪的,agent 沒寫時下一輪才不會讀到舊回應;失敗時快照會還原
551
+ const clearReplies = () => rmSync(flowFile(run, "plan-replies.md"), { force: true });
552
+ clearReplies();
553
+ let outcome;
554
+ try {
555
+ outcome = await agentStep(run, agent, "plan-fix", renderPrompt("plan-fix", { requirement: run.requirement, testPattern: cfg.testPattern }), { kind: "write", reset: async () => { await discardChanges(worktreeDir(run.id)); restorePlan(run, snap); clearReplies(); } });
556
+ }
557
+ catch (err) {
558
+ // 所有 agent 額度都用完而暫停:還原成進入時的內容,resume 前的計畫審查回應仍在
559
+ restorePlan(run, snap);
560
+ throw err;
561
+ }
345
562
  const { r, agent: actual } = outcome;
346
563
  await discardChanges(worktreeDir(run.id)); // 只允許改 .flow/
347
564
  if (!r.ok) {
348
565
  restorePlan(run, snap);
349
- return retry(run, "plan-fix", `${feedback}\n\n(上次修改時 Agent 執行失敗:${r.summary})`, "plan_fix");
566
+ return retry(run, "plan-fix", `${feedback}\n\n(上次修改時 Agent 執行失敗:${r.summary})`, "plan_fix", "agent_error");
350
567
  }
351
568
  const ordered = validatePlan(run);
352
569
  if (typeof ordered === "string") {
353
570
  restorePlan(run, snap);
354
- return retry(run, "plan-fix", `${feedback}\n\n另外,修改後的計畫沒有通過格式檢查,已還原:\n${ordered}`, "plan_fix");
571
+ return retry(run, "plan-fix", `${feedback}\n\n另外,修改後的計畫沒有通過格式檢查,已還原:\n${ordered}`, "plan_fix", "format_invalid");
355
572
  }
356
573
  const handoffError = finishHandoff(run, outcome, "writer");
357
574
  if (handoffError) {
358
575
  restorePlan(run, snap);
359
576
  // 計畫已還原,要保留原本的審查意見,否則下一次修正不知道要改什麼
360
- return retry(run, "plan-fix", `${feedback}\n\n另外,交接回覆不合格,本次修改已還原:${handoffError}`, "plan_fix");
577
+ return retry(run, "plan-fix", `${feedback}\n\n另外,交接回覆不合格,本次修改已還原:${handoffError}`, "plan_fix", "handoff_invalid");
361
578
  }
362
579
  acceptPlan(run, ordered);
363
580
  writeFileSync(flowFile(run, "feedback.md"), feedback); // 保留審查意見,讓下一輪審查者知道上次提了什麼
@@ -377,7 +594,7 @@ async function arbitratePlan(run) {
377
594
  const verdicts = [];
378
595
  for (const [slot, arbiter] of panel.entries()) {
379
596
  info(run, `⚖️ ${mode}(${arbiter})`);
380
- const snap = snapshotPlan(run);
597
+ const snap = snapshotPlan(run, PLAN_REPLY_FILES);
381
598
  rmSync(flowFile(run, "plan-arbiter.json"), { force: true });
382
599
  const outcome = await agentStep(run, arbiter, "plan-arbiter", renderPrompt("plan-arbiter", { requirement: run.requirement }), {
383
600
  kind: "review", slot, blind: true,
@@ -391,8 +608,14 @@ async function arbitratePlan(run) {
391
608
  if (!result?.ok || handoffError) {
392
609
  const outputError = !r.ok ? `Agent 執行失敗:${r.summary}` : !result ? "未產生有效裁決" : result.ok ? "未產生有效裁決" : result.error;
393
610
  const reason = handoffError ?? outputError;
394
- info(run, ` ⏸️ ${arbiter} 未產生有效裁決:${reason}`);
395
- return { ...run, stage: "paused", pausedStage: "plan_review", pauseReason: `${arbiter} 仲裁未產生有效裁決:${reason}` };
611
+ const category = handoffError ? "handoff_invalid" : !r.ok ? "agent_error" : "format_invalid";
612
+ info(run, ` ✗ ${arbiter} 未產生有效裁決:${reason}`);
613
+ // 無效的裁決移到 worktree 外保存,供事後查看;plan-arbitration.json 還在,重試時直接回到仲裁
614
+ if (existsSync(flowFile(run, "plan-arbiter.json"))) {
615
+ mkdirSync(join(runDir(run.id), "reviews"), { recursive: true });
616
+ renameSync(flowFile(run, "plan-arbiter.json"), join(runDir(run.id), "reviews", `plan-arbiter-${arbitrationRound}-${arbiter}-invalid.json`));
617
+ }
618
+ return retry(run, "plan-arbitration-run", `上次仲裁未產生有效裁決:${reason}`, "plan_review", category);
396
619
  }
397
620
  mkdirSync(join(runDir(run.id), "reviews"), { recursive: true });
398
621
  renameSync(flowFile(run, "plan-arbiter.json"), join(runDir(run.id), "reviews", `plan-arbiter-${arbitrationRound}-${arbiter}.json`));
@@ -403,6 +626,9 @@ async function arbitratePlan(run) {
403
626
  verdicts.push({ arbiter, verdict, notes, markdownNotes });
404
627
  }
405
628
  rmSync(flowFile(run, "dispute.md"), { force: true });
629
+ rmSync(planArbitrationPath(run.id), { force: true });
630
+ run = { ...run, attempts: { ...run.attempts } };
631
+ delete run.attempts["plan-arbitration-run"];
406
632
  const approvals = verdicts.filter((v) => v.verdict === "approve").length;
407
633
  const unanimous = approvals === verdicts.length;
408
634
  const decision = arbitrationDecision(verdicts.map((v) => v.verdict), cfg.tieBreak);
@@ -418,6 +644,8 @@ async function arbitratePlan(run) {
418
644
  const attempts = { ...run.attempts, "plan-arbitration": arbitrationRound };
419
645
  delete attempts["plan-review"];
420
646
  rmSync(flowFile(run, "plan-review-last.txt"), { force: true });
647
+ rmSync(planReviewStatePath(run.id), { force: true });
648
+ addRetry(run.id, { key: "plan-arbitration", backTo: "plan_fix", category: "arbitration_revise", attempt: arbitrationRound, final: false }, run.updatedAt);
421
649
  writeFileSync(flowFile(run, "feedback.md"), `# 仲裁要求修訂(第 ${arbitrationRound} 次)\n\n${summary}\n\n${feedbackRecord}\n`);
422
650
  info(run, ` → ${summary}`);
423
651
  return { ...run, attempts, stage: "plan_fix" };
@@ -427,7 +655,7 @@ async function arbitratePlan(run) {
427
655
  info(run, ` → ${summary}`);
428
656
  return planSettled(run, "plan-review");
429
657
  }
430
- return { ...run, stage: "failed", failedStage: "plan_review", failureReason: `${summary},需要人工決定(見 plan.md 的仲裁紀錄)` };
658
+ return { ...run, stage: "failed", failedStage: "plan_review", failureCategory: "arbitration_stop", failureReason: `${summary},需要人工決定(見 plan.md 的仲裁紀錄)` };
431
659
  }
432
660
  function planTamperedMessage(files) {
433
661
  return `計畫定案後不可修改規格與計畫檔,已還原你的變更:${files.map((f) => `.flow/${f}`).join(", ")}。若認為規格或驗收條件有誤,請寫進 .flow/handoff-response.json 的 newIssues。`;
@@ -456,6 +684,13 @@ async function implementStage(run) {
456
684
  if (run.taskPhase === "fix")
457
685
  return taskFixStep(run, task, progress);
458
686
  const agents = taskAgents(run.cycle, run.taskIndex, cfg.tddSplit, run.id);
687
+ const framework = hasTestFramework();
688
+ const tdd = taskUsesTdd(task, framework);
689
+ // ── 不走 TDD:略過紅燈,實作前的 HEAD 就是這個任務的起點 ──
690
+ if (!tdd && run.taskPhase === "tests") {
691
+ info(run, `⏭️ [${progress}] 略過 TDD(${framework ? "planner 標記不適合先寫測試" : "專案沒有測試框架"})`);
692
+ return { ...run, taskPhase: "code", taskBase: await headCommit(repo), testsCommit: undefined, lastTestsAuthor: undefined };
693
+ }
459
694
  // ── 紅燈:只寫測試,而且測試必須失敗 ──
460
695
  if (run.taskPhase === "tests") {
461
696
  const key = `${task.id}:tests`;
@@ -467,29 +702,29 @@ async function implementStage(run) {
467
702
  const tampered = restorePlan(run, snap);
468
703
  if (!r.ok) {
469
704
  await resetTo(repo, before);
470
- return retry(run, key, `Agent 執行失敗:${r.summary}`, "implement");
705
+ return retry(run, key, `Agent 執行失敗:${r.summary}`, "implement", "agent_error");
471
706
  }
472
707
  if (tampered.length) {
473
708
  await resetTo(repo, before);
474
- return retry(run, key, planTamperedMessage(tampered), "implement");
709
+ return retry(run, key, planTamperedMessage(tampered), "implement", "plan_tampered");
475
710
  }
476
711
  const commit = await commitAll(repo, `test(${task.id}): ${task.title} [${testsAuthor}]`);
477
712
  if (!commit)
478
- return retry(run, key, "沒有任何檔案變更,這個階段必須撰寫測試。", "implement");
713
+ return retry(run, key, "沒有任何檔案變更,這個階段必須撰寫測試。", "implement", "tests_not_written");
479
714
  const changed = await changedFiles(repo, before, commit);
480
715
  if (!changed.some((f) => testRe.test(f))) {
481
716
  await resetTo(repo, before);
482
- return retry(run, key, `沒有新增或修改任何符合 /${cfg.testPattern}/ 的測試檔。`, "implement");
717
+ return retry(run, key, `沒有新增或修改任何符合 /${cfg.testPattern}/ 的測試檔。`, "implement", "tests_not_written");
483
718
  }
484
719
  const red = await runCommand(target(run, `${task.id}-red`, CMD_AGENT), testCmd);
485
720
  if (red.ok) {
486
721
  await resetTo(repo, before);
487
- return retry(run, key, "測試在功能尚未實作前就全部通過,代表測試沒有驗證到新行為。請撰寫會因功能尚未實作而失敗的測試。", "implement");
722
+ return retry(run, key, "測試在功能尚未實作前就全部通過,代表測試沒有驗證到新行為。請撰寫會因功能尚未實作而失敗的測試。", "implement", "tests_not_red");
488
723
  }
489
724
  const handoffError = finishHandoff(run, outcome, "writer");
490
725
  if (handoffError) {
491
726
  await resetTo(repo, before);
492
- return retry(run, key, handoffError, "implement");
727
+ return retry(run, key, handoffError, "implement", "handoff_invalid");
493
728
  }
494
729
  writeFileSync(flowFile(run, "red-output.txt"), red.output);
495
730
  info(run, `🔴 [${progress}] 測試如預期失敗`);
@@ -497,38 +732,44 @@ async function implementStage(run) {
497
732
  }
498
733
  // ── 綠燈:實作到測試通過,而且不可動測試 ──
499
734
  const key = `${task.id}:code`;
500
- const testsCommit = run.testsCommit;
735
+ // 走 TDD 時以測試 commit 為基準;不走 TDD 時以任務起點為基準
736
+ const testsCommit = tdd ? run.testsCommit : run.taskBase;
501
737
  if (!testsCommit)
502
- throw new Error("缺少 testsCommit,狀態不一致");
503
- info(run, `🛠️ [${progress}] 實作(${agents.code},測試由 ${run.lastTestsAuthor ?? agents.tests} 撰寫)`);
738
+ throw new Error(tdd ? "缺少 testsCommit,狀態不一致" : "缺少 taskBase,狀態不一致");
739
+ info(run, tdd ? `🛠️ [${progress}] 實作(${agents.code},測試由 ${run.lastTestsAuthor ?? agents.tests} 撰寫)` : `🛠️ [${progress}] 實作(${agents.code},不走 TDD)`);
504
740
  const redOutput = existsSync(flowFile(run, "red-output.txt")) ? readFileSync(flowFile(run, "red-output.txt"), "utf8") : "";
505
741
  const snap = snapshotPlan(run, LOCKED_FILES);
506
- const outcome = await agentStep(run, agents.code, `${task.id}-code`, renderPrompt("implement-code", { task: taskJson, acceptance: acceptanceJson, testCmd, redOutput: tail(redOutput, 3000) }), { kind: "write", reset: async () => { await resetTo(repo, testsCommit); restorePlan(run, snap); } });
742
+ const outcome = await agentStep(run, agents.code, `${task.id}-code`, tdd
743
+ ? renderPrompt("implement-code", { task: taskJson, acceptance: acceptanceJson, testCmd, redOutput: tail(redOutput, 3000) })
744
+ : renderPrompt("implement-direct", { task: taskJson, acceptance: acceptanceJson, testCmd: framework ? testCmd : "" }), { kind: "write", reset: async () => { await resetTo(repo, testsCommit); restorePlan(run, snap); } });
507
745
  const { r, agent: codeAuthor } = outcome;
508
746
  const tampered = restorePlan(run, snap);
509
747
  if (!r.ok)
510
- return retry(run, key, `Agent 執行失敗:${r.summary}`, "implement");
748
+ return retry(run, key, `Agent 執行失敗:${r.summary}`, "implement", "agent_error");
511
749
  if (tampered.length) {
512
750
  await resetTo(repo, testsCommit);
513
- return retry(run, key, planTamperedMessage(tampered), "implement");
751
+ return retry(run, key, planTamperedMessage(tampered), "implement", "plan_tampered");
514
752
  }
515
- await commitAll(repo, `feat(${task.id}): ${task.title} [${codeAuthor}]`);
516
- const touched = (await changedFiles(repo, testsCommit, await headCommit(repo))).filter((f) => testRe.test(f));
753
+ const codeCommit = await commitAll(repo, `feat(${task.id}): ${task.title} [${codeAuthor}]`);
754
+ if (!tdd && !codeCommit)
755
+ return retry(run, key, "沒有任何檔案變更,這個任務必須完成實作。", "implement", "code_not_written");
756
+ const touched = tdd ? (await changedFiles(repo, testsCommit, await headCommit(repo))).filter((f) => testRe.test(f)) : [];
517
757
  if (touched.length) {
518
758
  await resetTo(repo, testsCommit);
519
- return retry(run, key, `實作階段不可修改測試檔,已還原你的變更:${touched.join(", ")}`, "implement");
759
+ return retry(run, key, `實作階段不可修改測試檔,已還原你的變更:${touched.join(", ")}`, "implement", "tests_modified");
520
760
  }
521
- const green = await runCommand(target(run, `${task.id}-green`, CMD_AGENT), testCmd);
761
+ // 沒有測試框架時沒有東西可跑,後面的驗證與審查照常把關
762
+ const green = framework ? await runCommand(target(run, `${task.id}-green`, CMD_AGENT), testCmd) : { ok: true, output: "", seq: 0 };
522
763
  if (!green.ok) {
523
764
  info(run, ` ✗ 測試仍未通過${logHint(run, green.seq)}`);
524
- return retry(run, key, `測試仍未通過:\n\n\`\`\`\n${tail(green.output)}\n\`\`\``, "implement");
765
+ return retry(run, key, `測試仍未通過:\n\n\`\`\`\n${tail(green.output)}\n\`\`\``, "implement", "tests_not_green");
525
766
  }
526
767
  const handoffError = finishHandoff(run, outcome, "writer");
527
768
  if (handoffError) {
528
769
  await resetTo(repo, testsCommit);
529
- return retry(run, key, handoffError, "implement");
770
+ return retry(run, key, handoffError, "implement", "handoff_invalid");
530
771
  }
531
- info(run, `🟢 [${progress}] 測試通過`);
772
+ info(run, framework ? `🟢 [${progress}] 測試通過` : `✔️ [${progress}] 實作完成`);
532
773
  return { ...succeed(run, key, "implement"), taskPhase: "review", lastWriter: codeAuthor };
533
774
  }
534
775
  // ── 任務審查:只看這個任務的變更與驗收條件 ──
@@ -558,7 +799,7 @@ async function taskReviewStep(run, task, progress, taskJson, acceptanceJson) {
558
799
  if (!result.objector)
559
800
  return { ...succeed(reviewed, key, "implement"), taskPhase: "verify" };
560
801
  return {
561
- ...retry(reviewed, key, `任務審查要求修改:\n\n${result.issues.join("\n\n")}`, "implement"),
802
+ ...retry(reviewed, key, `任務審查要求修改:\n\n${result.issues.join("\n\n")}`, "implement", "review_changes"),
562
803
  taskPhase: "fix",
563
804
  fixSource: "review",
564
805
  lastReviewer: result.objector,
@@ -570,7 +811,7 @@ async function taskVerifyStep(run, task, progress) {
570
811
  info(run, `🔍 [${progress}] 執行驗證`);
571
812
  const report = await runChecks(run, `${task.id}-`);
572
813
  if (report)
573
- return { ...retry(run, key, report, "implement"), taskPhase: "fix", fixSource: "verify" };
814
+ return { ...retry(run, key, report, "implement", "checks_failed"), taskPhase: "fix", fixSource: "verify" };
574
815
  info(run, `✅ [${progress}] 完成`);
575
816
  return {
576
817
  ...succeed(run, key, "implement"),
@@ -622,7 +863,7 @@ async function verifyStage(run) {
622
863
  const report = await runChecks(run);
623
864
  if (!report)
624
865
  return succeed(run, "verify", "review");
625
- return { ...retry(run, "verify", report, "fix"), fixSource: "verify" };
866
+ return { ...retry(run, "verify", report, "fix", "checks_failed"), fixSource: "verify" };
626
867
  }
627
868
  /** 修正驗證錯誤或審查意見;成功時回傳實際修正者,未通過時回傳重試後的 run */
628
869
  async function applyFix(run, opts) {
@@ -647,24 +888,24 @@ async function applyFix(run, opts) {
647
888
  });
648
889
  const { r, agent: actual } = outcome;
649
890
  const tampered = restorePlan(run, snap);
650
- const again = (reason) => ({ run: retry(run, opts.key, reason, opts.backTo) });
891
+ const again = (reason, category) => ({ run: retry(run, opts.key, reason, opts.backTo, category) });
651
892
  if (!r.ok)
652
- return again(`${feedback}\n\n(上次修正時 Agent 執行失敗:${r.summary})`);
893
+ return again(`${feedback}\n\n(上次修正時 Agent 執行失敗:${r.summary})`, "agent_error");
653
894
  if (tampered.length) {
654
895
  await resetTo(repo, before);
655
- return again(`${feedback}\n\n另外:${planTamperedMessage(tampered)}`);
896
+ return again(`${feedback}\n\n另外:${planTamperedMessage(tampered)}`, "plan_tampered");
656
897
  }
657
898
  await commitAll(repo, `${opts.commitScope}: ${why} [${actual}]`);
658
899
  const testRe = new RegExp(cfg.testPattern);
659
900
  const deleted = (await changedFiles(repo, before, await headCommit(repo), "D")).filter((f) => testRe.test(f));
660
901
  if (deleted.length) {
661
902
  await resetTo(repo, before);
662
- return again(`${feedback}\n\n另外:不可刪除測試檔來讓檢查通過,已還原:${deleted.join(", ")}`);
903
+ return again(`${feedback}\n\n另外:不可刪除測試檔來讓檢查通過,已還原:${deleted.join(", ")}`, "tests_deleted");
663
904
  }
664
905
  const handoffError = finishHandoff(run, outcome, "writer");
665
906
  if (handoffError) {
666
907
  await resetTo(repo, before);
667
- return again(`${feedback}\n\n另外,交接回覆不合格,本次修正已還原:${handoffError}`);
908
+ return again(`${feedback}\n\n另外,交接回覆不合格,本次修正已還原:${handoffError}`, "handoff_invalid");
668
909
  }
669
910
  return { agent: actual };
670
911
  }
@@ -709,14 +950,14 @@ async function codeReview(run, opts) {
709
950
  if (tampered.length)
710
951
  info(run, ` ↩️ 已還原審查者修改的檔案:${tampered.join(", ")}`);
711
952
  if (!r.ok)
712
- return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, `Agent 執行失敗:${r.summary}`, opts.backTo) };
953
+ return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, `Agent 執行失敗:${r.summary}`, opts.backTo, "agent_error") };
713
954
  const review = readJsonFile(flowFile(run, "review.json"), ConsistentReviewResult);
714
955
  if (!review.ok)
715
- return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, review.error, opts.backTo) };
956
+ return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, review.error, opts.backTo, "format_invalid") };
716
957
  const gate = opts.gate ? { target: "code", verdict: review.data.verdict } : undefined;
717
958
  const handoffError = finishHandoff(run, outcome, "reviewer", gate);
718
959
  if (handoffError)
719
- return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, handoffError, opts.backTo) };
960
+ return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, handoffError, opts.backTo, "handoff_invalid") };
720
961
  if (opts.step === "review")
721
962
  run = clearModelReviewFailure(run, "review", reviewer);
722
963
  renameSync(flowFile(run, "review.json"), flowFile(run, opts.saveAs(reviewer)));
@@ -753,7 +994,7 @@ async function reviewStage(run) {
753
994
  if (!result.objector)
754
995
  return succeed(result.state, "review", "pr");
755
996
  return {
756
- ...retry(result.state, "review", `程式碼審查要求修改:\n\n${result.issues.join("\n\n")}`, "fix"),
997
+ ...retry(result.state, "review", `程式碼審查要求修改:\n\n${result.issues.join("\n\n")}`, "fix", "review_changes"),
757
998
  fixSource: "review",
758
999
  lastReviewer: result.objector,
759
1000
  };
@@ -761,7 +1002,7 @@ async function reviewStage(run) {
761
1002
  async function prStage(run) {
762
1003
  const pending = openActions(readHandoff(run.id));
763
1004
  if (pending.length)
764
- return { ...run, stage: "failed", failedStage: "pr", failureReason: `仍有未結交接事項:${pending.map((item) => item.id).join("、")}` };
1005
+ return { ...run, stage: "failed", failedStage: "pr", failureCategory: "open_handoff", failureReason: `仍有未結交接事項:${pending.map((item) => item.id).join("、")}` };
765
1006
  const repo = worktreeDir(run.id);
766
1007
  const remotes = (await git(repo, "remote")).split("\n").filter(Boolean);
767
1008
  if (!remotes.includes("origin")) {
@@ -806,6 +1047,7 @@ export async function advance(initial) {
806
1047
  ...run,
807
1048
  stage: "failed",
808
1049
  failedStage: stage,
1050
+ failureCategory: "agent_budget",
809
1051
  failureReason: `已執行 agent ${runs} 次,達到上限 ${run.maxAgentRuns}(可用 resume --max-agent-runs 調高)`,
810
1052
  });
811
1053
  }
@@ -817,7 +1059,7 @@ export async function advance(initial) {
817
1059
  info(run, `⏸️ 暫停:${err.message}`);
818
1060
  return saveRun({ ...run, stage: "paused", pausedStage: stage, pauseReason: err.message });
819
1061
  }
820
- return saveRun({ ...run, stage: "failed", failedStage: stage, failureReason: err.message });
1062
+ return saveRun({ ...run, stage: "failed", failedStage: stage, failureCategory: "error", failureReason: err.message });
821
1063
  }
822
1064
  }
823
1065
  }