agentflowctl 0.13.1 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -9
- package/dist/cli.js +126 -5
- package/dist/config.js +5 -3
- package/dist/detect.js +21 -4
- package/dist/engine.js +347 -105
- package/dist/insights.js +86 -0
- package/dist/modelSelection.js +19 -11
- package/dist/paths.js +4 -0
- package/dist/planReview.js +333 -0
- package/dist/schemas.js +18 -0
- package/dist/stats.js +26 -0
- package/dist/store.js +62 -17
- package/dist/usageInsights.js +156 -0
- package/dist/util.js +12 -0
- package/examples/flow.config.json +1 -0
- package/package.json +1 -1
- package/prompts/implement-direct.md +67 -0
- package/prompts/plan-arbiter.md +6 -1
- package/prompts/plan-fix.md +5 -2
- package/prompts/plan-review-group.md +92 -0
- package/prompts/plan-review-index.md +93 -0
- package/prompts/plan-review.md +2 -1
- package/prompts/plan.md +6 -4
package/dist/engine.js
CHANGED
|
@@ -3,18 +3,19 @@ import { join } from "node:path";
|
|
|
3
3
|
import { z } from "zod";
|
|
4
4
|
import { config } from "./config.js";
|
|
5
5
|
import { arbitrationDecision } from "./arbitration.js";
|
|
6
|
-
import { detectProjectDefaults, withProjectDefaults } from "./detect.js";
|
|
6
|
+
import { detectProjectDefaults, usesTestFramework, withProjectDefaults } from "./detect.js";
|
|
7
7
|
import { escapeXml, opinion, reviewIssue } from "./feedback.js";
|
|
8
8
|
import { changedFiles, commitAll, discardChanges, git, headCommit, resetTo } from "./git.js";
|
|
9
9
|
import { acceptHandoff, openActions, prepareHandoff, previewHandoff, readHandoff, recoverHandoff, reviewHandoffGate, validateHandoffResponse } from "./handoff.js";
|
|
10
|
-
import { flowDir, logDir, projectRoot, runDir, worktreeDir } from "./paths.js";
|
|
10
|
+
import { flowDir, logDir, planArbitrationPath, planReviewStatePath, projectRoot, runDir, worktreeDir } from "./paths.js";
|
|
11
11
|
import { CMD_AGENT, nextLogFile } from "./logs.js";
|
|
12
12
|
import { exec } from "./proc.js";
|
|
13
13
|
import { arbiterPanel, availableAgent, fixAgent, planAgent, planFixAgent, reviewers, specAgent, taskAgents } from "./roles.js";
|
|
14
14
|
import { resolveAgent, runAgent, runCommand } from "./runner.js";
|
|
15
15
|
import { AcceptanceList, ArbiterResult, ConsistentReviewResult, RepoConfig, TaskList, } from "./schemas.js";
|
|
16
|
-
import { addSubstitution, addUsage, agentRuns, saveRun } from "./store.js";
|
|
16
|
+
import { addRetry, addSubstitution, addUsage, agentRuns, saveRun } from "./store.js";
|
|
17
17
|
import { clearModelReviewFailure, clearModelReviewStage, recordModelReviewFailure, selectModel } from "./modelSelection.js";
|
|
18
|
+
import { applyReviewVerdicts, dirtyGroups, dirtyTaskIds, extractPlanEvidence, groupReviewerCount, layeredReview, neighborTasks, planContentKey, planOverview, planReviewIndex, readPendingArbitration, readPlanReviewState, repliesForTasks, reviewFingerprint, roundProgress, } from "./planReview.js";
|
|
18
19
|
import { orderTasks, taskAcceptance, validateTaskComplexity } from "./tasks.js";
|
|
19
20
|
import { readJsonFile, renderPrompt, tail } from "./util.js";
|
|
20
21
|
// ───────────────────────── 共用工具 ─────────────────────────
|
|
@@ -33,6 +34,10 @@ export class QuotaPause extends Error {
|
|
|
33
34
|
}
|
|
34
35
|
/** 這次執行中已確認額度用完的 agent;程序結束即清空,resume 時會重新嘗試 */
|
|
35
36
|
const exhausted = new Set();
|
|
37
|
+
/** 只供測試使用:清掉這個程序內記下的額度用完 agent,模擬新啟動的程序 */
|
|
38
|
+
export function resetQuotaState() {
|
|
39
|
+
exhausted.clear();
|
|
40
|
+
}
|
|
36
41
|
function handoffTarget(run) {
|
|
37
42
|
return ["spec", "plan", "plan_review", "plan_fix"].includes(run.stage) ? "plan" : "code";
|
|
38
43
|
}
|
|
@@ -61,7 +66,7 @@ async function agentStep(run, planned, step, prompt, mode) {
|
|
|
61
66
|
info(run, `🔁 ${agent} 額度已用完,${step} 由 ${sub} 代打${note ? `(注意:${note})` : ""}`);
|
|
62
67
|
agent = sub;
|
|
63
68
|
}
|
|
64
|
-
const selected = selectModel(run, cfg, agent, step, /^T-\d+-/.test(step) ? loadOrderedTasks(run)[run.taskIndex]?.complexity : undefined, mode.kind === "review" ? agent : undefined);
|
|
69
|
+
const selected = selectModel(run, cfg, agent, step, /^T-\d+-/.test(step) ? loadOrderedTasks(run)[run.taskIndex]?.complexity : undefined, mode.kind === "review" ? agent : undefined, mode.modelScope);
|
|
65
70
|
info(run, `🤖 ${step}:${agent} 使用 ${selected.name ?? "CLI 預設(名稱未知)"}${selected.insufficient ? `(低於目標 ${selected.targetStrength})` : ""}`);
|
|
66
71
|
const callKey = handoffKey(run, step, mode.slot ?? 0, agent);
|
|
67
72
|
prepareHandoff(run.id, callKey, handoffTarget(run), mode.blind ?? false);
|
|
@@ -112,20 +117,32 @@ function reportMeta(run, agent, r) {
|
|
|
112
117
|
if (r.meta.concerns)
|
|
113
118
|
info(run, ` 💭 ${agent} 的疑慮:${r.meta.concerns}`);
|
|
114
119
|
}
|
|
115
|
-
|
|
116
|
-
|
|
120
|
+
/** 讀 .flow/ 下的文字檔,沒有檔就當空字串 */
|
|
121
|
+
function flowText(run, name) {
|
|
122
|
+
const p = flowFile(run, name);
|
|
117
123
|
return existsSync(p) ? readFileSync(p, "utf8") : "";
|
|
118
124
|
}
|
|
119
|
-
|
|
120
|
-
|
|
125
|
+
function readFeedback(run) {
|
|
126
|
+
return flowText(run, "feedback.md");
|
|
127
|
+
}
|
|
128
|
+
/** 計畫審查第幾輪仍有人要求修改時交付仲裁;固定值,不受重試上限影響 */
|
|
129
|
+
const PLAN_ARBITRATION_ROUND = 2;
|
|
130
|
+
/** 這個 run 的重試上限:`--max-attempts` 存在 run 裡,沒設定才用環境變數 */
|
|
131
|
+
function attemptLimit(run) {
|
|
132
|
+
return run.maxAttempts ?? config.maxAttempts;
|
|
133
|
+
}
|
|
134
|
+
/** 關卡未通過:寫入 feedback.md 給下一次嘗試參考,並把原因分類記進 retries.jsonl;超過上限就讓整個 run 失敗 */
|
|
135
|
+
function retry(run, key, reason, backTo, category) {
|
|
121
136
|
const n = (run.attempts[key] ?? 0) + 1;
|
|
122
137
|
const attempts = { ...run.attempts, [key]: n };
|
|
138
|
+
const final = n >= attemptLimit(run);
|
|
123
139
|
mkdirSync(flowDir(run.id), { recursive: true });
|
|
124
140
|
writeFileSync(flowFile(run, "feedback.md"), `# 前次嘗試未通過(第 ${n} 次)\n\n${reason}\n`);
|
|
125
|
-
|
|
126
|
-
|
|
141
|
+
addRetry(run.id, { key, backTo, category, attempt: n, final }, run.updatedAt);
|
|
142
|
+
if (final) {
|
|
143
|
+
return { ...run, attempts, stage: "failed", failedStage: backTo, failureCategory: "retry_limit", failureReason: `${key} 連續失敗 ${n} 次:${tail(reason, 500)}` };
|
|
127
144
|
}
|
|
128
|
-
info(run, `⚠️ ${key} 未通過,重試(${n}/${
|
|
145
|
+
info(run, `⚠️ ${key} 未通過,重試(${n}/${attemptLimit(run)})`);
|
|
129
146
|
return { ...run, attempts, stage: backTo };
|
|
130
147
|
}
|
|
131
148
|
function succeed(run, key, next) {
|
|
@@ -147,6 +164,17 @@ export function loadRepoConfig() {
|
|
|
147
164
|
throw new Error(r.error);
|
|
148
165
|
return r.data;
|
|
149
166
|
}
|
|
167
|
+
/** 專案有測試框架(偵測到或手動設定 test) */
|
|
168
|
+
function hasTestFramework() {
|
|
169
|
+
const root = projectRoot();
|
|
170
|
+
const p = join(root, "flow.config.json");
|
|
171
|
+
const r = existsSync(p) ? readJsonFile(p, z.unknown()) : undefined;
|
|
172
|
+
return usesTestFramework(r?.ok ? r.data : {}, detectProjectDefaults(root));
|
|
173
|
+
}
|
|
174
|
+
/** 這個任務要不要走紅綠燈:沒有測試框架一律不走,其餘依 planner 標記(沒標視為要走) */
|
|
175
|
+
function taskUsesTdd(task, framework) {
|
|
176
|
+
return framework && task.tdd !== false;
|
|
177
|
+
}
|
|
150
178
|
function loadOrderedTasks(run) {
|
|
151
179
|
const r = readJsonFile(flowFile(run, "tasks.ordered.json"), TaskList);
|
|
152
180
|
if (!r.ok)
|
|
@@ -161,24 +189,26 @@ async function specStage(run) {
|
|
|
161
189
|
const { r } = outcome;
|
|
162
190
|
await discardChanges(worktreeDir(run.id)); // 這個階段只允許寫 .flow/
|
|
163
191
|
if (!r.ok)
|
|
164
|
-
return retry(run, "spec", `Agent 執行失敗:${r.summary}`, "spec");
|
|
192
|
+
return retry(run, "spec", `Agent 執行失敗:${r.summary}`, "spec", "agent_error");
|
|
165
193
|
if (!existsSync(flowFile(run, "spec.md")))
|
|
166
|
-
return retry(run, "spec", "缺少 .flow/spec.md", "spec");
|
|
194
|
+
return retry(run, "spec", "缺少 .flow/spec.md", "spec", "missing_artifact");
|
|
167
195
|
const ac = readJsonFile(flowFile(run, "acceptance.json"), AcceptanceList);
|
|
168
196
|
if (!ac.ok)
|
|
169
|
-
return retry(run, "spec", ac.error, "spec");
|
|
197
|
+
return retry(run, "spec", ac.error, "spec", "format_invalid");
|
|
170
198
|
const ids = ac.data.map((a) => a.id);
|
|
171
199
|
if (new Set(ids).size !== ids.length)
|
|
172
|
-
return retry(run, "spec", "驗收條件 id 有重複", "spec");
|
|
200
|
+
return retry(run, "spec", "驗收條件 id 有重複", "spec", "format_invalid");
|
|
173
201
|
const handoffError = finishHandoff(run, outcome, "writer");
|
|
174
202
|
if (handoffError)
|
|
175
|
-
return retry(run, "spec", handoffError, "spec");
|
|
203
|
+
return retry(run, "spec", handoffError, "spec", "handoff_invalid");
|
|
176
204
|
return succeed(run, "spec", "plan");
|
|
177
205
|
}
|
|
178
206
|
// ── 規格與計畫檔案:計畫審查、仲裁與計畫定案後的所有階段只能讀,不能改 ──
|
|
179
207
|
const PLAN_FILES = ["spec.md", "acceptance.json", "plan.md", "tasks.json"];
|
|
180
208
|
/** 計畫定案後(實作、修正、程式碼審查)另外依賴排好的任務順序,同樣不能被改 */
|
|
181
209
|
const LOCKED_FILES = [...PLAN_FILES, "tasks.ordered.json"];
|
|
210
|
+
/** 審查、修訂與仲裁的快照另外包含審查回應;不要併進 PLAN_FILES,定案後的階段不依賴它 */
|
|
211
|
+
const PLAN_REPLY_FILES = [...PLAN_FILES, "plan-replies.md"];
|
|
182
212
|
function snapshotPlan(run, files = PLAN_FILES) {
|
|
183
213
|
return Object.fromEntries(files.map((f) => [f, existsSync(flowFile(run, f)) ? readFileSync(flowFile(run, f), "utf8") : undefined]));
|
|
184
214
|
}
|
|
@@ -223,16 +253,19 @@ function acceptPlan(run, ordered) {
|
|
|
223
253
|
function announceTasks(run, ordered) {
|
|
224
254
|
const cfg = loadRepoConfig();
|
|
225
255
|
info(run, `📋 共 ${ordered.length} 個任務:${ordered.map((t) => t.id).join(" → ")}`);
|
|
256
|
+
const framework = hasTestFramework();
|
|
226
257
|
ordered.forEach((t, i) => {
|
|
227
258
|
const a = taskAgents(run.cycle, i, cfg.tddSplit, run.id);
|
|
228
|
-
info(run,
|
|
259
|
+
info(run, taskUsesTdd(t, framework)
|
|
260
|
+
? ` ${t.id} 測試:${a.tests} 實作:${a.code} 審查:${a.review}`
|
|
261
|
+
: ` ${t.id} 略過 TDD 實作:${a.code} 審查:${a.review}`);
|
|
229
262
|
});
|
|
230
263
|
}
|
|
231
264
|
/** 計畫定案後:預設直接開始實作;--manual-plan 時才停下來等人 */
|
|
232
265
|
function planSettled(run, key) {
|
|
233
266
|
const pending = openActions(readHandoff(run.id), "plan");
|
|
234
267
|
if (pending.length)
|
|
235
|
-
return retry(run, "plan-handoff", `計畫仍有未結交接事項:${pending.map((item) => item.id).join("、")}`, "plan_fix");
|
|
268
|
+
return retry(run, "plan-handoff", `計畫仍有未結交接事項:${pending.map((item) => item.id).join("、")}`, "plan_fix", "open_handoff");
|
|
236
269
|
const next = run.autopilot ? "implement" : "awaiting_approval";
|
|
237
270
|
const ordered = loadOrderedTasks(run);
|
|
238
271
|
announceTasks(run, ordered);
|
|
@@ -252,59 +285,75 @@ async function planStage(run) {
|
|
|
252
285
|
const { r, agent: actual } = outcome;
|
|
253
286
|
await discardChanges(worktreeDir(run.id));
|
|
254
287
|
if (!r.ok)
|
|
255
|
-
return retry(run, "plan", `Agent 執行失敗:${r.summary}`, "plan");
|
|
288
|
+
return retry(run, "plan", `Agent 執行失敗:${r.summary}`, "plan", "agent_error");
|
|
256
289
|
const ordered = validatePlan(run);
|
|
257
290
|
if (typeof ordered === "string")
|
|
258
|
-
return retry(run, "plan", ordered, "plan");
|
|
291
|
+
return retry(run, "plan", ordered, "plan", "format_invalid");
|
|
259
292
|
const handoffError = finishHandoff(run, outcome, "writer");
|
|
260
293
|
if (handoffError)
|
|
261
|
-
return retry(run, "plan", handoffError, "plan");
|
|
294
|
+
return retry(run, "plan", handoffError, "plan", "handoff_invalid");
|
|
262
295
|
acceptPlan(run, ordered);
|
|
263
296
|
rmSync(flowFile(run, "plan-review-last.txt"), { force: true });
|
|
264
297
|
return { ...succeed(run, "plan", "plan_review"), planWriter: actual };
|
|
265
298
|
}
|
|
266
299
|
async function planReviewStage(run) {
|
|
267
300
|
const cfg = loadRepoConfig();
|
|
268
|
-
const
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
301
|
+
const pending = pendingArbitration(run, cfg);
|
|
302
|
+
if (pending) {
|
|
303
|
+
info(run, "⚖️ 上次仲裁沒有完成,直接回到仲裁(不重跑計畫審查)");
|
|
304
|
+
return arbitratePlan({ ...run, planReviewer: pending.planReviewer });
|
|
305
|
+
}
|
|
306
|
+
const layered = loadLayeredPlan(run, cfg);
|
|
307
|
+
return layered ? planReviewLayered(run, layered, cfg) : planReviewFull(run, cfg);
|
|
308
|
+
}
|
|
309
|
+
/**
|
|
310
|
+
* 整份審查與分層審查共用的單次呼叫:還原審查者改過的計畫檔、驗證裁決與交接。
|
|
311
|
+
* 失敗時回傳已呼叫 retry 的 run、沒有 collected,呼叫端應直接回傳這個 run。
|
|
312
|
+
*/
|
|
313
|
+
async function collectPlanReview(run, spec) {
|
|
314
|
+
const { reviewer, step } = spec;
|
|
315
|
+
const snap = snapshotPlan(run, PLAN_REPLY_FILES);
|
|
316
|
+
rmSync(flowFile(run, spec.output), { force: true });
|
|
317
|
+
const outcome = await agentStep(run, reviewer, step, spec.prompt, {
|
|
318
|
+
kind: "review", slot: spec.slot, modelScope: spec.scope,
|
|
319
|
+
reset: async () => { await discardChanges(worktreeDir(run.id)); restorePlan(run, snap); },
|
|
320
|
+
});
|
|
321
|
+
await discardChanges(worktreeDir(run.id));
|
|
322
|
+
const tampered = restorePlan(run, snap);
|
|
323
|
+
if (tampered.length)
|
|
324
|
+
info(run, ` ↩️ 已還原審查者修改的檔案:${tampered.join(", ")}`);
|
|
325
|
+
const stop = (category, reason) => ({
|
|
326
|
+
run: retry(recordModelReviewFailure(run, step, reviewer, spec.scope), "plan-review-run", reason, "plan_review", category),
|
|
327
|
+
});
|
|
328
|
+
if (!outcome.r.ok)
|
|
329
|
+
return stop("agent_error", `Agent 執行失敗:${outcome.r.summary}`);
|
|
330
|
+
const review = readJsonFile(flowFile(run, spec.output), ConsistentReviewResult);
|
|
331
|
+
if (!review.ok)
|
|
332
|
+
return stop("format_invalid", review.error);
|
|
333
|
+
// 群審查不帶關卡:索引要求修改並新增事項後,群的核准不算矛盾;最後由 planSettled 檢查未結事項
|
|
334
|
+
const handoffError = finishHandoff(run, outcome, "reviewer", spec.gated ? { target: "plan", verdict: review.data.verdict } : undefined);
|
|
335
|
+
if (handoffError)
|
|
336
|
+
return stop("handoff_invalid", handoffError);
|
|
337
|
+
run = clearModelReviewFailure(run, step, reviewer, spec.scope);
|
|
338
|
+
// 審查紀錄移到 worktree 外面:之後的仲裁者看不到是哪一家提的意見
|
|
339
|
+
mkdirSync(join(runDir(run.id), "reviews"), { recursive: true });
|
|
340
|
+
renameSync(flowFile(run, spec.output), join(runDir(run.id), "reviews", spec.archive));
|
|
341
|
+
if (review.data.verdict === "approve") {
|
|
342
|
+
info(run, ` ✓ ${reviewer} 核准${spec.subject}`);
|
|
343
|
+
return { run, collected: { reviewer, verdict: "approve", issueLines: [] } };
|
|
307
344
|
}
|
|
345
|
+
info(run, ` ✗ ${reviewer} 要求修改${spec.subject}`);
|
|
346
|
+
const issueLines = review.data.items
|
|
347
|
+
.filter((i) => i.status !== "met")
|
|
348
|
+
.map((i) => reviewIssue(i.criterion, i.status, i.note));
|
|
349
|
+
return { run, collected: { reviewer, verdict: "changes_requested", issueLines } };
|
|
350
|
+
}
|
|
351
|
+
/** 一輪審查都產出合法裁決後:全部核准就定案,否則退回修訂,僵持或達輪數上限時交付仲裁 */
|
|
352
|
+
async function concludePlanReview(run, cfg, round, calls) {
|
|
353
|
+
const objections = calls.filter((call) => call.verdict !== "approve");
|
|
354
|
+
const issueLines = objections.flatMap((call) => call.issueLines);
|
|
355
|
+
const issues = objections.map((call) => opinion(call.reviewer, call.issueLines));
|
|
356
|
+
const firstObjector = objections[0]?.reviewer;
|
|
308
357
|
run = clearModelReviewStage(run, "plan-review");
|
|
309
358
|
const attempts = { ...run.attempts };
|
|
310
359
|
delete attempts["plan-review-run"];
|
|
@@ -314,10 +363,11 @@ async function planReviewStage(run) {
|
|
|
314
363
|
const report = `計畫審查要求修改:\n\n${issues.join("\n\n")}`;
|
|
315
364
|
// 僵持偵測:意見和上一輪完全相同,代表修改沒有進展
|
|
316
365
|
const lastPath = flowFile(run, "plan-review-last.txt");
|
|
317
|
-
const fingerprint =
|
|
366
|
+
const fingerprint = reviewFingerprint(issueLines);
|
|
318
367
|
const stalled = existsSync(lastPath) && readFileSync(lastPath, "utf8") === fingerprint;
|
|
319
368
|
writeFileSync(lastPath, fingerprint);
|
|
320
|
-
|
|
369
|
+
// 修訂過一次仍被要求修改就交付仲裁,不等到重試上限
|
|
370
|
+
const exhausted = round >= PLAN_ARBITRATION_ROUND;
|
|
321
371
|
if ((stalled || exhausted) && cfg.planArbiter) {
|
|
322
372
|
// 雙盲:帶有審查者名稱的 feedback.md 不留在 worktree,完整報告另存到 worktree 外
|
|
323
373
|
rmSync(flowFile(run, "feedback.md"), { force: true });
|
|
@@ -326,9 +376,165 @@ async function planReviewStage(run) {
|
|
|
326
376
|
// 給仲裁者的爭議清單不含任何模型名稱
|
|
327
377
|
writeFileSync(flowFile(run, "dispute.md"), `# 尚未解決的審查意見\n\n${[...new Set(issueLines)].join("\n")}\n`);
|
|
328
378
|
info(run, `⚖️ 計畫審查${stalled ? "意見沒有變化" : `已達 ${round} 輪`},交付仲裁`);
|
|
379
|
+
// 仲裁暫停(裁決無效或額度用完)後 resume 仍停在 plan_review,靠這份紀錄直接回到仲裁
|
|
380
|
+
mkdirSync(runDir(run.id), { recursive: true });
|
|
381
|
+
writeFileSync(planArbitrationPath(run.id), JSON.stringify({ planReviewer: firstObjector, planKey: currentPlanKey(run) }, null, 2));
|
|
329
382
|
return arbitratePlan({ ...run, planReviewer: firstObjector });
|
|
330
383
|
}
|
|
331
|
-
return { ...retry(run, "plan-review", report, "plan_fix"), planReviewer: firstObjector };
|
|
384
|
+
return { ...retry(run, "plan-review", report, "plan_fix", "review_changes"), planReviewer: firstObjector };
|
|
385
|
+
}
|
|
386
|
+
/** 整份審查:每位審查者讀完整份規格與計畫 */
|
|
387
|
+
async function planReviewFull(run, cfg) {
|
|
388
|
+
// 整份審查不維護分層狀態;之後若又切回分層,不能沿用這之前的 approve
|
|
389
|
+
rmSync(planReviewStatePath(run.id), { force: true });
|
|
390
|
+
const author = run.planWriter ?? planAgent(run.cycle, run.id);
|
|
391
|
+
const round = (run.attempts["plan-review"] ?? 0) + 1;
|
|
392
|
+
const panel = reviewers(run.cycle, author, cfg.planReviewQuorum, `${run.id}:plan-review:${round}`);
|
|
393
|
+
const calls = [];
|
|
394
|
+
for (const [slot, reviewer] of panel.entries()) {
|
|
395
|
+
info(run, `🧐 計畫審查第 ${round} 輪(${reviewer},作者 ${author})`);
|
|
396
|
+
const passed = await collectPlanReview(run, {
|
|
397
|
+
reviewer, step: "plan-review", slot, gated: true, subject: "計畫",
|
|
398
|
+
prompt: renderPrompt("plan-review", { reviewer, author, requirement: run.requirement }),
|
|
399
|
+
output: "plan-review.json", archive: `plan-review-${round}-${reviewer}.json`,
|
|
400
|
+
});
|
|
401
|
+
run = passed.run;
|
|
402
|
+
if (!passed.collected)
|
|
403
|
+
return run;
|
|
404
|
+
calls.push(passed.collected);
|
|
405
|
+
}
|
|
406
|
+
return concludePlanReview(run, cfg, round, calls);
|
|
407
|
+
}
|
|
408
|
+
/** 本輪進度的雜湊涵蓋的檔案:任一份變了,就不能沿用先前的審查結果 */
|
|
409
|
+
const PLAN_KEY_FILES = ["tasks.json", "acceptance.json", "plan.md", "plan-replies.md"];
|
|
410
|
+
function currentPlanKey(run) {
|
|
411
|
+
return planContentKey(PLAN_KEY_FILES.map((name) => flowText(run, name)));
|
|
412
|
+
}
|
|
413
|
+
/**
|
|
414
|
+
* 上次交付仲裁卻沒有得出裁決(暫停)時的紀錄。計畫在這之間被改過、或爭議清單不見了,
|
|
415
|
+
* 就當成沒有待完成的仲裁並刪掉紀錄,重新審查。暫停期間關掉了 planArbiter 也一樣,
|
|
416
|
+
* 連同爭議清單一起刪掉。
|
|
417
|
+
*/
|
|
418
|
+
function pendingArbitration(run, cfg) {
|
|
419
|
+
const path = planArbitrationPath(run.id);
|
|
420
|
+
if (!existsSync(path))
|
|
421
|
+
return undefined;
|
|
422
|
+
const pending = readPendingArbitration(readFileSync(path, "utf8"));
|
|
423
|
+
if (cfg.planArbiter && pending && pending.planKey === currentPlanKey(run) && existsSync(flowFile(run, "dispute.md")))
|
|
424
|
+
return pending;
|
|
425
|
+
rmSync(path, { force: true });
|
|
426
|
+
if (!cfg.planArbiter)
|
|
427
|
+
rmSync(flowFile(run, "dispute.md"), { force: true });
|
|
428
|
+
return undefined;
|
|
429
|
+
}
|
|
430
|
+
/** 符合分層條件時讀出分層審查需要的資料;不符合時回傳 undefined,已達門檻的會印出原因 */
|
|
431
|
+
function loadLayeredPlan(run, cfg) {
|
|
432
|
+
const tasks = readJsonFile(flowFile(run, "tasks.json"), TaskList);
|
|
433
|
+
const acceptance = readJsonFile(flowFile(run, "acceptance.json"), AcceptanceList);
|
|
434
|
+
if (!tasks.ok || !acceptance.ok)
|
|
435
|
+
return undefined;
|
|
436
|
+
const planMd = flowText(run, "plan.md");
|
|
437
|
+
const decision = layeredReview(tasks.data, planMd, cfg.planReviewLayers);
|
|
438
|
+
if (!decision.layered) {
|
|
439
|
+
if (decision.reason)
|
|
440
|
+
info(run, `📋 這次計畫審查讀整份計畫:${decision.reason}`);
|
|
441
|
+
return undefined;
|
|
442
|
+
}
|
|
443
|
+
const statePath = planReviewStatePath(run.id);
|
|
444
|
+
const state = existsSync(statePath) ? readPlanReviewState(readFileSync(statePath, "utf8")) : undefined;
|
|
445
|
+
return {
|
|
446
|
+
tasks: tasks.data, acceptance: acceptance.data, planMd, state,
|
|
447
|
+
dirty: dirtyGroups(decision.groups, dirtyTaskIds(state?.reviewed, tasks.data, acceptance.data, planMd)),
|
|
448
|
+
};
|
|
449
|
+
}
|
|
450
|
+
/** 分層審查:每輪一次索引審查,再只審有變動的任務群 */
|
|
451
|
+
async function planReviewLayered(run, layered, cfg) {
|
|
452
|
+
const author = run.planWriter ?? planAgent(run.cycle, run.id);
|
|
453
|
+
const round = (run.attempts["plan-review"] ?? 0) + 1;
|
|
454
|
+
const panel = reviewers(run.cycle, author, cfg.planReviewQuorum, `${run.id}:plan-review:${round}`);
|
|
455
|
+
const replies = flowText(run, "plan-replies.md");
|
|
456
|
+
const statePath = planReviewStatePath(run.id);
|
|
457
|
+
const reviewed = layered.state?.reviewed;
|
|
458
|
+
const planKey = currentPlanKey(run);
|
|
459
|
+
// 同一輪因某一呼叫失敗而重跑時,沿用已成功的呼叫:不重複付費,已套用的交接也不會被重新提出
|
|
460
|
+
const progress = roundProgress(layered.state, round, planKey) ?? { round, planKey, calls: [] };
|
|
461
|
+
const saveState = (state) => {
|
|
462
|
+
mkdirSync(runDir(run.id), { recursive: true });
|
|
463
|
+
writeFileSync(statePath, JSON.stringify(state, null, 2));
|
|
464
|
+
};
|
|
465
|
+
const done = new Map(progress.calls.map((call) => [call.key, call]));
|
|
466
|
+
const calls = [];
|
|
467
|
+
// 沿用的呼叫也佔一格,重跑時每個呼叫的 slot 才不會變
|
|
468
|
+
let nextSlot = 0;
|
|
469
|
+
/** 執行或沿用一次呼叫;失敗時回傳 false,run 已是 retry 後的狀態 */
|
|
470
|
+
const runCall = async (key, taskIds, label, spec) => {
|
|
471
|
+
const slot = nextSlot++;
|
|
472
|
+
const reused = done.get(key);
|
|
473
|
+
if (reused) {
|
|
474
|
+
info(run, ` ↪ 沿用本輪已完成的${label}(${reused.reviewer})`);
|
|
475
|
+
calls.push(reused);
|
|
476
|
+
return true;
|
|
477
|
+
}
|
|
478
|
+
info(run, `🧐 ${label}第 ${round} 輪(${spec.reviewer},作者 ${author})`);
|
|
479
|
+
// run 是外層參數,刻意在閉包裡更新:後續呼叫與最後的彙總都要看到 retry、clearModelReviewFailure 之後的 run
|
|
480
|
+
const passed = await collectPlanReview(run, { ...spec, slot });
|
|
481
|
+
run = passed.run;
|
|
482
|
+
if (!passed.collected)
|
|
483
|
+
return false;
|
|
484
|
+
// 真的執行並成功就是有進展:同一輪不同呼叫輪流失敗時,不會累計到重試上限而讓 run 失敗。
|
|
485
|
+
// 進度寫在 round 裡、不會重跑,所以一輪最多失敗「呼叫數 × maxAttempts」次。
|
|
486
|
+
// 整份審查每次重跑整輪,不能這樣歸零,否則同一位審查者反覆失敗會無限重試。
|
|
487
|
+
const attempts = { ...run.attempts };
|
|
488
|
+
delete attempts["plan-review-run"];
|
|
489
|
+
run = { ...run, attempts };
|
|
490
|
+
const call = { key, ...passed.collected, ...(taskIds ? { taskIds } : {}) };
|
|
491
|
+
progress.calls.push(call);
|
|
492
|
+
calls.push(call);
|
|
493
|
+
saveState({ version: 1, ...(reviewed ? { reviewed } : {}), round: progress });
|
|
494
|
+
return true;
|
|
495
|
+
};
|
|
496
|
+
for (const reviewer of panel) {
|
|
497
|
+
const ok = await runCall(`index:${reviewer}`, undefined, "計畫索引審查", {
|
|
498
|
+
reviewer, step: "plan-review", gated: true, subject: "計畫索引",
|
|
499
|
+
prompt: renderPrompt("plan-review-index", {
|
|
500
|
+
reviewer, author, requirement: run.requirement,
|
|
501
|
+
index: planReviewIndex(layered.tasks),
|
|
502
|
+
acceptance: JSON.stringify(layered.acceptance, null, 2),
|
|
503
|
+
overview: planOverview(layered.planMd),
|
|
504
|
+
replies,
|
|
505
|
+
}),
|
|
506
|
+
output: "plan-review.json", archive: `plan-review-${round}-${reviewer}.json`,
|
|
507
|
+
});
|
|
508
|
+
if (!ok)
|
|
509
|
+
return run;
|
|
510
|
+
}
|
|
511
|
+
for (const group of layered.dirty) {
|
|
512
|
+
const groupTasks = layered.tasks.filter((task) => group.taskIds.includes(task.id));
|
|
513
|
+
const acceptanceIds = new Set(groupTasks.flatMap((task) => task.acceptance));
|
|
514
|
+
const neighbors = neighborTasks(layered.tasks, group.taskIds).map(({ id, title, description, dependsOn }) => ({ id, title, description, dependsOn }));
|
|
515
|
+
const groupPanel = reviewers(run.cycle, author, groupReviewerCount(groupTasks, cfg.planReviewQuorum), `${run.id}:plan-group:${group.id}:${round}`);
|
|
516
|
+
for (const reviewer of groupPanel) {
|
|
517
|
+
const ok = await runCall(`group:${group.id}:${group.taskIds.join(",")}:${reviewer}`, group.taskIds, `計畫群 ${group.id} 審查`, {
|
|
518
|
+
reviewer, step: "plan-review-group", gated: false, subject: `任務群 ${group.id}`, scope: group.id,
|
|
519
|
+
prompt: renderPrompt("plan-review-group", {
|
|
520
|
+
reviewer, author, groupId: group.id,
|
|
521
|
+
files: group.files.join("、") || "(這群的描述沒有點名檔案)",
|
|
522
|
+
tasks: JSON.stringify(groupTasks, null, 2),
|
|
523
|
+
neighbors: neighbors.length ? JSON.stringify(neighbors, null, 2) : "(沒有跨群的直接相依)",
|
|
524
|
+
acceptance: JSON.stringify(layered.acceptance.filter((item) => acceptanceIds.has(item.id)), null, 2),
|
|
525
|
+
evidence: extractPlanEvidence(layered.planMd, group.taskIds),
|
|
526
|
+
replies: repliesForTasks(replies, group.taskIds),
|
|
527
|
+
}),
|
|
528
|
+
output: "plan-review-group.json", archive: `plan-review-${round}-${group.id}-${reviewer}.json`,
|
|
529
|
+
});
|
|
530
|
+
if (!ok)
|
|
531
|
+
return run;
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
// 只放這一輪真的審過的群(含沿用的),沒審到的任務才留得住前次 verdict
|
|
535
|
+
const groupVerdicts = calls.flatMap((call) => call.taskIds ? [{ taskIds: call.taskIds, verdict: call.verdict }] : []);
|
|
536
|
+
saveState({ version: 1, reviewed: applyReviewVerdicts(reviewed, layered.tasks, layered.acceptance, layered.planMd, groupVerdicts) });
|
|
537
|
+
return concludePlanReview(run, cfg, round, calls);
|
|
332
538
|
}
|
|
333
539
|
async function planFixStage(run) {
|
|
334
540
|
const cfg = loadRepoConfig();
|
|
@@ -340,24 +546,35 @@ async function planFixStage(run) {
|
|
|
340
546
|
});
|
|
341
547
|
info(run, `✏️ 依 ${run.planReviewer ?? "審查者"} 的意見修改計畫(${agent})`);
|
|
342
548
|
const feedback = readFeedback(run);
|
|
343
|
-
const snap = snapshotPlan(run);
|
|
344
|
-
|
|
549
|
+
const snap = snapshotPlan(run, PLAN_REPLY_FILES);
|
|
550
|
+
// 回應每輪整份覆寫:先刪掉上一輪的,agent 沒寫時下一輪才不會讀到舊回應;失敗時快照會還原
|
|
551
|
+
const clearReplies = () => rmSync(flowFile(run, "plan-replies.md"), { force: true });
|
|
552
|
+
clearReplies();
|
|
553
|
+
let outcome;
|
|
554
|
+
try {
|
|
555
|
+
outcome = await agentStep(run, agent, "plan-fix", renderPrompt("plan-fix", { requirement: run.requirement, testPattern: cfg.testPattern }), { kind: "write", reset: async () => { await discardChanges(worktreeDir(run.id)); restorePlan(run, snap); clearReplies(); } });
|
|
556
|
+
}
|
|
557
|
+
catch (err) {
|
|
558
|
+
// 所有 agent 額度都用完而暫停:還原成進入時的內容,resume 前的計畫審查回應仍在
|
|
559
|
+
restorePlan(run, snap);
|
|
560
|
+
throw err;
|
|
561
|
+
}
|
|
345
562
|
const { r, agent: actual } = outcome;
|
|
346
563
|
await discardChanges(worktreeDir(run.id)); // 只允許改 .flow/
|
|
347
564
|
if (!r.ok) {
|
|
348
565
|
restorePlan(run, snap);
|
|
349
|
-
return retry(run, "plan-fix", `${feedback}\n\n(上次修改時 Agent 執行失敗:${r.summary})`, "plan_fix");
|
|
566
|
+
return retry(run, "plan-fix", `${feedback}\n\n(上次修改時 Agent 執行失敗:${r.summary})`, "plan_fix", "agent_error");
|
|
350
567
|
}
|
|
351
568
|
const ordered = validatePlan(run);
|
|
352
569
|
if (typeof ordered === "string") {
|
|
353
570
|
restorePlan(run, snap);
|
|
354
|
-
return retry(run, "plan-fix", `${feedback}\n\n另外,修改後的計畫沒有通過格式檢查,已還原:\n${ordered}`, "plan_fix");
|
|
571
|
+
return retry(run, "plan-fix", `${feedback}\n\n另外,修改後的計畫沒有通過格式檢查,已還原:\n${ordered}`, "plan_fix", "format_invalid");
|
|
355
572
|
}
|
|
356
573
|
const handoffError = finishHandoff(run, outcome, "writer");
|
|
357
574
|
if (handoffError) {
|
|
358
575
|
restorePlan(run, snap);
|
|
359
576
|
// 計畫已還原,要保留原本的審查意見,否則下一次修正不知道要改什麼
|
|
360
|
-
return retry(run, "plan-fix", `${feedback}\n\n另外,交接回覆不合格,本次修改已還原:${handoffError}`, "plan_fix");
|
|
577
|
+
return retry(run, "plan-fix", `${feedback}\n\n另外,交接回覆不合格,本次修改已還原:${handoffError}`, "plan_fix", "handoff_invalid");
|
|
361
578
|
}
|
|
362
579
|
acceptPlan(run, ordered);
|
|
363
580
|
writeFileSync(flowFile(run, "feedback.md"), feedback); // 保留審查意見,讓下一輪審查者知道上次提了什麼
|
|
@@ -377,7 +594,7 @@ async function arbitratePlan(run) {
|
|
|
377
594
|
const verdicts = [];
|
|
378
595
|
for (const [slot, arbiter] of panel.entries()) {
|
|
379
596
|
info(run, `⚖️ ${mode}(${arbiter})`);
|
|
380
|
-
const snap = snapshotPlan(run);
|
|
597
|
+
const snap = snapshotPlan(run, PLAN_REPLY_FILES);
|
|
381
598
|
rmSync(flowFile(run, "plan-arbiter.json"), { force: true });
|
|
382
599
|
const outcome = await agentStep(run, arbiter, "plan-arbiter", renderPrompt("plan-arbiter", { requirement: run.requirement }), {
|
|
383
600
|
kind: "review", slot, blind: true,
|
|
@@ -391,8 +608,14 @@ async function arbitratePlan(run) {
|
|
|
391
608
|
if (!result?.ok || handoffError) {
|
|
392
609
|
const outputError = !r.ok ? `Agent 執行失敗:${r.summary}` : !result ? "未產生有效裁決" : result.ok ? "未產生有效裁決" : result.error;
|
|
393
610
|
const reason = handoffError ?? outputError;
|
|
394
|
-
|
|
395
|
-
|
|
611
|
+
const category = handoffError ? "handoff_invalid" : !r.ok ? "agent_error" : "format_invalid";
|
|
612
|
+
info(run, ` ✗ ${arbiter} 未產生有效裁決:${reason}`);
|
|
613
|
+
// 無效的裁決移到 worktree 外保存,供事後查看;plan-arbitration.json 還在,重試時直接回到仲裁
|
|
614
|
+
if (existsSync(flowFile(run, "plan-arbiter.json"))) {
|
|
615
|
+
mkdirSync(join(runDir(run.id), "reviews"), { recursive: true });
|
|
616
|
+
renameSync(flowFile(run, "plan-arbiter.json"), join(runDir(run.id), "reviews", `plan-arbiter-${arbitrationRound}-${arbiter}-invalid.json`));
|
|
617
|
+
}
|
|
618
|
+
return retry(run, "plan-arbitration-run", `上次仲裁未產生有效裁決:${reason}`, "plan_review", category);
|
|
396
619
|
}
|
|
397
620
|
mkdirSync(join(runDir(run.id), "reviews"), { recursive: true });
|
|
398
621
|
renameSync(flowFile(run, "plan-arbiter.json"), join(runDir(run.id), "reviews", `plan-arbiter-${arbitrationRound}-${arbiter}.json`));
|
|
@@ -403,6 +626,9 @@ async function arbitratePlan(run) {
|
|
|
403
626
|
verdicts.push({ arbiter, verdict, notes, markdownNotes });
|
|
404
627
|
}
|
|
405
628
|
rmSync(flowFile(run, "dispute.md"), { force: true });
|
|
629
|
+
rmSync(planArbitrationPath(run.id), { force: true });
|
|
630
|
+
run = { ...run, attempts: { ...run.attempts } };
|
|
631
|
+
delete run.attempts["plan-arbitration-run"];
|
|
406
632
|
const approvals = verdicts.filter((v) => v.verdict === "approve").length;
|
|
407
633
|
const unanimous = approvals === verdicts.length;
|
|
408
634
|
const decision = arbitrationDecision(verdicts.map((v) => v.verdict), cfg.tieBreak);
|
|
@@ -418,6 +644,8 @@ async function arbitratePlan(run) {
|
|
|
418
644
|
const attempts = { ...run.attempts, "plan-arbitration": arbitrationRound };
|
|
419
645
|
delete attempts["plan-review"];
|
|
420
646
|
rmSync(flowFile(run, "plan-review-last.txt"), { force: true });
|
|
647
|
+
rmSync(planReviewStatePath(run.id), { force: true });
|
|
648
|
+
addRetry(run.id, { key: "plan-arbitration", backTo: "plan_fix", category: "arbitration_revise", attempt: arbitrationRound, final: false }, run.updatedAt);
|
|
421
649
|
writeFileSync(flowFile(run, "feedback.md"), `# 仲裁要求修訂(第 ${arbitrationRound} 次)\n\n${summary}\n\n${feedbackRecord}\n`);
|
|
422
650
|
info(run, ` → ${summary}`);
|
|
423
651
|
return { ...run, attempts, stage: "plan_fix" };
|
|
@@ -427,7 +655,7 @@ async function arbitratePlan(run) {
|
|
|
427
655
|
info(run, ` → ${summary}`);
|
|
428
656
|
return planSettled(run, "plan-review");
|
|
429
657
|
}
|
|
430
|
-
return { ...run, stage: "failed", failedStage: "plan_review", failureReason: `${summary},需要人工決定(見 plan.md 的仲裁紀錄)` };
|
|
658
|
+
return { ...run, stage: "failed", failedStage: "plan_review", failureCategory: "arbitration_stop", failureReason: `${summary},需要人工決定(見 plan.md 的仲裁紀錄)` };
|
|
431
659
|
}
|
|
432
660
|
function planTamperedMessage(files) {
|
|
433
661
|
return `計畫定案後不可修改規格與計畫檔,已還原你的變更:${files.map((f) => `.flow/${f}`).join(", ")}。若認為規格或驗收條件有誤,請寫進 .flow/handoff-response.json 的 newIssues。`;
|
|
@@ -456,6 +684,13 @@ async function implementStage(run) {
|
|
|
456
684
|
if (run.taskPhase === "fix")
|
|
457
685
|
return taskFixStep(run, task, progress);
|
|
458
686
|
const agents = taskAgents(run.cycle, run.taskIndex, cfg.tddSplit, run.id);
|
|
687
|
+
const framework = hasTestFramework();
|
|
688
|
+
const tdd = taskUsesTdd(task, framework);
|
|
689
|
+
// ── 不走 TDD:略過紅燈,實作前的 HEAD 就是這個任務的起點 ──
|
|
690
|
+
if (!tdd && run.taskPhase === "tests") {
|
|
691
|
+
info(run, `⏭️ [${progress}] 略過 TDD(${framework ? "planner 標記不適合先寫測試" : "專案沒有測試框架"})`);
|
|
692
|
+
return { ...run, taskPhase: "code", taskBase: await headCommit(repo), testsCommit: undefined, lastTestsAuthor: undefined };
|
|
693
|
+
}
|
|
459
694
|
// ── 紅燈:只寫測試,而且測試必須失敗 ──
|
|
460
695
|
if (run.taskPhase === "tests") {
|
|
461
696
|
const key = `${task.id}:tests`;
|
|
@@ -467,29 +702,29 @@ async function implementStage(run) {
|
|
|
467
702
|
const tampered = restorePlan(run, snap);
|
|
468
703
|
if (!r.ok) {
|
|
469
704
|
await resetTo(repo, before);
|
|
470
|
-
return retry(run, key, `Agent 執行失敗:${r.summary}`, "implement");
|
|
705
|
+
return retry(run, key, `Agent 執行失敗:${r.summary}`, "implement", "agent_error");
|
|
471
706
|
}
|
|
472
707
|
if (tampered.length) {
|
|
473
708
|
await resetTo(repo, before);
|
|
474
|
-
return retry(run, key, planTamperedMessage(tampered), "implement");
|
|
709
|
+
return retry(run, key, planTamperedMessage(tampered), "implement", "plan_tampered");
|
|
475
710
|
}
|
|
476
711
|
const commit = await commitAll(repo, `test(${task.id}): ${task.title} [${testsAuthor}]`);
|
|
477
712
|
if (!commit)
|
|
478
|
-
return retry(run, key, "沒有任何檔案變更,這個階段必須撰寫測試。", "implement");
|
|
713
|
+
return retry(run, key, "沒有任何檔案變更,這個階段必須撰寫測試。", "implement", "tests_not_written");
|
|
479
714
|
const changed = await changedFiles(repo, before, commit);
|
|
480
715
|
if (!changed.some((f) => testRe.test(f))) {
|
|
481
716
|
await resetTo(repo, before);
|
|
482
|
-
return retry(run, key, `沒有新增或修改任何符合 /${cfg.testPattern}/ 的測試檔。`, "implement");
|
|
717
|
+
return retry(run, key, `沒有新增或修改任何符合 /${cfg.testPattern}/ 的測試檔。`, "implement", "tests_not_written");
|
|
483
718
|
}
|
|
484
719
|
const red = await runCommand(target(run, `${task.id}-red`, CMD_AGENT), testCmd);
|
|
485
720
|
if (red.ok) {
|
|
486
721
|
await resetTo(repo, before);
|
|
487
|
-
return retry(run, key, "測試在功能尚未實作前就全部通過,代表測試沒有驗證到新行為。請撰寫會因功能尚未實作而失敗的測試。", "implement");
|
|
722
|
+
return retry(run, key, "測試在功能尚未實作前就全部通過,代表測試沒有驗證到新行為。請撰寫會因功能尚未實作而失敗的測試。", "implement", "tests_not_red");
|
|
488
723
|
}
|
|
489
724
|
const handoffError = finishHandoff(run, outcome, "writer");
|
|
490
725
|
if (handoffError) {
|
|
491
726
|
await resetTo(repo, before);
|
|
492
|
-
return retry(run, key, handoffError, "implement");
|
|
727
|
+
return retry(run, key, handoffError, "implement", "handoff_invalid");
|
|
493
728
|
}
|
|
494
729
|
writeFileSync(flowFile(run, "red-output.txt"), red.output);
|
|
495
730
|
info(run, `🔴 [${progress}] 測試如預期失敗`);
|
|
@@ -497,38 +732,44 @@ async function implementStage(run) {
|
|
|
497
732
|
}
|
|
498
733
|
// ── 綠燈:實作到測試通過,而且不可動測試 ──
|
|
499
734
|
const key = `${task.id}:code`;
|
|
500
|
-
|
|
735
|
+
// 走 TDD 時以測試 commit 為基準;不走 TDD 時以任務起點為基準
|
|
736
|
+
const testsCommit = tdd ? run.testsCommit : run.taskBase;
|
|
501
737
|
if (!testsCommit)
|
|
502
|
-
throw new Error("缺少 testsCommit,狀態不一致");
|
|
503
|
-
info(run, `🛠️ [${progress}] 實作(${agents.code},測試由 ${run.lastTestsAuthor ?? agents.tests} 撰寫)`);
|
|
738
|
+
throw new Error(tdd ? "缺少 testsCommit,狀態不一致" : "缺少 taskBase,狀態不一致");
|
|
739
|
+
info(run, tdd ? `🛠️ [${progress}] 實作(${agents.code},測試由 ${run.lastTestsAuthor ?? agents.tests} 撰寫)` : `🛠️ [${progress}] 實作(${agents.code},不走 TDD)`);
|
|
504
740
|
const redOutput = existsSync(flowFile(run, "red-output.txt")) ? readFileSync(flowFile(run, "red-output.txt"), "utf8") : "";
|
|
505
741
|
const snap = snapshotPlan(run, LOCKED_FILES);
|
|
506
|
-
const outcome = await agentStep(run, agents.code, `${task.id}-code`,
|
|
742
|
+
const outcome = await agentStep(run, agents.code, `${task.id}-code`, tdd
|
|
743
|
+
? renderPrompt("implement-code", { task: taskJson, acceptance: acceptanceJson, testCmd, redOutput: tail(redOutput, 3000) })
|
|
744
|
+
: renderPrompt("implement-direct", { task: taskJson, acceptance: acceptanceJson, testCmd: framework ? testCmd : "" }), { kind: "write", reset: async () => { await resetTo(repo, testsCommit); restorePlan(run, snap); } });
|
|
507
745
|
const { r, agent: codeAuthor } = outcome;
|
|
508
746
|
const tampered = restorePlan(run, snap);
|
|
509
747
|
if (!r.ok)
|
|
510
|
-
return retry(run, key, `Agent 執行失敗:${r.summary}`, "implement");
|
|
748
|
+
return retry(run, key, `Agent 執行失敗:${r.summary}`, "implement", "agent_error");
|
|
511
749
|
if (tampered.length) {
|
|
512
750
|
await resetTo(repo, testsCommit);
|
|
513
|
-
return retry(run, key, planTamperedMessage(tampered), "implement");
|
|
751
|
+
return retry(run, key, planTamperedMessage(tampered), "implement", "plan_tampered");
|
|
514
752
|
}
|
|
515
|
-
await commitAll(repo, `feat(${task.id}): ${task.title} [${codeAuthor}]`);
|
|
516
|
-
|
|
753
|
+
const codeCommit = await commitAll(repo, `feat(${task.id}): ${task.title} [${codeAuthor}]`);
|
|
754
|
+
if (!tdd && !codeCommit)
|
|
755
|
+
return retry(run, key, "沒有任何檔案變更,這個任務必須完成實作。", "implement", "code_not_written");
|
|
756
|
+
const touched = tdd ? (await changedFiles(repo, testsCommit, await headCommit(repo))).filter((f) => testRe.test(f)) : [];
|
|
517
757
|
if (touched.length) {
|
|
518
758
|
await resetTo(repo, testsCommit);
|
|
519
|
-
return retry(run, key, `實作階段不可修改測試檔,已還原你的變更:${touched.join(", ")}`, "implement");
|
|
759
|
+
return retry(run, key, `實作階段不可修改測試檔,已還原你的變更:${touched.join(", ")}`, "implement", "tests_modified");
|
|
520
760
|
}
|
|
521
|
-
|
|
761
|
+
// 沒有測試框架時沒有東西可跑,後面的驗證與審查照常把關
|
|
762
|
+
const green = framework ? await runCommand(target(run, `${task.id}-green`, CMD_AGENT), testCmd) : { ok: true, output: "", seq: 0 };
|
|
522
763
|
if (!green.ok) {
|
|
523
764
|
info(run, ` ✗ 測試仍未通過${logHint(run, green.seq)}`);
|
|
524
|
-
return retry(run, key, `測試仍未通過:\n\n\`\`\`\n${tail(green.output)}\n\`\`\``, "implement");
|
|
765
|
+
return retry(run, key, `測試仍未通過:\n\n\`\`\`\n${tail(green.output)}\n\`\`\``, "implement", "tests_not_green");
|
|
525
766
|
}
|
|
526
767
|
const handoffError = finishHandoff(run, outcome, "writer");
|
|
527
768
|
if (handoffError) {
|
|
528
769
|
await resetTo(repo, testsCommit);
|
|
529
|
-
return retry(run, key, handoffError, "implement");
|
|
770
|
+
return retry(run, key, handoffError, "implement", "handoff_invalid");
|
|
530
771
|
}
|
|
531
|
-
info(run, `🟢 [${progress}] 測試通過`);
|
|
772
|
+
info(run, framework ? `🟢 [${progress}] 測試通過` : `✔️ [${progress}] 實作完成`);
|
|
532
773
|
return { ...succeed(run, key, "implement"), taskPhase: "review", lastWriter: codeAuthor };
|
|
533
774
|
}
|
|
534
775
|
// ── 任務審查:只看這個任務的變更與驗收條件 ──
|
|
@@ -558,7 +799,7 @@ async function taskReviewStep(run, task, progress, taskJson, acceptanceJson) {
|
|
|
558
799
|
if (!result.objector)
|
|
559
800
|
return { ...succeed(reviewed, key, "implement"), taskPhase: "verify" };
|
|
560
801
|
return {
|
|
561
|
-
...retry(reviewed, key, `任務審查要求修改:\n\n${result.issues.join("\n\n")}`, "implement"),
|
|
802
|
+
...retry(reviewed, key, `任務審查要求修改:\n\n${result.issues.join("\n\n")}`, "implement", "review_changes"),
|
|
562
803
|
taskPhase: "fix",
|
|
563
804
|
fixSource: "review",
|
|
564
805
|
lastReviewer: result.objector,
|
|
@@ -570,7 +811,7 @@ async function taskVerifyStep(run, task, progress) {
|
|
|
570
811
|
info(run, `🔍 [${progress}] 執行驗證`);
|
|
571
812
|
const report = await runChecks(run, `${task.id}-`);
|
|
572
813
|
if (report)
|
|
573
|
-
return { ...retry(run, key, report, "implement"), taskPhase: "fix", fixSource: "verify" };
|
|
814
|
+
return { ...retry(run, key, report, "implement", "checks_failed"), taskPhase: "fix", fixSource: "verify" };
|
|
574
815
|
info(run, `✅ [${progress}] 完成`);
|
|
575
816
|
return {
|
|
576
817
|
...succeed(run, key, "implement"),
|
|
@@ -622,7 +863,7 @@ async function verifyStage(run) {
|
|
|
622
863
|
const report = await runChecks(run);
|
|
623
864
|
if (!report)
|
|
624
865
|
return succeed(run, "verify", "review");
|
|
625
|
-
return { ...retry(run, "verify", report, "fix"), fixSource: "verify" };
|
|
866
|
+
return { ...retry(run, "verify", report, "fix", "checks_failed"), fixSource: "verify" };
|
|
626
867
|
}
|
|
627
868
|
/** 修正驗證錯誤或審查意見;成功時回傳實際修正者,未通過時回傳重試後的 run */
|
|
628
869
|
async function applyFix(run, opts) {
|
|
@@ -647,24 +888,24 @@ async function applyFix(run, opts) {
|
|
|
647
888
|
});
|
|
648
889
|
const { r, agent: actual } = outcome;
|
|
649
890
|
const tampered = restorePlan(run, snap);
|
|
650
|
-
const again = (reason) => ({ run: retry(run, opts.key, reason, opts.backTo) });
|
|
891
|
+
const again = (reason, category) => ({ run: retry(run, opts.key, reason, opts.backTo, category) });
|
|
651
892
|
if (!r.ok)
|
|
652
|
-
return again(`${feedback}\n\n(上次修正時 Agent 執行失敗:${r.summary}
|
|
893
|
+
return again(`${feedback}\n\n(上次修正時 Agent 執行失敗:${r.summary})`, "agent_error");
|
|
653
894
|
if (tampered.length) {
|
|
654
895
|
await resetTo(repo, before);
|
|
655
|
-
return again(`${feedback}\n\n另外:${planTamperedMessage(tampered)}
|
|
896
|
+
return again(`${feedback}\n\n另外:${planTamperedMessage(tampered)}`, "plan_tampered");
|
|
656
897
|
}
|
|
657
898
|
await commitAll(repo, `${opts.commitScope}: ${why} [${actual}]`);
|
|
658
899
|
const testRe = new RegExp(cfg.testPattern);
|
|
659
900
|
const deleted = (await changedFiles(repo, before, await headCommit(repo), "D")).filter((f) => testRe.test(f));
|
|
660
901
|
if (deleted.length) {
|
|
661
902
|
await resetTo(repo, before);
|
|
662
|
-
return again(`${feedback}\n\n另外:不可刪除測試檔來讓檢查通過,已還原:${deleted.join(", ")}
|
|
903
|
+
return again(`${feedback}\n\n另外:不可刪除測試檔來讓檢查通過,已還原:${deleted.join(", ")}`, "tests_deleted");
|
|
663
904
|
}
|
|
664
905
|
const handoffError = finishHandoff(run, outcome, "writer");
|
|
665
906
|
if (handoffError) {
|
|
666
907
|
await resetTo(repo, before);
|
|
667
|
-
return again(`${feedback}\n\n另外,交接回覆不合格,本次修正已還原:${handoffError}
|
|
908
|
+
return again(`${feedback}\n\n另外,交接回覆不合格,本次修正已還原:${handoffError}`, "handoff_invalid");
|
|
668
909
|
}
|
|
669
910
|
return { agent: actual };
|
|
670
911
|
}
|
|
@@ -709,14 +950,14 @@ async function codeReview(run, opts) {
|
|
|
709
950
|
if (tampered.length)
|
|
710
951
|
info(run, ` ↩️ 已還原審查者修改的檔案:${tampered.join(", ")}`);
|
|
711
952
|
if (!r.ok)
|
|
712
|
-
return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, `Agent 執行失敗:${r.summary}`, opts.backTo) };
|
|
953
|
+
return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, `Agent 執行失敗:${r.summary}`, opts.backTo, "agent_error") };
|
|
713
954
|
const review = readJsonFile(flowFile(run, "review.json"), ConsistentReviewResult);
|
|
714
955
|
if (!review.ok)
|
|
715
|
-
return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, review.error, opts.backTo) };
|
|
956
|
+
return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, review.error, opts.backTo, "format_invalid") };
|
|
716
957
|
const gate = opts.gate ? { target: "code", verdict: review.data.verdict } : undefined;
|
|
717
958
|
const handoffError = finishHandoff(run, outcome, "reviewer", gate);
|
|
718
959
|
if (handoffError)
|
|
719
|
-
return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, handoffError, opts.backTo) };
|
|
960
|
+
return { run: retry(opts.step === "review" ? recordModelReviewFailure(run, "review", reviewer) : run, opts.runKey, handoffError, opts.backTo, "handoff_invalid") };
|
|
720
961
|
if (opts.step === "review")
|
|
721
962
|
run = clearModelReviewFailure(run, "review", reviewer);
|
|
722
963
|
renameSync(flowFile(run, "review.json"), flowFile(run, opts.saveAs(reviewer)));
|
|
@@ -753,7 +994,7 @@ async function reviewStage(run) {
|
|
|
753
994
|
if (!result.objector)
|
|
754
995
|
return succeed(result.state, "review", "pr");
|
|
755
996
|
return {
|
|
756
|
-
...retry(result.state, "review", `程式碼審查要求修改:\n\n${result.issues.join("\n\n")}`, "fix"),
|
|
997
|
+
...retry(result.state, "review", `程式碼審查要求修改:\n\n${result.issues.join("\n\n")}`, "fix", "review_changes"),
|
|
757
998
|
fixSource: "review",
|
|
758
999
|
lastReviewer: result.objector,
|
|
759
1000
|
};
|
|
@@ -761,7 +1002,7 @@ async function reviewStage(run) {
|
|
|
761
1002
|
async function prStage(run) {
|
|
762
1003
|
const pending = openActions(readHandoff(run.id));
|
|
763
1004
|
if (pending.length)
|
|
764
|
-
return { ...run, stage: "failed", failedStage: "pr", failureReason: `仍有未結交接事項:${pending.map((item) => item.id).join("、")}` };
|
|
1005
|
+
return { ...run, stage: "failed", failedStage: "pr", failureCategory: "open_handoff", failureReason: `仍有未結交接事項:${pending.map((item) => item.id).join("、")}` };
|
|
765
1006
|
const repo = worktreeDir(run.id);
|
|
766
1007
|
const remotes = (await git(repo, "remote")).split("\n").filter(Boolean);
|
|
767
1008
|
if (!remotes.includes("origin")) {
|
|
@@ -806,6 +1047,7 @@ export async function advance(initial) {
|
|
|
806
1047
|
...run,
|
|
807
1048
|
stage: "failed",
|
|
808
1049
|
failedStage: stage,
|
|
1050
|
+
failureCategory: "agent_budget",
|
|
809
1051
|
failureReason: `已執行 agent ${runs} 次,達到上限 ${run.maxAgentRuns}(可用 resume --max-agent-runs 調高)`,
|
|
810
1052
|
});
|
|
811
1053
|
}
|
|
@@ -817,7 +1059,7 @@ export async function advance(initial) {
|
|
|
817
1059
|
info(run, `⏸️ 暫停:${err.message}`);
|
|
818
1060
|
return saveRun({ ...run, stage: "paused", pausedStage: stage, pauseReason: err.message });
|
|
819
1061
|
}
|
|
820
|
-
return saveRun({ ...run, stage: "failed", failedStage: stage, failureReason: err.message });
|
|
1062
|
+
return saveRun({ ...run, stage: "failed", failedStage: stage, failureCategory: "error", failureReason: err.message });
|
|
821
1063
|
}
|
|
822
1064
|
}
|
|
823
1065
|
}
|