agentflowctl 0.13.1 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/store.js CHANGED
@@ -36,9 +36,21 @@ export function addUsage(id, entry) {
36
36
  appendFileSync(usagePath(id), `${JSON.stringify({ at: new Date().toISOString(), ...entry })}\n`);
37
37
  }
38
38
  const emptySummary = () => ({ tokens: 0, inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0, runs: 0, reportedRuns: 0, unreportedRuns: 0, legacyRuns: 0, legacyTokens: 0 });
39
+ /** 逐行讀取 jsonl;寫到一半被中斷的殘行直接略過,不讓 status、insights 整個讀不出來 */
40
+ function readJsonl(p) {
41
+ if (!existsSync(p))
42
+ return [];
43
+ return readFileSync(p, "utf8").split("\n").filter(Boolean).flatMap((l) => {
44
+ try {
45
+ return [JSON.parse(l)];
46
+ }
47
+ catch {
48
+ return [];
49
+ }
50
+ });
51
+ }
39
52
  export function listUsage(id) {
40
- const p = usagePath(id);
41
- return existsSync(p) ? readFileSync(p, "utf8").split("\n").filter(Boolean).map((line) => JSON.parse(line)) : [];
53
+ return readJsonl(usagePath(id));
42
54
  }
43
55
  function addSummary(acc, e) {
44
56
  acc.runs += 1;
@@ -58,23 +70,37 @@ function addSummary(acc, e) {
58
70
  }
59
71
  }
60
72
  /** 只加總明確回報的 token,舊資料的 0 不視為已回報。 */
61
- function groupUsage(id, keyOf) {
73
+ export function totalUsage(entries) {
74
+ const acc = emptySummary();
75
+ for (const e of entries)
76
+ addSummary(acc, e);
77
+ return acc;
78
+ }
79
+ export function summarizeUsage(entries, keyOf) {
62
80
  const out = {};
63
- for (const e of listUsage(id))
81
+ for (const e of entries)
64
82
  addSummary(out[keyOf(e)] ??= emptySummary(), e);
65
83
  return out;
66
84
  }
67
- export const usageByAgent = (id) => groupUsage(id, (e) => e.agent);
68
- export const usageByModelStage = (id) => groupUsage(id, (e) => `${e.model ?? "CLI 預設(名稱未知)"} / ${e.stage}`);
69
- export const usageByStrength = (id) => groupUsage(id, (e) => e.strength ?? "未知");
70
- /** 依階段鍵加總(所有任務的同一步合在一起);認不得的步驟歸為「其他」。 */
71
- export const usageByStage = (id) => groupUsage(id, (e) => stageOfStep(e.stage) ?? "其他");
72
- /** 依任務加總寫測試、實作、任務審查與任務修正;規格、計畫與整體階段歸為「非任務步驟」。 */
73
- export const usageByTask = (id) => groupUsage(id, (e) => /^(T-\d+)-/.exec(e.stage)?.[1] ?? "非任務步驟");
85
+ export const usageKeyAgent = (e) => e.agent;
86
+ export const usageKeyStrength = (e) => e.strength ?? "未知";
87
+ export const usageKeyStage = (e) => stageOfStep(e.stage) ?? "其他";
88
+ export const usageKeyTask = (e) => /^(T-\d+)-/.exec(e.stage)?.[1] ?? "非任務步驟";
89
+ const modelName = (e) => e.model ?? "CLI 預設(名稱未知)";
90
+ export const usageKeyModelStage = (e) => `${modelName(e)} / ${e.stage}`;
91
+ /** 跨 run 用:任務步驟換成步驟種類(T-1-code → taskCode),不同 run 的 task id 不互相合併 */
92
+ export const usageKeyModelStageKind = (e) => `${modelName(e)} / ${stageOfStep(e.stage) ?? e.stage}`;
93
+ function groupUsage(id, keyOf) {
94
+ return summarizeUsage(listUsage(id), keyOf);
95
+ }
96
+ export const usageByAgent = (id) => groupUsage(id, usageKeyAgent);
97
+ export const usageByModelStage = (id) => groupUsage(id, usageKeyModelStage);
98
+ export const usageByStrength = (id) => groupUsage(id, usageKeyStrength);
99
+ export const usageByStage = (id) => groupUsage(id, usageKeyStage);
100
+ export const usageByTask = (id) => groupUsage(id, usageKeyTask);
74
101
  /** 這個 run 已執行 agent 的次數(每次執行都會記一筆用量) */
75
102
  export function agentRuns(id) {
76
- const p = usagePath(id);
77
- return existsSync(p) ? readFileSync(p, "utf8").split("\n").filter(Boolean).length : 0;
103
+ return listUsage(id).length;
78
104
  }
79
105
  const subPath = (id) => join(runDir(id), "substitutions.jsonl");
80
106
  export function addSubstitution(id, s) {
@@ -82,9 +108,28 @@ export function addSubstitution(id, s) {
82
108
  appendFileSync(subPath(id), `${JSON.stringify({ at: new Date().toISOString(), ...s })}\n`);
83
109
  }
84
110
  export function listSubstitutions(id) {
85
- const p = subPath(id);
86
- if (!existsSync(p))
87
- return [];
88
- return readFileSync(p, "utf8").split("\n").filter(Boolean).map((l) => JSON.parse(l));
111
+ return readJsonl(subPath(id));
112
+ }
113
+ export const RetryCategories = [
114
+ "agent_error", "missing_artifact", "format_invalid", "handoff_invalid", "open_handoff",
115
+ "review_changes", "arbitration_revise", "plan_tampered", "tests_not_written", "code_not_written", "tests_not_red", "tests_modified",
116
+ "tests_not_green", "tests_deleted", "checks_failed",
117
+ ];
118
+ const retryPath = (id) => join(runDir(id), "retries.jsonl");
119
+ /**
120
+ * savedAt 是目前 state.json 的 updatedAt。上一筆同 key、同 attempt 的紀錄若晚於它,
121
+ * 代表上次寫入重試後還沒存到 state 就中斷了,resume 重跑同一步時不再重複記一筆。
122
+ */
123
+ export function addRetry(id, entry, savedAt) {
124
+ if (savedAt) {
125
+ const last = listRetries(id).filter((r) => r.key === entry.key).at(-1);
126
+ if (last && last.attempt === entry.attempt && last.at > savedAt)
127
+ return;
128
+ }
129
+ mkdirSync(runDir(id), { recursive: true });
130
+ appendFileSync(retryPath(id), `${JSON.stringify({ at: new Date().toISOString(), ...entry })}\n`);
131
+ }
132
+ export function listRetries(id) {
133
+ return readJsonl(retryPath(id));
89
134
  }
90
135
  //# sourceMappingURL=store.js.map
@@ -0,0 +1,156 @@
1
+ import { retryLabel } from "./insights.js";
2
+ import { RetryCategories } from "./store.js";
3
+ import { summarizeUsage, totalUsage, usageKeyAgent, usageKeyModelStageKind, usageKeyStage, usageKeyStrength, usageKeyTask } from "./store.js";
4
+ import { mergeStats } from "./stats.js";
5
+ export const FindingCodes = [
6
+ "coverage_low", "high_strength_share", "retry_waste", "input_heavy",
7
+ "hot_stage", "hot_task", "substitution_waste", "cache_unread",
8
+ "hot_model_step", "hot_failing_step",
9
+ ];
10
+ export const FINDING_LABEL = {
11
+ coverage_low: "用量回報不完整",
12
+ high_strength_share: "高強度模型占比偏高",
13
+ retry_waste: "重試可能比單次 prompt 更耗 token",
14
+ input_heavy: "輸入遠大於輸出",
15
+ hot_stage: "單一階段佔用量過高",
16
+ hot_task: "單一任務佔用量過高",
17
+ substitution_waste: "額度代打造成重做",
18
+ cache_unread: "cache 寫入多、讀取少",
19
+ hot_model_step: "單一模型與步驟佔用量過高",
20
+ hot_failing_step: "單一執行步驟失敗次數過多",
21
+ };
22
+ /** 會算進 retry_waste 的重試:排除審查要求修改與仲裁要求修訂這類正常往返 */
23
+ const WASTEFUL_RETRY = (r) => r.category !== "review_changes" && r.category !== "arbitration_revise";
24
+ export function topTaskOf(usage) {
25
+ const entries = Object.entries(summarizeUsage(usage, usageKeyTask)).filter(([key, v]) => key !== "非任務步驟" && v.tokens > 0);
26
+ const taskTotal = entries.reduce((n, [, v]) => n + v.tokens, 0);
27
+ const top = entries.sort((a, b) => b[1].tokens - a[1].tokens)[0];
28
+ return top ? { task: top[0], tokens: top[1].tokens, share: top[1].tokens / taskTotal, tasks: entries.length } : undefined;
29
+ }
30
+ const pct = (part, whole) => (whole ? `${(part / whole * 100).toFixed(1)}%` : "無法計算");
31
+ function topByTokens(record) {
32
+ return Object.entries(record)
33
+ .filter(([, v]) => v.tokens > 0)
34
+ .sort((a, b) => b[1].tokens - a[1].tokens)[0];
35
+ }
36
+ export function usageFindings(input) {
37
+ const { total, byStrength, byStage, substitutions, topTasks = [], byModelStage = {}, steps = [] } = input;
38
+ const retries = input.retries.filter(WASTEFUL_RETRY);
39
+ const out = [];
40
+ const avg = total.reportedRuns ? Math.round(total.tokens / total.reportedRuns) : 0;
41
+ const unknownCalls = total.unreportedRuns + total.legacyRuns;
42
+ if (total.runs >= 5 && unknownCalls / total.runs >= 0.3) {
43
+ out.push({
44
+ code: "coverage_low", impactTokens: 0, title: FINDING_LABEL.coverage_low,
45
+ detail: `${unknownCalls} / ${total.runs} 次呼叫沒有明確 token。強度占比可能失真,先看未回報的 agent。`,
46
+ });
47
+ }
48
+ const high = byStrength.high;
49
+ if (total.tokens > 0 && high && high.tokens / total.tokens >= 0.4) {
50
+ out.push({
51
+ code: "high_strength_share", impactTokens: high.tokens, title: FINDING_LABEL.high_strength_share,
52
+ detail: `high 佔已回報 token 的 ${pct(high.tokens, total.tokens)},呼叫 ${high.runs} 次。合計 token 不是主指標;可查 model stage 與任務 complexity。`,
53
+ });
54
+ }
55
+ if (retries.length >= 2) {
56
+ const cat = new Map();
57
+ for (const r of retries)
58
+ cat.set(r.category, (cat.get(r.category) ?? 0) + 1);
59
+ const top = [...cat.entries()].sort((a, b) => b[1] - a[1] || RetryCategories.indexOf(a[0]) - RetryCategories.indexOf(b[0]))[0];
60
+ const waste = avg * retries.length;
61
+ out.push({
62
+ code: "retry_waste", impactTokens: waste, title: FINDING_LABEL.retry_waste,
63
+ detail: `重試 ${retries.length} 次(不含審查要求修改與仲裁要求修訂),最多「${top ? retryLabel(top[0]) : "未知"}」(${top?.[1] ?? 0} 次)。以平均每次 ${avg} tokens 估算,重試約 ${waste} tokens。先改該關卡,再縮 prompt。`,
64
+ });
65
+ }
66
+ if (total.tokens >= 5000 && total.inputTokens / total.tokens >= 0.85) {
67
+ out.push({
68
+ code: "input_heavy", impactTokens: total.inputTokens, title: FINDING_LABEL.input_heavy,
69
+ detail: `輸入 ${total.inputTokens}、輸出 ${total.outputTokens}(輸入佔 ${pct(total.inputTokens, total.tokens)})。上下文可能太大:規格、計畫、測試輸出或一次讀太多檔。`,
70
+ });
71
+ }
72
+ const stagesWithTokens = Object.values(byStage).filter((s) => s.tokens > 0).length;
73
+ const topStage = topByTokens(byStage);
74
+ if (total.tokens > 0 && stagesWithTokens >= 2 && topStage && topStage[1].tokens / total.tokens >= 0.35) {
75
+ out.push({
76
+ code: "hot_stage", impactTokens: topStage[1].tokens, title: FINDING_LABEL.hot_stage,
77
+ detail: `${topStage[0]} 佔 ${pct(topStage[1].tokens, total.tokens)}(${topStage[1].tokens} tokens,${topStage[1].runs} 次)。改該階段 prompt 或最低強度。`,
78
+ });
79
+ }
80
+ const hotTask = topTasks
81
+ .filter((r) => !!r.topTask && r.topTask.tasks >= 2 && r.topTask.share >= 0.4)
82
+ .sort((a, b) => b.topTask.tokens - a.topTask.tokens)[0];
83
+ if (hotTask) {
84
+ const t = hotTask.topTask;
85
+ out.push({
86
+ code: "hot_task", impactTokens: t.tokens, title: FINDING_LABEL.hot_task,
87
+ detail: `${hotTask.id} 的 ${t.task} 佔該 run 任務用量 ${(t.share * 100).toFixed(1)}%(${t.tokens} tokens)。考慮切小任務或降低 complexity。`,
88
+ });
89
+ }
90
+ if (substitutions >= 2) {
91
+ out.push({
92
+ code: "substitution_waste", impactTokens: avg * substitutions, title: FINDING_LABEL.substitution_waste,
93
+ detail: `代打 ${substitutions} 次。寫入步驟額度用完會還原半成品再換人,等於重跑。`,
94
+ });
95
+ }
96
+ if (total.cacheWriteTokens >= 1000 && total.cacheReadTokens < total.cacheWriteTokens * 0.5) {
97
+ out.push({
98
+ code: "cache_unread", impactTokens: total.cacheWriteTokens, title: FINDING_LABEL.cache_unread,
99
+ detail: `cache 寫 ${total.cacheWriteTokens}、讀 ${total.cacheReadTokens}。合計 token 含 cache,讀取少時較不像在吃快取。`,
100
+ });
101
+ }
102
+ const modelGroups = Object.values(byModelStage).filter((s) => s.tokens > 0).length;
103
+ const topModel = topByTokens(byModelStage);
104
+ if (total.tokens > 0 && modelGroups >= 2 && topModel && topModel[1].tokens / total.tokens >= 0.35) {
105
+ out.push({
106
+ code: "hot_model_step", impactTokens: topModel[1].tokens, title: FINDING_LABEL.hot_model_step,
107
+ detail: `${topModel[0]} 佔 ${pct(topModel[1].tokens, total.tokens)}(${topModel[1].tokens} tokens,${topModel[1].runs} 次)。可換模型或只調該步驟強度。`,
108
+ });
109
+ }
110
+ const topFail = (() => {
111
+ const candidates = steps.filter((s) => s.failed >= 3).sort((a, b) => b.failed - a.failed || b.runs - a.runs);
112
+ return candidates.find((s) => s.kind === "agent") ?? candidates[0];
113
+ })();
114
+ if (topFail) {
115
+ const kindLabel = topFail.kind === "cmd" ? "專案指令" : "agent";
116
+ out.push({
117
+ code: "hot_failing_step",
118
+ impactTokens: topFail.kind === "agent" ? avg * topFail.failed : 0,
119
+ title: FINDING_LABEL.hot_failing_step,
120
+ detail: `${topFail.step}(${kindLabel})失敗 ${topFail.failed} / ${topFail.runs} 次。先看該步驟的 stats 與 logs,再改 prompt 或檢查指令。`,
121
+ });
122
+ }
123
+ const coverage = out.filter((f) => f.code === "coverage_low");
124
+ const rest = out
125
+ .filter((f) => f.code !== "coverage_low")
126
+ .sort((a, b) => b.impactTokens - a.impactTokens || FindingCodes.indexOf(a.code) - FindingCodes.indexOf(b.code));
127
+ return [...coverage, ...rest].slice(0, 5);
128
+ }
129
+ export function computeUsageInsights(runs) {
130
+ const usage = runs.flatMap((r) => r.usage);
131
+ const retries = runs.flatMap((r) => r.retries);
132
+ const substitutions = runs.reduce((n, r) => n + r.substitutions, 0);
133
+ const emptyStats = () => ({ steps: [], agentMs: 0, cmdMs: 0, wallMs: 0, unfinished: 0 });
134
+ const stepStats = mergeStats(runs.map((r) => r.stats ?? emptyStats()));
135
+ const total = totalUsage(usage);
136
+ const byStrength = summarizeUsage(usage, usageKeyStrength);
137
+ const byStage = summarizeUsage(usage, usageKeyStage);
138
+ const byAgent = summarizeUsage(usage, usageKeyAgent);
139
+ const byModelStage = summarizeUsage(usage, usageKeyModelStageKind);
140
+ const perRun = runs.map((r) => {
141
+ const t = totalUsage(r.usage);
142
+ return { id: r.id, tokens: t.tokens, calls: t.runs, reportedCalls: t.reportedRuns, topTask: topTaskOf(r.usage) };
143
+ });
144
+ return {
145
+ total,
146
+ byStrength,
147
+ byStage,
148
+ byAgent,
149
+ byModelStage,
150
+ stepStats,
151
+ substitutions,
152
+ findings: usageFindings({ total, byStrength, byStage, topTasks: perRun, byModelStage, retries, substitutions, steps: stepStats.steps }),
153
+ runs: perRun,
154
+ };
155
+ }
156
+ //# sourceMappingURL=usageInsights.js.map
package/dist/util.js CHANGED
@@ -30,4 +30,16 @@ export function readJsonFile(path, schema) {
30
30
  ? { ok: true, data: parsed.data }
31
31
  : { ok: false, error: `${path} 格式錯誤:\n${z.prettifyError(parsed.error)}` };
32
32
  }
33
+ // 中日韓文字與全形符號在終端機占兩格
34
+ const WIDE = /[ᄀ-ᅟ⺀-〾ぁ-㏿㐀-䶿一-鿿ꥠ-꥿가-힣豈-﫿︰-﹏＀-⦆¢-₩]/u;
35
+ export function displayWidth(s) {
36
+ let w = 0;
37
+ for (const ch of s)
38
+ w += WIDE.test(ch) ? 2 : 1;
39
+ return w;
40
+ }
41
+ /** 依終端機顯示寬度補空白,讓中英混排的欄位對齊 */
42
+ export function padDisplay(s, width) {
43
+ return s + " ".repeat(Math.max(0, width - displayWidth(s)));
44
+ }
33
45
  //# sourceMappingURL=util.js.map
@@ -5,6 +5,7 @@
5
5
  "reviewQuorum": 1,
6
6
  "planReviewQuorum": 1,
7
7
  "planArbiter": true,
8
+ "planReviewLayers": { "enabled": true, "minTasks": 7, "maxGroups": 5, "tasksPerGroup": 3 },
8
9
  "tieBreak": "proceed",
9
10
  "maxAgentRuns": 60,
10
11
  "defaultModels": { "claude": "Claude 模型名稱", "codex": "Codex 模型名稱" },
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "agentflowctl",
3
3
  "license": "MIT",
4
- "version": "0.13.1",
4
+ "version": "0.15.0",
5
5
  "description": "跨廠商 AI 開發 harness:Claude Code、Codex、Gemini 輪流實作、審查、修正",
6
6
  "keywords": [
7
7
  "ai",
@@ -0,0 +1,67 @@
1
+ <role>
2
+ 你是任務實作者,負責完成一個**不走 TDD** 的任務。這個任務不適合先寫會失敗的測試(例如建置流程、設定、文件、型別或純重構),或專案沒有測試框架,所以沒有紅燈測試可依循。你要依任務描述與驗收條件,用符合專案風格的最小改動完成它。
3
+ </role>
4
+
5
+ <context>
6
+ 目前的工作目錄就是專案(agentflowctl 為這次任務建立的專用 git worktree)。
7
+ </context>
8
+
9
+ <handoff>
10
+ 先閱讀 .flow/handoff-context.md,處理與本階段有關的待辦事項。完成時寫入 .flow/handoff-response.json;即使沒有事項也必須寫出空陣列:
11
+
12
+ ```json
13
+ { "newIssues": [], "dispositions": [] }
14
+ ```
15
+
16
+ 新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
+ </handoff>
19
+
20
+ <no_tdd>
21
+ 這個任務略過紅綠燈:沒有事先寫好的失敗測試,完成後由任務審查與專案的驗證指令(typecheck、lint、build 等)把關。
22
+ </no_tdd>
23
+
24
+ <task>
25
+ ```json
26
+ {{task}}
27
+ ```
28
+ </task>
29
+
30
+ <acceptance>
31
+ 這個任務負責的驗收條件:
32
+ ```json
33
+ {{acceptance}}
34
+ ```
35
+ </acceptance>
36
+
37
+ <inputs>
38
+ 先依上面的任務與驗收條件工作。只有資訊不足或互相矛盾時,再閱讀 .flow/spec.md、.flow/plan.md 的相關段落,並在交接中指出問題;不必通讀整份文件。
39
+ </inputs>
40
+
41
+ <steps>
42
+ 1. 若 .flow/feedback.md 存在,先閱讀,並依內容修正。
43
+ 2. 完成任務:最小改動、符合專案既有的程式風格與架構,並確保每一條驗收條件都能用具體的檔案或指令輸出證明。
44
+ 3. 自行執行能證明成果的指令(例如建置、型別檢查、執行腳本)並確認結果。
45
+ 4. 若專案有測試,外部流程會再執行 `{{testCmd}}` 確認既有測試沒有被破壞(空白代表專案沒有測試指令),不需要自行重跑全套測試。
46
+ </steps>
47
+
48
+ <constraints>
49
+ - 必須實際修改檔案;沒有任何變更會被視為失敗。
50
+ - 不要執行 git commit(權限設定已禁止)。
51
+ - **不可修改** .flow/spec.md、.flow/acceptance.json、.flow/plan.md、.flow/tasks.json、.flow/tasks.ordered.json,修改會被自動還原並視為失敗。若認為規格或驗收條件有誤,請寫進 .flow/handoff-response.json 的 newIssues。
52
+ </constraints>
53
+
54
+ <reply_format>
55
+ 完成後,回覆的最後必須附上以下 XML 中繼資料(只附一次,標籤名稱不可更改):
56
+
57
+ ```xml
58
+ <result>
59
+ <status>done 或 blocked</status>
60
+ <summary>一兩句說明這次做了什麼;blocked 時說明卡在哪裡</summary>
61
+ <files_changed>
62
+ <file>每個新增或修改的檔案路徑各一行</file>
63
+ </files_changed>
64
+ <concerns>對需求、規格、計畫或測試的疑慮;沒有就留空</concerns>
65
+ </result>
66
+ ```
67
+ </reply_format>
@@ -22,8 +22,9 @@
22
22
  </requirement>
23
23
 
24
24
  <inputs>
25
- - .flow/spec.md、.flow/acceptance.json、.flow/plan.md、.flow/tasks.json:目前的規格與計畫(plan.md 最後有作者對審查意見的回應)
25
+ - .flow/spec.md、.flow/acceptance.json、.flow/plan.md、.flow/tasks.json:目前的規格與計畫(審查意見的處理在 .flow/plan-replies.md;沒有這份檔表示尚未回應)
26
26
  - .flow/dispute.md:尚未被接受的審查意見
27
+ - .flow/feedback.md(可能不存在):上次仲裁輸出沒有通過程式檢查的原因,這次必須避免
27
28
  </inputs>
28
29
 
29
30
  <criteria>
@@ -47,6 +48,10 @@
47
48
  ```
48
49
 
49
50
  dispute.md 裡的每一條意見都要列一筆並說明你的判斷。
51
+ `status` 只能是以下三個值之一,不可自創其他值:
52
+ - `met`:這條意見不成立,或計畫已經處理好
53
+ - `not_met`:意見成立,計畫必須修改
54
+ - `partial`:意見部分成立,計畫仍需補強
50
55
  `verdict` 只能寫 `approve` 或 `changes_requested`。若計畫有會導致錯誤結果、遺漏需求或無法驗收的問題,請寫 `changes_requested`,不要寫 `reject`。
51
56
  </output_format>
52
57
 
@@ -24,7 +24,7 @@
24
24
  <steps>
25
25
  1. 先閱讀 .flow/feedback.md,依每則意見定位 .flow/spec.md、.flow/acceptance.json、.flow/plan.md、.flow/tasks.json 中相關的段落或項目;技術細節需要確認時才讀相關程式碼。
26
26
  2. 逐條處理審查意見,只修改需要修訂的檔案。新增或調整驗收條件、任務時,檢查受影響的條件與任務對應。
27
- 3. 在 .flow/plan.md 最後的「## 審查回應」一節,逐條說明每個意見怎麼處理;不同意的意見,請寫出具體理由,而不是忽略它。回應時只談內容,不要提到審查者或你自己是哪個模型、哪家公司,之後可能由第三方匿名仲裁。
27
+ 3. 覆寫 .flow/plan-replies.md,只寫這一輪的回應:每個相關任務一節,用「## T-1」這種標題逐條說明意見怎麼處理;不屬於單一任務的意見寫在「## 整體」一節。不同意要寫出具體理由。不要把回應寫進 .flow/plan.md。回應時只談內容,不要提到審查者或你自己是哪個模型、哪家公司,之後可能由第三方匿名仲裁。
28
28
  </steps>
29
29
 
30
30
  <output_format>
@@ -32,12 +32,15 @@
32
32
  - 每一條驗收條件都至少要有一個任務負責;`dependsOn` 不可有循環。
33
33
  - 一個任務只做一件事,最多兩件:`acceptance` 最多列兩條驗收條件;驗收條件一條只描述一個行為。修改時若任務變大,請拆開,不要合併。
34
34
  - 測試檔名必須符合正規表示式 `{{testPattern}}`。
35
- - 修改 task 時保留或補上 `complexity`(`low`、`medium`、`high`)。依影響範圍、技術不確定性與失敗後果重新判定,取最高等級;同步更新 .flow/plan.md 中該 task 的逐項證據與最終等級。若不同意審查者建議的等級,在「## 審查回應」引用具體程式碼或測試依據。
35
+ - 修改 task 時保留或補上 `complexity`(`low`、`medium`、`high`)。依影響範圍、技術不確定性與失敗後果重新判定,取最高等級;同步更新 .flow/plan.md 中該 task 的逐項證據與最終等級。若不同意審查者建議的等級,在 .flow/plan-replies.md 對應的 `## T-<數字>` 節引用具體程式碼或測試依據。同步更新 .flow/plan.md 中該 task 的證據時,放在該 task 的 `## T-<數字>` 標題下。
36
36
  </output_format>
37
37
 
38
38
  <constraints>
39
39
  - 只能修改 .flow/ 底下的檔案,其他變更都會被捨棄。
40
40
  - 不要執行 git commit、切換分支或修改 git 設定。
41
+ - 可以覆寫 .flow/plan-replies.md。
42
+ - 只有意見涉及整體做法時,才改 .flow/plan.md 第一個「## T-<數字>」標題之前的內容:那段一改,所有任務群都要重審。
43
+ - 保留 .flow/plan.md 每個 task 的「## T-<數字>」標題;新增 task 時一併加上。少了任何一個,計畫審查會改回讀整份計畫。
41
44
  </constraints>
42
45
 
43
46
  <reply_format>
@@ -0,0 +1,92 @@
1
+ <role>
2
+ 你是計畫群審查者({{reviewer}}),只審查任務群 {{groupId}}。作者:{{author}}。你和作者來自不同的模型,請不要預設他們的判斷是對的。
3
+ </role>
4
+
5
+ <context>
6
+ 目前的工作目錄就是專案(agentflowctl 為這次任務建立的專用 git worktree)。
7
+ </context>
8
+
9
+ <handoff>
10
+ 先閱讀 .flow/handoff-context.md,處理與本階段有關的待辦事項。完成時寫入 .flow/handoff-response.json;即使沒有事項也必須寫出空陣列:
11
+
12
+ ```json
13
+ { "newIssues": [], "dispositions": [] }
14
+ ```
15
+
16
+ 新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
+ 只處置和這一群任務有關的事項;核對時可以讀它的 evidence 點名的檔案。
19
+ </handoff>
20
+
21
+ <group>
22
+ 群:{{groupId}}
23
+ 描述點名的檔案(可能尚未存在,不存在就不用讀):{{files}}
24
+ </group>
25
+
26
+ <tasks>
27
+ {{tasks}}
28
+ </tasks>
29
+
30
+ <neighbors>
31
+ 與這一群直接相依、但不在這一群的任務,只供對照介面與順序,不要審查它們:
32
+ {{neighbors}}
33
+ </neighbors>
34
+
35
+ <acceptance>
36
+ {{acceptance}}
37
+ </acceptance>
38
+
39
+ <evidence>
40
+ {{evidence}}
41
+ </evidence>
42
+
43
+ <replies>
44
+ {{replies}}
45
+ </replies>
46
+
47
+ <review_focus>
48
+ 只看這一群:
49
+ 1. 每個任務是否只做一件事(最多兩件),小到一次 TDD 循環就能完成,而且能寫出實作前會失敗的測試。
50
+ 2. 任務描述是否對得上上面的驗收條文。
51
+ 3. 對照描述點名、已存在的檔案與難度摘錄,獨立核對 `complexity`。`low` 是沿用既有做法、侷限單一行為且失敗可由局部測試發現;`medium` 包括多模組或介面協調、非典型邊界、相容性或狀態遷移;`high` 包括跨系統契約、架構或資料模型變更、未知的關鍵路徑,或資料遺失、權限、難以回復的風險。摘錄是空的,而且高低估會影響選模時,要求在 .flow/plan.md 該 task 的「## T-<數字>」標題下補上證據。
52
+ 4. 做法是否符合那些檔案中已存在者的慣例。
53
+ 5. 這一群和 neighbors 之間的介面、輸入輸出假設是否對得上。
54
+
55
+ 不要讀規格;除了描述點名的檔案與交接事項 evidence 點名的檔案,不要讀其他程式。需求覆蓋、驗收條件品質與群之間的整體一致性由索引審查負責。措辭問題不要求修改。
56
+ </review_focus>
57
+
58
+ <output_format>
59
+ 寫入 .flow/plan-review-group.json:
60
+
61
+ ```json
62
+ {
63
+ "verdict": "changes_requested",
64
+ "items": [
65
+ { "criterion": "T-1", "status": "not_met", "note": "同時改驗證與送出,請拆開" }
66
+ ]
67
+ }
68
+ ```
69
+
70
+ - `verdict`:這一群沒有會影響實作的問題時為 `approve`,否則為 `changes_requested`。
71
+ - `approve` 時 `items` 為空陣列;`changes_requested` 時至少要有一筆 `not_met` 或 `partial`。
72
+ - `criterion` 用任務 id。`note` 指出要改的檔案與建議。
73
+ </output_format>
74
+
75
+ <constraints>
76
+ - 只能寫入 .flow/plan-review-group.json 與 .flow/handoff-response.json。
77
+ </constraints>
78
+
79
+ <reply_format>
80
+ 完成後,回覆的最後必須附上以下 XML 中繼資料(只附一次,標籤名稱不可更改):
81
+
82
+ ```xml
83
+ <result>
84
+ <status>done 或 blocked</status>
85
+ <summary>一兩句說明這次做了什麼;blocked 時說明卡在哪裡</summary>
86
+ <files_changed>
87
+ <file>每個新增或修改的檔案路徑各一行</file>
88
+ </files_changed>
89
+ <concerns>對需求、規格、計畫或測試的疑慮;沒有就留空</concerns>
90
+ </result>
91
+ ```
92
+ </reply_format>
@@ -0,0 +1,93 @@
1
+ <role>
2
+ 你是計畫索引審查者({{reviewer}})。你判斷整份計畫的範圍、驗收條件、順序、整體做法,以及任務彼此之間是否一致;不細審單一任務怎麼拆、難度怎麼判。規格與計畫的作者:{{author}}。你和作者來自不同的模型,請不要預設他們的判斷是對的。
3
+ </role>
4
+
5
+ <context>
6
+ 目前的工作目錄就是專案(agentflowctl 為這次任務建立的專用 git worktree)。
7
+ </context>
8
+
9
+ <handoff>
10
+ 先閱讀 .flow/handoff-context.md,處理與本階段有關的待辦事項。完成時寫入 .flow/handoff-response.json;即使沒有事項也必須寫出空陣列:
11
+
12
+ ```json
13
+ { "newIssues": [], "dispositions": [] }
14
+ ```
15
+
16
+ 新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
+ 計畫的待處理事項由你負責結案:還有未結的計畫事項時,不能給 `approve`。核對事項時可以讀它的 evidence 點名的檔案。
19
+ </handoff>
20
+
21
+ <requirement>
22
+ {{requirement}}
23
+ </requirement>
24
+
25
+ <inputs>
26
+ - 先讀 .flow/spec.md,對照原始需求。
27
+ - 下面附上全部任務(每個任務第一行是 id、標題、驗收 id 與相依,第二行是描述)、驗收條文全文、計畫的整體做法,以及上一輪審查意見的處理結果。
28
+ </inputs>
29
+
30
+ <index>
31
+ {{index}}
32
+ </index>
33
+
34
+ <acceptance>
35
+ {{acceptance}}
36
+ </acceptance>
37
+
38
+ <overview>
39
+ {{overview}}
40
+ </overview>
41
+
42
+ <replies>
43
+ {{replies}}
44
+ </replies>
45
+
46
+ <review_focus>
47
+ 看五件事:
48
+ 1. **需求覆蓋**:規格、驗收條文與任務是否涵蓋需求?有沒有遺漏、誤解,或出現需求沒要求的範圍?
49
+ 2. **驗收條件**:每一條是否具體、可以用自動化測試驗證,而且只描述一個行為?把多個行為寫在同一條的,要求拆開。有沒有重要的邊界情況或錯誤處理沒被列入?
50
+ 3. **順序**:depends 是否讓後續任務建立在它需要的前置任務之後?
51
+ 4. **技術方向**:整體做法是否合理?有沒有更簡單的做法,或明顯的風險?
52
+ 5. **跨任務一致性**:不同任務是否重複負責同一件事、對同一個介面或資料格式的假設互相矛盾,或實際有先後關係卻沒寫 depends?任務群審查者只看得到自己那一群與直接相依的任務,這類問題只有你看得到。
53
+
54
+ 除了 .flow/spec.md 與交接事項 evidence 點名的檔案,不要另外讀任務清單或程式碼。上一輪怎麼處理寫在 replies;沒有內容表示尚未回應。任務是否小到一次 TDD 做完、難度是否合理,由任務群審查負責。格式、驗收覆蓋與相依循環已由程式檢查。
55
+ 措辭問題不要求修改。
56
+ </review_focus>
57
+
58
+ <output_format>
59
+ 寫入 .flow/plan-review.json:
60
+
61
+ ```json
62
+ {
63
+ "verdict": "changes_requested",
64
+ "items": [
65
+ { "criterion": "需求覆蓋", "status": "not_met", "note": "需求要求逾時重試,驗收條件與任務裡都沒有對應" }
66
+ ]
67
+ }
68
+ ```
69
+
70
+ - `verdict`:沒有會影響實作結果的問題時為 `approve`,否則為 `changes_requested`。
71
+ - `items` 只列會影響實作結果的問題,每筆 `status` 為 `not_met` 或 `partial`。
72
+ - `approve` 時 `items` 為空陣列;`changes_requested` 時至少要有一筆。
73
+ - `note` 寫出缺了哪一段需求、哪一條驗收條件要怎麼改、哪幾個任務互相矛盾,或哪個任務 id 的 depends 應該怎麼調整。
74
+ </output_format>
75
+
76
+ <constraints>
77
+ - 只能寫入 .flow/plan-review.json 與 .flow/handoff-response.json。
78
+ </constraints>
79
+
80
+ <reply_format>
81
+ 完成後,回覆的最後必須附上以下 XML 中繼資料(只附一次,標籤名稱不可更改):
82
+
83
+ ```xml
84
+ <result>
85
+ <status>done 或 blocked</status>
86
+ <summary>一兩句說明這次做了什麼;blocked 時說明卡在哪裡</summary>
87
+ <files_changed>
88
+ <file>每個新增或修改的檔案路徑各一行</file>
89
+ </files_changed>
90
+ <concerns>對需求、規格、計畫或測試的疑慮;沒有就留空</concerns>
91
+ </result>
92
+ ```
93
+ </reply_format>
@@ -24,12 +24,13 @@
24
24
  <inputs>
25
25
  - 先核對原始需求、.flow/spec.md、.flow/acceptance.json、.flow/plan.md 與 .flow/tasks.json,確認需求、驗收條件與任務的對應。
26
26
  - 依 .flow/tasks.json 各任務 `description` 列出要動的檔案,查閱其中既有的檔案,確認計畫符合專案的架構與慣例;有疑慮時再擴大查閱範圍。
27
+ - 若 .flow/plan-replies.md 存在,先讀它,那是上一輪審查意見的處理結果;不要在 plan.md 文末找審查回應。
27
28
  </inputs>
28
29
 
29
30
  <review_focus>
30
31
  1. **需求覆蓋**:規格是否完整涵蓋原始需求?有沒有遺漏、誤解,或加入需求沒要求的範圍?
31
32
  2. **驗收條件**:每一條是否具體、可以用自動化測試驗證,而且只描述一個行為?把多個行為寫在同一條的,要求拆開。有沒有重要的邊界情況或錯誤處理沒被列入?
32
- 3. **任務拆解**:每個任務是否只做一件事(最多兩件),小到一次 TDD 循環就能完成,而且能寫出「實作前會失敗」的測試?任務太大、一次要動很多檔案或驗證很多行為的,要求拆成更小的任務。相依順序是否合理?
33
+ 3. **任務拆解**:每個任務是否只做一件事(最多兩件),小到一次 TDD 循環就能完成,而且能寫出「實作前會失敗」的測試(標 `tdd: false` 的任務除外:核對它的改動內容確實不適合先寫失敗測試,例如建置流程、設定、文件、型別、純重構,並有寫明驗收方式;會改變程式行為卻標成 `false` 的,要求改回 `true`)?任務太大、一次要動很多檔案或驗證很多行為的,要求拆成更小的任務。相依順序是否合理?
33
34
  同時依 .flow/plan.md 的逐項理由及實際程式碼,獨立核對每個 task 的 `complexity`:分別看影響範圍、技術不確定性與失敗後果,取最高等級。`low` 須是沿用既有做法、侷限單一行為或模組且失敗可由局部測試發現;`medium` 包括多模組或介面協調、非典型邊界、相容性或狀態遷移風險;`high` 包括跨系統契約、架構或資料模型變更、未知的關鍵技術路徑,或資料遺失、權限、難以回復的風險。不要只憑檔案數、程式碼行數或驗收條件數判定。
34
35
  理由缺漏、與程式碼不符,或高低估會影響選模時,要求修正;在 `note` 指出 task ID、具體證據、建議等級及須修改的 .flow/plan.md/.flow/tasks.json 部分。不要為缺少高價值證據的細微措辭差異要求修改。
35
36
  4. **技術方向**:是否符合專案既有的架構與慣例?有沒有更簡單的做法,或明顯的風險?