agentflowctl 0.10.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,74 @@
1
+ import { mkdtempSync, rmSync } from "node:fs";
2
+ import { tmpdir } from "node:os";
3
+ import { join } from "node:path";
4
+ import { ADAPTERS } from "./agents/index.js";
5
+ import { exec } from "./proc.js";
6
+ export function probeFailureReason(message) {
7
+ if (/\b429\b|quota|rate.?limit|usage limit/i.test(message))
8
+ return `額度或速率限制:${message}`;
9
+ if (/invalid model|model.*(not found|unknown|unavailable)|unknown model/i.test(message))
10
+ return `模型名稱無效或目前不可用:${message}`;
11
+ if (/auth|unauthori[sz]ed|forbidden|\b401\b|\b403\b/i.test(message))
12
+ return `認證或權限失敗:${message}`;
13
+ if (/network|connect|timed? out|dns|enotfound/i.test(message))
14
+ return `網路連線失敗:${message}`;
15
+ return message;
16
+ }
17
+ /** 在獨立暫存目錄對指定模型送出最短請求;安全能力不足時直接拒絕。 */
18
+ export async function probeModel(def, model, timeoutMs = 30000) {
19
+ const dir = mkdtempSync(join(tmpdir(), "agentflowctl-model-"));
20
+ try {
21
+ const adapter = ADAPTERS[def.adapter];
22
+ let inv;
23
+ if (def.adapter === "command") {
24
+ if (!def.modelProbe?.length || !def.modelProbe.some((arg) => arg.includes("{model}"))) {
25
+ return { status: "unverifiable", reason: "command agent 缺少含 {model} 的 modelProbe" };
26
+ }
27
+ const [cmd, ...rest] = def.modelProbe;
28
+ inv = { cmd: cmd.replaceAll("{model}", model), args: rest.map((arg) => arg.replaceAll("{model}", model)) };
29
+ }
30
+ else {
31
+ if (!adapter.invokeModelProbe)
32
+ return { status: "unverifiable", reason: `${def.adapter} CLI 無法安全停用工具,因此不能驗證模型` };
33
+ inv = adapter.invokeModelProbe(model, dir);
34
+ }
35
+ const events = [];
36
+ const result = await exec(inv.cmd, inv.args, {
37
+ cwd: dir, env: inv.env, input: inv.input, timeoutMs,
38
+ onStdoutLine: (line) => { if (def.adapter !== "command")
39
+ events.push(...adapter.parse(line)); },
40
+ });
41
+ if (result.code !== 0)
42
+ return { status: "failed", reason: result.code === 124 ? "模型檢查逾時" : probeFailureReason(result.stderr.trim() || result.stdout.trim() || `結束碼 ${result.code}`) };
43
+ if (def.adapter === "command") {
44
+ try {
45
+ const data = JSON.parse(result.stdout.trim());
46
+ if (data.requestedModel !== model || typeof data.resolvedModel !== "string" || !data.resolvedModel.trim()) {
47
+ return { status: "failed", reason: "自訂探測命令回報的模型名稱不符" };
48
+ }
49
+ return { status: "ok", resolvedModel: data.resolvedModel };
50
+ }
51
+ catch {
52
+ return { status: "failed", reason: "自訂探測命令未輸出有效 JSON" };
53
+ }
54
+ }
55
+ if (events.some((e) => e.kind === "tool"))
56
+ return { status: "failed", reason: "模型檢查期間出現工具呼叫,未通過無工具驗證" };
57
+ const done = events.filter((e) => e.kind === "done").at(-1);
58
+ if (done?.kind !== "done")
59
+ return { status: "failed", reason: "CLI 沒有回報完成" };
60
+ if (!done.ok)
61
+ return { status: "failed", reason: done.summary ?? "CLI 回報失敗" };
62
+ if (!events.some((e) => e.kind === "text" && e.text.trim()))
63
+ return { status: "failed", reason: "CLI 沒有回傳文字內容" };
64
+ const resolved = events.find((e) => e.kind === "model");
65
+ return { status: "ok", resolvedModel: resolved?.kind === "model" ? resolved.id : undefined };
66
+ }
67
+ catch (error) {
68
+ return { status: "failed", reason: error.message };
69
+ }
70
+ finally {
71
+ rmSync(dir, { recursive: true, force: true });
72
+ }
73
+ }
74
+ //# sourceMappingURL=modelProbe.js.map
@@ -0,0 +1,107 @@
1
+ const LEVEL = { low: 0, medium: 1, high: 2 };
2
+ const STRENGTHS = ["low", "medium", "high"];
3
+ export const DEFAULT_STAGE_STRENGTH = {
4
+ spec: "medium", plan: "high", planReview: "high", planFix: "medium", planArbiter: "high",
5
+ taskTests: "low", taskCode: "low", taskReview: "medium", taskFix: "low", fix: "medium", review: "high",
6
+ };
7
+ export function effectiveStageStrengths(cfg) {
8
+ const custom = cfg.modelSelection.stageStrength;
9
+ return Object.fromEntries(Object.keys(DEFAULT_STAGE_STRENGTH).map((stage) => [stage, { strength: custom[stage] ?? DEFAULT_STAGE_STRENGTH[stage], custom: custom[stage] !== undefined }]));
10
+ }
11
+ /** 小組審查失敗只影響該審查者的下一次選模。 */
12
+ export function recordModelReviewFailure(run, step, reviewer) {
13
+ const key = `${step}:${reviewer}`;
14
+ return { ...run, modelRetryAttempts: { ...run.modelRetryAttempts, [key]: (run.modelRetryAttempts?.[key] ?? 0) + 1 } };
15
+ }
16
+ export function clearModelReviewFailure(run, step, reviewer) {
17
+ const attempts = { ...run.modelRetryAttempts };
18
+ delete attempts[`${step}:${reviewer}`];
19
+ return { ...run, modelRetryAttempts: attempts };
20
+ }
21
+ export function clearModelReviewStage(run, step) {
22
+ const attempts = Object.fromEntries(Object.entries(run.modelRetryAttempts ?? {}).filter(([key]) => !key.startsWith(`${step}:`)));
23
+ return { ...run, modelRetryAttempts: attempts };
24
+ }
25
+ /** 把步驟名稱(例如 `T-1-code`、`plan-review`)對應到階段鍵;不認得時回傳 undefined。 */
26
+ export function stageOfStep(step) {
27
+ const task = /^T-\d+-(tests|code|review|fix)$/.exec(step);
28
+ if (task)
29
+ return { tests: "taskTests", code: "taskCode", review: "taskReview", fix: "taskFix" }[task[1]];
30
+ const stage = {
31
+ spec: "spec", plan: "plan", "plan-review": "planReview", "plan-fix": "planFix",
32
+ "plan-arbiter": "planArbiter", fix: "fix", review: "review",
33
+ };
34
+ return stage[step];
35
+ }
36
+ function stepStage(step) {
37
+ const found = stageOfStep(step);
38
+ if (!found)
39
+ throw new Error(`未知的 LLM 步驟:${step}`);
40
+ return found;
41
+ }
42
+ function escalation(run, stage, step, reviewer) {
43
+ const a = run.attempts;
44
+ if (stage === "planReview" || stage === "review")
45
+ return run.modelRetryAttempts?.[`${step}:${reviewer ?? ""}`] ?? 0;
46
+ if (stage === "planArbiter")
47
+ return 0;
48
+ if (stage === "planFix")
49
+ return (a["plan-fix"] ?? 0) + Math.max((a["plan-review"] ?? 0) + (a["plan-handoff"] ?? 0) - 1, 0);
50
+ if (stage === "taskFix") {
51
+ const id = step.slice(0, -4);
52
+ return (a[`${id}:fix`] ?? 0) + Math.max((a[`${id}:review`] ?? 0) + (a[`${id}:verify`] ?? 0) - 1, 0);
53
+ }
54
+ if (stage === "fix")
55
+ return (a.fix ?? 0) + Math.max((a.verify ?? 0) + (a.review ?? 0) - 1, 0);
56
+ if (stage === "taskTests" || stage === "taskCode" || stage === "taskReview") {
57
+ const id = step.slice(0, step.lastIndexOf("-"));
58
+ const suffix = step.slice(step.lastIndexOf("-") + 1);
59
+ return a[`${id}:${suffix === "review" ? "review-run" : suffix}`] ?? 0;
60
+ }
61
+ return a[step] ?? 0;
62
+ }
63
+ /** 每次呼叫前依目前設定與 run 狀態選擇模型;不修改任何狀態。 */
64
+ export function selectModel(run, cfg, agent, step, complexity, reviewer) {
65
+ const def = cfg.agents[agent];
66
+ if (!def)
67
+ throw new Error(`未定義的 agent:${agent}`);
68
+ const mode = run.modelMode ?? "balanced";
69
+ if (mode === "balanced") {
70
+ const fallback = def.adapter === "command" ? undefined : cfg.defaultModels[def.adapter];
71
+ return { mode, name: def.model ?? fallback, insufficient: false };
72
+ }
73
+ if (!def.models?.length)
74
+ throw new Error(`agent ${agent} 沒有設定 models`);
75
+ const stage = stepStage(step);
76
+ const floor = effectiveStageStrengths(cfg)[stage].strength;
77
+ const baseline = stage.startsWith("task") ? Math.max(LEVEL[floor], LEVEL[complexity ?? "medium"]) : LEVEL[floor];
78
+ const targetStrength = STRENGTHS[Math.min(2, baseline + escalation(run, stage, step, reviewer))];
79
+ const candidates = def.models.filter((m) => LEVEL[m.strength] >= LEVEL[targetStrength]);
80
+ const min = Math.min(...candidates.map((m) => LEVEL[m.strength]));
81
+ const chosen = candidates.find((m) => LEVEL[m.strength] === min)
82
+ ?? def.models.reduce((best, current) => LEVEL[current.strength] > LEVEL[best.strength] ? current : best);
83
+ return { mode, name: chosen.name, strength: chosen.strength, targetStrength, insufficient: LEVEL[chosen.strength] < LEVEL[targetStrength] };
84
+ }
85
+ /** 啟用 adaptive 前檢查本次實際參與的 agent。 */
86
+ export function validateAdaptiveConfig(cfg, cycle) {
87
+ for (const name of cycle) {
88
+ const def = cfg.agents[name];
89
+ if (!def)
90
+ throw new Error(`未定義的 agent:${name}`);
91
+ if (!def.models?.length)
92
+ throw new Error(`agent ${name} 的 models 至少要有一個模型`);
93
+ const names = def.models.map((m) => m.name);
94
+ if (new Set(names).size !== names.length)
95
+ throw new Error(`agent ${name} 的 models 有重複名稱`);
96
+ if (def.adapter === "command") {
97
+ if (!def.command?.some((arg) => arg.includes("{model}")))
98
+ throw new Error(`agent ${name} 的 command 缺少 {model}`);
99
+ if (!def.modelProbe?.length || !def.modelProbe.some((arg) => arg.includes("{model}")))
100
+ throw new Error(`agent ${name} 的 modelProbe 缺少 {model}`);
101
+ }
102
+ else if (def.extraArgs.some((arg) => arg === "--model" || arg === "-m" || arg.startsWith("--model="))) {
103
+ throw new Error(`agent ${name} 的 extraArgs 與 adaptive 模型參數衝突`);
104
+ }
105
+ }
106
+ }
107
+ //# sourceMappingURL=modelSelection.js.map
package/dist/proc.js CHANGED
@@ -6,6 +6,8 @@ export function exec(cmd, args, opts = {}) {
6
6
  env: { ...process.env, ...opts.env },
7
7
  shell: opts.shell ?? false,
8
8
  });
9
+ let timedOut = false;
10
+ const timer = opts.timeoutMs ? setTimeout(() => { timedOut = true; child.kill("SIGKILL"); }, opts.timeoutMs) : undefined;
9
11
  let stdout = "";
10
12
  let stderr = "";
11
13
  let pending = "";
@@ -22,11 +24,14 @@ export function exec(cmd, args, opts = {}) {
22
24
  child.stderr.on("data", (d) => {
23
25
  stderr += d.toString();
24
26
  });
25
- child.on("error", reject);
27
+ child.on("error", (error) => { if (timer)
28
+ clearTimeout(timer); reject(error); });
26
29
  child.on("close", (code) => {
30
+ if (timer)
31
+ clearTimeout(timer);
27
32
  if (opts.onStdoutLine && pending)
28
33
  opts.onStdoutLine(pending);
29
- resolve({ code: code ?? 1, stdout, stderr });
34
+ resolve({ code: timedOut ? 124 : code ?? 1, stdout, stderr: timedOut ? `${stderr}\n執行逾時` : stderr });
30
35
  });
31
36
  child.stdin.on("error", () => { }); // 子程序不讀 stdin 時忽略 EPIPE
32
37
  child.stdin.end(opts.input ?? "");
package/dist/roles.js CHANGED
@@ -1,10 +1,11 @@
1
1
  /**
2
2
  * 決定每個步驟由哪個 agent 執行。
3
3
  *
4
- * 規則只有三條:
4
+ * 規則只有四條:
5
5
  * 1. 審查者不能是最後寫程式的 agent,多位審查者彼此不重複。
6
6
  * 2. 開啟 tddSplit 時,同一個任務的測試與實作由不同 agent 負責。
7
7
  * 3. 人選隨機決定,不依 cycle 的順序。cycle 只代表有哪些 agent 參與。
8
+ * 4. 任務的測試、實作、任務審查依同一個隨機順序輪流交換,讓各家用量平均。
8
9
  *
9
10
  * 隨機以 seed(run id 加上步驟與輪次)決定:同一個 run 的同一步驟重算時會得到同樣的人,
10
11
  * 所以 resume、或同一步驟在不同地方重算時結果一致。
@@ -48,28 +49,30 @@ export function pick(cycle, seed, exclude = []) {
48
49
  export const specAgent = (cycle, seed) => pick(cycle, `${seed}:author`);
49
50
  export const planAgent = specAgent;
50
51
  /**
51
- * 每 cycle.length 個任務為一輪,每輪洗一次牌,所以每家輪完一次才會再輪到;
52
- * 換輪時若新一輪第一位和上一輪最後一位相同,就把它往後移,避免同一家連續撰寫測試。
52
+ * 任務的角色輪流交換,讓各家用量平均:整個 run 用同一個洗好的順序,
53
+ * 第 i 個任務由 order[i] 寫測試、order[i+1] 寫實作、order[i+2] 做任務審查(都取餘數)。
54
+ * 三家時每三個任務各家剛好把三種角色各做一次;兩家時測試與實作每個任務互換。
53
55
  */
54
- function taskBag(cycle, seed, bag) {
55
- const order = shuffled(cycle, `${seed}:tasks:${bag}`);
56
- if (bag === 0 || order.length < 2)
57
- return order;
58
- const prevLast = taskBag(cycle, seed, bag - 1).at(-1);
59
- return order[0] === prevLast ? [...order.slice(1), order[0]] : order;
60
- }
61
- /** 第 i 個任務的測試作者從洗好的牌依序取;實作者隨機挑一位測試作者以外的 agent */
62
56
  export function taskAgents(cycle, taskIndex, tddSplit, seed) {
63
- const n = cycle.length;
64
- const tests = taskBag(cycle, seed, Math.floor(taskIndex / n))[taskIndex % n];
65
- return { tests, code: tddSplit ? pick(cycle, `${seed}:code:${taskIndex}`, [tests]) : tests };
57
+ const order = shuffled(cycle, `${seed}:tasks`);
58
+ const at = (offset) => order[(taskIndex + offset) % order.length];
59
+ const tests = at(0);
60
+ const code = tddSplit ? at(1) : tests;
61
+ // 兩家時沒有第三方,任務審查只能由實作者以外的測試作者負責
62
+ const review = order.length >= 3 ? at(tddSplit ? 2 : 1) : order.find((c) => c !== code) ?? code;
63
+ return { tests, code, review };
66
64
  }
67
- /** 隨機挑出 quorum 位彼此不重複、而且不是最後作者的 reviewer;任務審查優先避開測試作者 */
68
- export function reviewers(cycle, lastWriter, quorum, seed, testAuthor) {
65
+ /**
66
+ * 隨機挑出 quorum 位彼此不重複、而且不是最後作者的 reviewer;任務審查優先避開測試作者。
67
+ * prefer 是輪到的審查者:只要不是最後作者就排第一位(任務修正後重審仍由同一位審查)。
68
+ */
69
+ export function reviewers(cycle, lastWriter, quorum, seed, testAuthor, prefer) {
69
70
  const candidates = shuffled(cycle.filter((c) => c !== lastWriter), seed);
70
71
  const preferred = testAuthor ? candidates.filter((c) => c !== testAuthor) : candidates;
71
72
  const fallback = testAuthor ? candidates.filter((c) => c === testAuthor) : [];
72
- const picked = [...preferred, ...fallback].slice(0, quorum);
73
+ const ranked = [...preferred, ...fallback];
74
+ const first = prefer && ranked.includes(prefer) ? [prefer] : [];
75
+ const picked = [...first, ...ranked.filter((c) => c !== prefer)].slice(0, quorum);
73
76
  return picked.length ? picked : [lastWriter ?? cycle[0]];
74
77
  }
75
78
  /**
package/dist/runner.js CHANGED
@@ -63,11 +63,14 @@ export async function runAgent(name, def, t, prompt) {
63
63
  projectRoot: projectRoot(),
64
64
  command: def.command,
65
65
  });
66
- appendLog(t.logFile, headerLine({ stage: t.stage, step: t.step, agent: name, adapter: def.adapter, startedAt: new Date().toISOString() }));
66
+ appendLog(t.logFile, headerLine({ stage: t.stage, step: t.step, agent: name, adapter: def.adapter, model: def.model, strength: t.strength, targetStrength: t.targetStrength, startedAt: new Date().toISOString() }));
67
67
  let done;
68
68
  let lastText = "";
69
69
  let inputTokens = 0;
70
70
  let outputTokens = 0;
71
+ let inputReported = false;
72
+ let outputReported = false;
73
+ let resolvedModel;
71
74
  const r = await exec(inv.cmd, inv.args, {
72
75
  cwd: t.cwd,
73
76
  env: inv.env,
@@ -87,8 +90,17 @@ export async function runAgent(name, def, t, prompt) {
87
90
  console.log(formatToolLine(name, ev));
88
91
  }
89
92
  else if (ev.kind === "usage") {
90
- inputTokens += ev.inputTokens ?? 0;
91
- outputTokens += ev.outputTokens ?? 0;
93
+ if (ev.inputTokens !== undefined) {
94
+ inputReported = true;
95
+ inputTokens += ev.inputTokens;
96
+ }
97
+ if (ev.outputTokens !== undefined) {
98
+ outputReported = true;
99
+ outputTokens += ev.outputTokens;
100
+ }
101
+ }
102
+ else if (ev.kind === "model") {
103
+ resolvedModel = ev.id;
92
104
  }
93
105
  else if (ev.kind === "done") {
94
106
  done = { ok: ev.ok, summary: ev.summary };
@@ -105,7 +117,8 @@ export async function runAgent(name, def, t, prompt) {
105
117
  const summary = done?.summary || lastText || tail(r.stdout, 2000) || tail(r.stderr, 2000);
106
118
  const quotaExhausted = !ok && isQuotaError(`${summary}\n${done?.summary ?? ""}\n${r.stderr}\n${tail(r.stdout, 4000)}`);
107
119
  const meta = parseResultMeta(summary) ?? parseResultMeta(lastText);
108
- return { ok, quotaExhausted, summary, meta, inputTokens, outputTokens };
120
+ return { ok, quotaExhausted, summary, meta, usageReported: inputReported && outputReported,
121
+ inputTokens: inputReported ? inputTokens : undefined, outputTokens: outputReported ? outputTokens : undefined, resolvedModel };
109
122
  }
110
123
  /**
111
124
  * 在 worktree 內執行專案指令(安裝、測試、建置)。
package/dist/schemas.js CHANGED
@@ -71,6 +71,7 @@ export const TaskItem = z.object({
71
71
  description: z.string().min(1),
72
72
  dependsOn: z.array(z.string()).default([]),
73
73
  acceptance: z.array(z.string()).min(1, "每個任務至少要對應一條驗收條件"),
74
+ complexity: z.enum(["low", "medium", "high"]).optional(),
74
75
  });
75
76
  /** Agent 在 plan 階段產出的 .flow/tasks.json */
76
77
  export const TaskList = z.array(TaskItem).min(1);
@@ -97,12 +98,17 @@ export const ArbiterResult = ReviewResult.extend({
97
98
  .transform((verdict) => verdict === "reject" ? "changes_requested" : verdict),
98
99
  });
99
100
  /** 一個 agent 的定義;名稱(agents 的 key)用在 cycle 裡 */
101
+ export const ModelStrength = z.enum(["low", "medium", "high"]);
102
+ export const ModelStage = z.enum(["spec", "plan", "planReview", "planFix", "planArbiter", "taskTests", "taskCode", "taskReview", "taskFix", "fix", "review"]);
103
+ export const ModelEntry = z.object({ name: z.string().trim().min(1), strength: ModelStrength });
100
104
  export const AgentDef = z.object({
101
105
  adapter: z.enum(["claude", "codex", "gemini", "command"]),
102
106
  model: z.string().optional(),
107
+ models: z.array(ModelEntry).optional(),
103
108
  extraArgs: z.array(z.string()).default([]),
104
109
  /** 只有 command adapter 使用,`{prompt}` 會被替換成 prompt */
105
110
  command: z.array(z.string()).optional(),
111
+ modelProbe: z.array(z.string()).optional(),
106
112
  });
107
113
  /** 目標專案可選的 flow.config.json,預設值對應 Vite + TypeScript + Vitest 專案 */
108
114
  export const RepoConfig = z.object({
@@ -114,6 +120,15 @@ export const RepoConfig = z.object({
114
120
  codex: z.string().trim().min(1).optional(),
115
121
  gemini: z.string().trim().min(1).optional(),
116
122
  }).default({}),
123
+ modelSelection: z.object({
124
+ mode: z.enum(["balanced", "adaptive"]).default("balanced"),
125
+ stageStrength: z.strictObject({
126
+ spec: ModelStrength.optional(), plan: ModelStrength.optional(), planReview: ModelStrength.optional(),
127
+ planFix: ModelStrength.optional(), planArbiter: ModelStrength.optional(), taskTests: ModelStrength.optional(),
128
+ taskCode: ModelStrength.optional(), taskReview: ModelStrength.optional(), taskFix: ModelStrength.optional(),
129
+ fix: ModelStrength.optional(), review: ModelStrength.optional(),
130
+ }).default({}),
131
+ }).default({ mode: "balanced", stageStrength: {} }),
117
132
  /** 參與的 agent(順序不影響分工);未設定時取 agents 裡已安裝的 CLI */
118
133
  cycle: z.array(z.string()).min(1).optional(),
119
134
  /** review 後的修正由誰做:ring=輪到下一位;author=最後寫程式的 agent */
@@ -171,6 +186,8 @@ export const FlowRun = z.object({
171
186
  fixSource: z.enum(["verify", "review"]).optional(),
172
187
  /** 各關卡的連續失敗次數 */
173
188
  attempts: z.record(z.string(), z.number()),
189
+ modelMode: z.enum(["balanced", "adaptive"]).optional(),
190
+ modelRetryAttempts: z.record(z.string(), z.number().int().nonnegative()).optional(),
174
191
  taskIndex: z.number().int().nonnegative(),
175
192
  /** 目前任務進行到哪一步:寫測試 → 實作 → 審查 → 驗證,審查或驗證未通過時進入修正 */
176
193
  taskPhase: z.enum(["tests", "code", "review", "verify", "fix"]),
package/dist/store.js CHANGED
@@ -1,5 +1,6 @@
1
1
  import { appendFileSync, existsSync, mkdirSync, readdirSync, readFileSync, renameSync, writeFileSync } from "node:fs";
2
2
  import { dirname, join } from "node:path";
3
+ import { stageOfStep } from "./modelSelection.js";
3
4
  import { agentflowctlDir, runDir } from "./paths.js";
4
5
  import { FlowRun } from "./schemas.js";
5
6
  const statePath = (id) => join(runDir(id), "state.json");
@@ -34,21 +35,40 @@ export function addUsage(id, entry) {
34
35
  mkdirSync(runDir(id), { recursive: true });
35
36
  appendFileSync(usagePath(id), `${JSON.stringify({ at: new Date().toISOString(), ...entry })}\n`);
36
37
  }
37
- /** 依 agent 加總 token 與執行次數,方便比較各家模型 */
38
- export function usageByAgent(id) {
38
+ const emptySummary = () => ({ tokens: 0, inputTokens: 0, outputTokens: 0, runs: 0, reportedRuns: 0, unreportedRuns: 0, legacyRuns: 0, legacyTokens: 0 });
39
+ export function listUsage(id) {
39
40
  const p = usagePath(id);
40
- const out = {};
41
- if (!existsSync(p))
42
- return out;
43
- for (const line of readFileSync(p, "utf8").split("\n").filter(Boolean)) {
44
- const e = JSON.parse(line);
45
- const key = e.agent ?? "?";
46
- const acc = (out[key] ??= { tokens: 0, runs: 0 });
41
+ return existsSync(p) ? readFileSync(p, "utf8").split("\n").filter(Boolean).map((line) => JSON.parse(line)) : [];
42
+ }
43
+ function addSummary(acc, e) {
44
+ acc.runs += 1;
45
+ if (e.usageReported === true && typeof e.inputTokens === "number" && typeof e.outputTokens === "number") {
46
+ acc.reportedRuns += 1;
47
+ acc.inputTokens += e.inputTokens ?? 0;
48
+ acc.outputTokens += e.outputTokens ?? 0;
47
49
  acc.tokens += (e.inputTokens ?? 0) + (e.outputTokens ?? 0);
48
- acc.runs += 1;
49
50
  }
51
+ else if (e.usageReported === false || e.usageReported === true)
52
+ acc.unreportedRuns += 1;
53
+ else {
54
+ acc.legacyRuns += 1;
55
+ acc.legacyTokens += (e.inputTokens ?? 0) + (e.outputTokens ?? 0);
56
+ }
57
+ }
58
+ /** 只加總明確回報的 token,舊資料的 0 不視為已回報。 */
59
+ function groupUsage(id, keyOf) {
60
+ const out = {};
61
+ for (const e of listUsage(id))
62
+ addSummary(out[keyOf(e)] ??= emptySummary(), e);
50
63
  return out;
51
64
  }
65
+ export const usageByAgent = (id) => groupUsage(id, (e) => e.agent);
66
+ export const usageByModelStage = (id) => groupUsage(id, (e) => `${e.model ?? "CLI 預設(名稱未知)"} / ${e.stage}`);
67
+ export const usageByStrength = (id) => groupUsage(id, (e) => e.strength ?? "未知");
68
+ /** 依階段鍵加總(所有任務的同一步合在一起);認不得的步驟歸為「其他」。 */
69
+ export const usageByStage = (id) => groupUsage(id, (e) => stageOfStep(e.stage) ?? "其他");
70
+ /** 依任務加總寫測試、實作、任務審查與任務修正;規格、計畫與整體階段歸為「非任務步驟」。 */
71
+ export const usageByTask = (id) => groupUsage(id, (e) => /^(T-\d+)-/.exec(e.stage)?.[1] ?? "非任務步驟");
52
72
  /** 這個 run 已執行 agent 的次數(每次執行都會記一筆用量) */
53
73
  export function agentRuns(id) {
54
74
  const p = usagePath(id);
package/dist/tasks.js CHANGED
@@ -1,5 +1,12 @@
1
1
  /** 一個任務最多做兩件事:對應的驗收條件超過這個數量就要再拆 */
2
2
  export const MAX_TASK_ACCEPTANCE = 2;
3
+ /** adaptive 的新計畫必須明確標註難度;舊 run 仍可讀取缺少欄位的 task。 */
4
+ export function validateTaskComplexity(tasks, mode) {
5
+ if (mode === "balanced")
6
+ return undefined;
7
+ const missing = tasks.filter((task) => !task.complexity).map((task) => task.id);
8
+ return missing.length ? `任務缺少 complexity:${missing.join("、")}` : undefined;
9
+ }
3
10
  /** 只取目前任務負責的驗收條件,避免每次實作都重讀整份清單。 */
4
11
  export function taskAcceptance(task, acceptance) {
5
12
  const byId = new Map(acceptance.map((item) => [item.id, item]));
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "agentflowctl",
3
3
  "license": "MIT",
4
- "version": "0.10.0",
4
+ "version": "0.12.0",
5
5
  "description": "跨廠商 AI 開發 harness:Claude Code、Codex、Gemini 輪流實作、審查、修正",
6
6
  "keywords": [
7
7
  "ai",
package/prompts/fix.md CHANGED
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <inputs>
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <task>
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <task>
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <requirement>
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <requirement>
@@ -32,6 +32,7 @@
32
32
  - 每一條驗收條件都至少要有一個任務負責;`dependsOn` 不可有循環。
33
33
  - 一個任務只做一件事,最多兩件:`acceptance` 最多列兩條驗收條件;驗收條件一條只描述一個行為。修改時若任務變大,請拆開,不要合併。
34
34
  - 測試檔名必須符合正規表示式 `{{testPattern}}`。
35
+ - 修改 task 時保留或補上 `complexity`(`low`、`medium`、`high`)。依影響範圍、技術不確定性與失敗後果重新判定,取最高等級;同步更新 .flow/plan.md 中該 task 的逐項證據與最終等級。若不同意審查者建議的等級,在「## 審查回應」引用具體程式碼或測試依據。
35
36
  </output_format>
36
37
 
37
38
  <constraints>
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <requirement>
@@ -30,6 +30,8 @@
30
30
  1. **需求覆蓋**:規格是否完整涵蓋原始需求?有沒有遺漏、誤解,或加入需求沒要求的範圍?
31
31
  2. **驗收條件**:每一條是否具體、可以用自動化測試驗證,而且只描述一個行為?把多個行為寫在同一條的,要求拆開。有沒有重要的邊界情況或錯誤處理沒被列入?
32
32
  3. **任務拆解**:每個任務是否只做一件事(最多兩件),小到一次 TDD 循環就能完成,而且能寫出「實作前會失敗」的測試?任務太大、一次要動很多檔案或驗證很多行為的,要求拆成更小的任務。相依順序是否合理?
33
+ 同時依 .flow/plan.md 的逐項理由及實際程式碼,獨立核對每個 task 的 `complexity`:分別看影響範圍、技術不確定性與失敗後果,取最高等級。`low` 須是沿用既有做法、侷限單一行為或模組且失敗可由局部測試發現;`medium` 包括多模組或介面協調、非典型邊界、相容性或狀態遷移風險;`high` 包括跨系統契約、架構或資料模型變更、未知的關鍵技術路徑,或資料遺失、權限、難以回復的風險。不要只憑檔案數、程式碼行數或驗收條件數判定。
34
+ 理由缺漏、與程式碼不符,或高低估會影響選模時,要求修正;在 `note` 指出 task ID、具體證據、建議等級及須修改的 .flow/plan.md/.flow/tasks.json 部分。不要為缺少高價值證據的細微措辭差異要求修改。
33
35
  4. **技術方向**:是否符合專案既有的架構與慣例?有沒有更簡單的做法,或明顯的風險?
34
36
 
35
37
  措辭、格式這類不影響實作結果的小問題,不需要要求修改。
package/prompts/plan.md CHANGED
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <inputs>
@@ -25,7 +25,7 @@
25
25
  <steps>
26
26
  1. 若 .flow/feedback.md 存在,先閱讀,並依內容修正前次的產出。
27
27
  2. 閱讀規格與相關程式碼。
28
- 3. 撰寫 .flow/plan.md:整體實作方式、要新增或修改的模組、任務順序的理由。
28
+ 3. 撰寫 .flow/plan.md:整體實作方式、要新增或修改的模組、任務順序的理由;逐一記錄 task 的難度判定依據。
29
29
  4. 撰寫 .flow/tasks.json。
30
30
  </steps>
31
31
 
@@ -38,6 +38,7 @@
38
38
  "id": "T-1",
39
39
  "title": "建立表單驗證 schema",
40
40
  "description": "具體要做什麼、要動哪些檔案、測試要驗證什麼行為",
41
+ "complexity": "low",
41
42
  "dependsOn": [],
42
43
  "acceptance": ["AC-1"]
43
44
  }
@@ -51,6 +52,11 @@
51
52
  - `title` 用一句話說出這件事;需要用「並且」「以及」串起來的,就是兩個任務。
52
53
  - `description` 寫清楚要動哪些檔案、測試要驗證哪個行為,以及這個任務不做什麼。
53
54
  - 每個任務都必須能寫出「在實作前會失敗」的測試;純設定或重構類工作請併入相關任務。
55
+ - 每個任務先檢查預計修改的程式碼,再依「影響範圍、技術不確定性、失敗後果」三個面向判定 `complexity`,取其中最高的等級;不要只憑檔案數、程式碼行數或驗收條件數判定。
56
+ - `low`:沿用現有做法,變更侷限在單一行為或模組,失敗容易由局部測試發現且不影響既有資料或對外契約。
57
+ - `medium`:需要協調多個模組或既有介面、處理非典型邊界,或有相容性與狀態遷移風險,但可依已知做法實作與驗證。
58
+ - `high`:涉及跨系統契約、架構或資料模型變更;關鍵技術路徑尚不確定;或失敗可能造成資料遺失、權限問題或難以回復的影響。任一面向符合就標 `high`。
59
+ - 在 .flow/plan.md 逐一列出 task ID、三個面向的具體證據與最終等級;缺少證據時先查閱相關程式碼,不要一律標 `low` 或憑猜測調高。
54
60
  - 測試檔名必須符合正規表示式 `{{testPattern}}`。
55
61
  - 每一條驗收條件都至少要有一個任務負責;`dependsOn` 不可有循環。
56
62
  </guidelines>
package/prompts/review.md CHANGED
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <inputs>
package/prompts/spec.md CHANGED
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <requirement>
@@ -14,7 +14,7 @@
14
14
  ```
15
15
 
16
16
  新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
17
- 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
17
+ 處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved、resolved 或 accepted", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置:撰寫者只能用 proposed_resolved 提出修正;審查者可以用 resolved 或 accepted 結案。「參考資訊」(info)只供參考,不要放進 dispositions。重要疑慮必須放在這份檔案,不能只寫在回覆的 <concerns>。
18
18
  </handoff>
19
19
 
20
20
  <inputs>