agentflowctl 0.18.0 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -2
- package/dist/engine.js +42 -10
- package/dist/logs.js +3 -0
- package/dist/stats.js +4 -2
- package/dist/usageInsights.js +5 -3
- package/package.json +1 -1
- package/prompts/handoff-repair.md +51 -0
package/README.md
CHANGED
|
@@ -60,9 +60,9 @@ agentflowctl resume f-xxxx # 從暫停、中斷或失敗處接續
|
|
|
60
60
|
|
|
61
61
|
`status` 會列出目前階段、未結的交接事項與下一步指令;失敗或暫停時也會顯示原因。任務進度與「待你確認(不進實作)」分成兩個區塊:實作任務在「任務」,`kind: confirm` 的項目在「待你確認(不進實作)」,兩邊都會列出。計畫還沒通過首次驗證時,這個區塊的標題會多帶「(計畫尚未定案,以下為草稿)」,代表清單是從尚未經檢查的草稿蒐集,項目最終可能不會定案。只想看要人眼確認的項目時,用 `confirmations <id>`。要看某一步的詳細輸出,可用 `logs <id> <編號>`;加 `--full` 看完整工具內容,或加 `--raw` 看原始輸出。
|
|
62
62
|
|
|
63
|
-
`stats` 依 log 的開始與結束時間統計每個步驟的執行次數、失敗次數、總耗時與最長一次,並分開列出 agent 與專案指令(install、測試、checks)各占多少時間,最耗時的步驟排在最前面。沒有結束紀錄的 log
|
|
63
|
+
`stats` 依 log 的開始與結束時間統計每個步驟的執行次數、失敗次數、總耗時與最長一次,並分開列出 agent 與專案指令(install、測試、checks)各占多少時間,最耗時的步驟排在最前面。沒有結束紀錄的 log 列為未完成,不計入耗時;總經過時間包含暫停與等待核准。紅燈階段的測試指令(`T-<n>-red`)失敗是預期結果,只有測試意外通過才計為失敗。
|
|
64
64
|
|
|
65
|
-
`insights` 把所有 run 分成獨立區塊彙總:最終狀態、失敗原因(重試達上限、仲裁停止、agent 次數用完等)、用量(輸入/輸出/cache、強度占比、階段/agent,各 run 用量最高的任務)、各模型與步驟、步驟執行與失敗(與 `stats` 相同取 log 的檔頭檔尾,跨 run 不計總經過時間)、關卡重試原因。任務關卡與任務步驟不分 task id 合併計算。最後列出最多五則建議,規則由程式套門檻,不是再請 agent
|
|
65
|
+
`insights` 把所有 run 分成獨立區塊彙總:最終狀態、失敗原因(重試達上限、仲裁停止、agent 次數用完等)、用量(輸入/輸出/cache、強度占比、階段/agent,各 run 用量最高的任務)、各模型與步驟、步驟執行與失敗(與 `stats` 相同取 log 的檔頭檔尾,跨 run 不計總經過時間)、關卡重試原因。任務關卡與任務步驟不分 task id 合併計算。最後列出最多五則建議,規則由程式套門檻,不是再請 agent 分析。「輸入遠大於輸出」不計入 cache 讀取(Claude 的 cache 寫入仍計入),門檻是非 cache 的輸入與輸出合計至少 5000 tokens 且輸入佔 85% 以上;cache 讀取量另外列在說明中。合計 token 不是主指標。覆蓋不足時會先警告占比可能失真。各分組是同一批呼叫的不同切片,不要跨組相加。舊 run 沒有重試或失敗原因紀錄,不會回填。單一 run 的全量明細仍用 `status <id>` 與 `stats <id>`。
|
|
66
66
|
|
|
67
67
|
執行紀錄在 `.agentflowctl/runs/<id>/`,工作分支在 `.agentflowctl/worktrees/<id>/`。不再需要某次 run 時,可用 `agentflowctl clean <id>` 清除 worktree 與紀錄;`agentflowctl clean --all` 一次清除所有 done、failed 的 run,以及沒有紀錄的 worktree(進行中、暫停、等待核准的不動)。`flow/<id>` 分支會保留。
|
|
68
68
|
|
|
@@ -80,6 +80,8 @@ agentflowctl resume f-xxxx # 從暫停、中斷或失敗處接續
|
|
|
80
80
|
| `failed`:測試、檢查、審查或 agent 執行失敗 | 依 `status` 提示查看失敗的 log,處理原因後執行 `agentflowctl resume <id>`;失敗階段會重試 |
|
|
81
81
|
| `failed`:已達 agent 執行次數上限 | 用 `agentflowctl resume <id> --max-agent-runs 100` 調高上限後接續,數字須大於已執行次數 |
|
|
82
82
|
|
|
83
|
+
寫作類步驟(規格、計畫、計畫修正、紅燈測試、綠燈實作、fix)完成且通過關卡後,若 `.flow/handoff-response.json` 不合格,會請同一家 agent 再呼叫一次,只補寫交接(不重做工作,也不換人代打);補寫期間對其他檔案的變更與 commit 一律丟棄(規格與計畫類步驟沒有 commit,補寫者會被告知範圍為空,不會把事項標成已處理,補寫期間對規格與計畫檔的修改同樣丟棄)。補寫仍不合格或額度用完,才照舊還原這一步並重試。補寫呼叫記在原步驟名稱底下,`stats` 與 `logs` 會多一筆,也會計入 `--max-agent-runs` 的次數。
|
|
84
|
+
|
|
83
85
|
例如失敗時,可照終端機列出的 log 編號查看原因:
|
|
84
86
|
|
|
85
87
|
```bash
|
package/dist/engine.js
CHANGED
|
@@ -7,7 +7,7 @@ import { escapeXml, opinion, reviewIssue } from "./feedback.js";
|
|
|
7
7
|
import { changedFiles, commitAll, discardChanges, git, headCommit, resetTo } from "./git.js";
|
|
8
8
|
import { acceptHandoff, openActions, prepareHandoff, previewHandoff, readHandoff, recoverHandoff, responsePath, reviewHandoffGate, validateHandoffResponse } from "./handoff.js";
|
|
9
9
|
import { flowDir, logDir, planArbitrationPath, planReviewStatePath, projectRoot, runDir, worktreeDir } from "./paths.js";
|
|
10
|
-
import { CMD_AGENT, nextLogFile } from "./logs.js";
|
|
10
|
+
import { CMD_AGENT, nextLogFile, redStepName } from "./logs.js";
|
|
11
11
|
import { exec } from "./proc.js";
|
|
12
12
|
import { arbiterPanel, availableAgent, fixAgent, planAgent, planFixAgent, reviewers, specAgent, taskAgents } from "./roles.js";
|
|
13
13
|
import { dropCall, loadCalls, openRound, runPool, saveCall, storedCallValid } from "./parallelReview.js";
|
|
@@ -57,8 +57,8 @@ async function agentStep(run, planned, step, prompt, mode) {
|
|
|
57
57
|
let agent = planned;
|
|
58
58
|
for (;;) {
|
|
59
59
|
if (exhausted.has(agent)) {
|
|
60
|
-
if (mode.kind === "review") {
|
|
61
|
-
throw new QuotaPause(`${agent} 的額度已用完;${step}
|
|
60
|
+
if (mode.kind === "review" || mode.pinned) {
|
|
61
|
+
throw new QuotaPause(`${agent} 的額度已用完;${step} ${mode.pinned ? "不由另一家代打" : "是審查步驟,不由另一家代打"}`);
|
|
62
62
|
}
|
|
63
63
|
const sub = availableAgent(run.cycle, agent, [...exhausted], `${run.id}:${step}:sub`);
|
|
64
64
|
if (!sub)
|
|
@@ -112,6 +112,38 @@ base) {
|
|
|
112
112
|
return error.message;
|
|
113
113
|
}
|
|
114
114
|
}
|
|
115
|
+
/**
|
|
116
|
+
* 寫作步驟的交接:關卡已通過,交接回覆不合格時不丟掉工作,請同一家 agent 只補寫一次。
|
|
117
|
+
* 補寫前記下 HEAD 與 files 的內容,補寫後一律 reset 回去並還原這些檔案,連補寫自行建立的 commit 也丟棄,
|
|
118
|
+
* 所以補寫無法改變已通過的關卡。補寫仍失敗(或額度用完)就回傳錯誤,由呼叫端照舊還原並重試。
|
|
119
|
+
* base 是這一步開始前的 commit;沒有 commit 的步驟(規格、計畫)省略,補寫者會被告知範圍為空。
|
|
120
|
+
*/
|
|
121
|
+
async function settleHandoff(run, outcome, files, base) {
|
|
122
|
+
const first = finishHandoff(run, outcome, "writer");
|
|
123
|
+
if (!first)
|
|
124
|
+
return undefined;
|
|
125
|
+
const repo = worktreeDir(run.id);
|
|
126
|
+
const settled = await headCommit(repo);
|
|
127
|
+
const snap = snapshotPlan(run, files);
|
|
128
|
+
const cleanup = async () => { await resetTo(repo, settled); restorePlan(run, snap); };
|
|
129
|
+
info(run, `📎 交接回覆不合格,請 ${outcome.agent} 只補寫交接(不重做工作):${first}`);
|
|
130
|
+
let repair;
|
|
131
|
+
try {
|
|
132
|
+
// pinned 不找代打,額度用完直接暫停,所以不需要 reset;善後一律交給 finally
|
|
133
|
+
repair = await agentStep(run, outcome.agent, outcome.step, renderPrompt("handoff-repair", { step: outcome.step, error: first, range: `${base ?? settled}..${settled}` }), { kind: "write", pinned: true, reset: () => { } });
|
|
134
|
+
}
|
|
135
|
+
catch (error) {
|
|
136
|
+
if (error instanceof QuotaPause)
|
|
137
|
+
return first;
|
|
138
|
+
throw error;
|
|
139
|
+
}
|
|
140
|
+
finally {
|
|
141
|
+
await cleanup();
|
|
142
|
+
}
|
|
143
|
+
if (!repair.r.ok)
|
|
144
|
+
return `${first}(補寫交接時 Agent 執行失敗:${repair.r.summary})`;
|
|
145
|
+
return finishHandoff(run, repair, "writer");
|
|
146
|
+
}
|
|
115
147
|
/** 印出回覆裡的 XML 中繼資料;只供人檢視,關卡仍由程式檢查決定 */
|
|
116
148
|
function reportMeta(run, agent, r) {
|
|
117
149
|
if (!r.meta)
|
|
@@ -267,7 +299,7 @@ async function specStage(run) {
|
|
|
267
299
|
const ids = ac.data.map((a) => a.id);
|
|
268
300
|
if (new Set(ids).size !== ids.length)
|
|
269
301
|
return retry(run, "spec", "驗收條件 id 有重複", "spec", "format_invalid");
|
|
270
|
-
const handoffError =
|
|
302
|
+
const handoffError = await settleHandoff(run, outcome, PLAN_FILES);
|
|
271
303
|
if (handoffError)
|
|
272
304
|
return retry(run, "spec", handoffError, "spec", "handoff_invalid");
|
|
273
305
|
return succeed(run, "spec", "plan");
|
|
@@ -365,7 +397,7 @@ async function planStage(run) {
|
|
|
365
397
|
const ordered = validatePlan(run);
|
|
366
398
|
if (typeof ordered === "string")
|
|
367
399
|
return retry(run, "plan", ordered, "plan", "format_invalid");
|
|
368
|
-
const handoffError =
|
|
400
|
+
const handoffError = await settleHandoff(run, outcome, PLAN_FILES);
|
|
369
401
|
if (handoffError)
|
|
370
402
|
return retry(run, "plan", handoffError, "plan", "handoff_invalid");
|
|
371
403
|
acceptPlan(run, ordered);
|
|
@@ -730,7 +762,7 @@ async function planFixStage(run) {
|
|
|
730
762
|
restorePlan(run, snap);
|
|
731
763
|
return retry(run, "plan-fix", `${feedback}\n\n另外,修改後的計畫沒有通過格式檢查,已還原:\n${ordered}`, "plan_fix", "format_invalid");
|
|
732
764
|
}
|
|
733
|
-
const handoffError =
|
|
765
|
+
const handoffError = await settleHandoff(run, outcome, PLAN_REPLY_FILES);
|
|
734
766
|
if (handoffError) {
|
|
735
767
|
restorePlan(run, snap);
|
|
736
768
|
// 計畫已還原,要保留原本的審查意見,否則下一次修正不知道要改什麼
|
|
@@ -888,12 +920,12 @@ async function implementStage(run) {
|
|
|
888
920
|
await resetTo(repo, before);
|
|
889
921
|
return retry(run, key, `沒有新增或修改任何符合 /${cfg.testPattern}/ 的測試檔。`, "implement", "tests_not_written");
|
|
890
922
|
}
|
|
891
|
-
const red = await runCommand(target(run,
|
|
923
|
+
const red = await runCommand(target(run, redStepName(task.id), CMD_AGENT), testCmd);
|
|
892
924
|
if (red.ok && !waiveRed) {
|
|
893
925
|
await resetTo(repo, before);
|
|
894
926
|
return retry(run, key, "測試在功能尚未實作前就全部通過,代表測試沒有驗證到新行為。請撰寫會因功能尚未實作而失敗的測試。", "implement", "tests_not_red");
|
|
895
927
|
}
|
|
896
|
-
const handoffError =
|
|
928
|
+
const handoffError = await settleHandoff(run, outcome, LOCKED_FILES, before);
|
|
897
929
|
if (handoffError) {
|
|
898
930
|
await resetTo(repo, before);
|
|
899
931
|
return retry(run, key, handoffError, "implement", "handoff_invalid");
|
|
@@ -940,7 +972,7 @@ async function implementStage(run) {
|
|
|
940
972
|
info(run, ` ✗ 測試仍未通過${logHint(run, green.seq)}`);
|
|
941
973
|
return retry(run, key, `測試仍未通過:\n\n\`\`\`\n${tail(green.output)}\n\`\`\``, "implement", "tests_not_green");
|
|
942
974
|
}
|
|
943
|
-
const handoffError =
|
|
975
|
+
const handoffError = await settleHandoff(run, outcome, LOCKED_FILES, testsCommit);
|
|
944
976
|
if (handoffError) {
|
|
945
977
|
await resetTo(repo, testsCommit);
|
|
946
978
|
return retry(run, key, handoffError, "implement", "handoff_invalid");
|
|
@@ -1082,7 +1114,7 @@ async function applyFix(run, opts) {
|
|
|
1082
1114
|
await resetTo(repo, before);
|
|
1083
1115
|
return again(`${feedback}\n\n另外:不可刪除測試檔來讓檢查通過,已還原:${deleted.join(", ")}`, "tests_deleted");
|
|
1084
1116
|
}
|
|
1085
|
-
const handoffError =
|
|
1117
|
+
const handoffError = await settleHandoff(run, outcome, LOCKED_FILES, before);
|
|
1086
1118
|
if (handoffError) {
|
|
1087
1119
|
await resetTo(repo, before);
|
|
1088
1120
|
return again(`${feedback}\n\n另外,交接回覆不合格,本次修正已還原:${handoffError}`, "handoff_invalid");
|
package/dist/logs.js
CHANGED
|
@@ -15,6 +15,9 @@ export const FOOTER_PREFIX = "# exit ";
|
|
|
15
15
|
export const STDERR_MARK = "[stderr]";
|
|
16
16
|
/** 專案指令(install、測試、checks)的 agent 欄位 */
|
|
17
17
|
export const CMD_AGENT = "cmd";
|
|
18
|
+
/** 紅燈測試指令的步驟名稱;engine 寫 log 與 stats 判斷預期失敗都用這裡,命名不會各改各的 */
|
|
19
|
+
export const redStepName = (taskId) => `${taskId}-red`;
|
|
20
|
+
export const isRedStep = (step) => /^T-\d+-red$/.test(step);
|
|
18
21
|
const segment = (s) => s.replace(/[\s/\\:*?"<>|]+/g, "_");
|
|
19
22
|
/** 下一份 log 的完整路徑;序號接在目錄裡最大的序號之後 */
|
|
20
23
|
export function nextLogFile(dir, stage, step, agent) {
|
package/dist/stats.js
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
import { CMD_AGENT } from "./logs.js";
|
|
1
|
+
import { CMD_AGENT, isRedStep } from "./logs.js";
|
|
2
2
|
const time = (iso) => (iso ? new Date(iso).getTime() : NaN);
|
|
3
|
+
/** 紅燈階段的測試指令:失敗才是預期結果,意外通過才代表這一步沒過關 */
|
|
4
|
+
const isRedCommand = (kind, step) => kind === "cmd" && isRedStep(step);
|
|
3
5
|
/** 從 log 的檔頭與檔尾算出每個步驟的次數、失敗與耗時,不讀 log 內容 */
|
|
4
6
|
export function computeStats(entries) {
|
|
5
7
|
const byKey = new Map();
|
|
@@ -23,7 +25,7 @@ export function computeStats(entries) {
|
|
|
23
25
|
unfinished += 1;
|
|
24
26
|
continue;
|
|
25
27
|
}
|
|
26
|
-
if (!footer.ok)
|
|
28
|
+
if (isRedCommand(kind, header.step) ? footer.ok : !footer.ok)
|
|
27
29
|
s.failed += 1;
|
|
28
30
|
const ms = Math.max(0, end - start);
|
|
29
31
|
s.totalMs += ms;
|
package/dist/usageInsights.js
CHANGED
|
@@ -63,10 +63,12 @@ export function usageFindings(input) {
|
|
|
63
63
|
detail: `重試 ${retries.length} 次(不含審查要求修改與仲裁要求修訂),最多「${top ? retryLabel(top[0]) : "未知"}」(${top?.[1] ?? 0} 次)。以平均每次 ${avg} tokens 估算,重試約 ${waste} tokens。先改該關卡,再縮 prompt。`,
|
|
64
64
|
});
|
|
65
65
|
}
|
|
66
|
-
|
|
66
|
+
const freshInput = Math.max(0, total.inputTokens - total.cacheReadTokens);
|
|
67
|
+
const freshTotal = freshInput + total.outputTokens;
|
|
68
|
+
if (freshTotal >= 5000 && freshInput / freshTotal >= 0.85) {
|
|
67
69
|
out.push({
|
|
68
|
-
code: "input_heavy", impactTokens:
|
|
69
|
-
detail:
|
|
70
|
+
code: "input_heavy", impactTokens: freshInput, title: FINDING_LABEL.input_heavy,
|
|
71
|
+
detail: `不含 cache 讀取的輸入 ${freshInput}、輸出 ${total.outputTokens}(輸入佔 ${pct(freshInput, freshTotal)});另有 cache 讀取 ${total.cacheReadTokens} 未計入。上下文可能太大:規格、計畫、測試輸出或一次讀太多檔。`,
|
|
70
72
|
});
|
|
71
73
|
}
|
|
72
74
|
const stagesWithTokens = Object.values(byStage).filter((s) => s.tokens > 0).length;
|
package/package.json
CHANGED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
<role>
|
|
2
|
+
你是交接紀錄員,只負責補寫上一個步驟沒有寫好的 .flow/handoff-response.json。工作本身已經完成並通過檢查,不要重做。
|
|
3
|
+
</role>
|
|
4
|
+
|
|
5
|
+
<context>
|
|
6
|
+
目前的工作目錄就是專案(agentflowctl 為這次任務建立的專用 git worktree)。
|
|
7
|
+
步驟 {{step}} 的工作已完成,變更範圍是 `{{range}}`,但交接回覆沒有通過檢查,原因:
|
|
8
|
+
|
|
9
|
+
{{error}}
|
|
10
|
+
</context>
|
|
11
|
+
|
|
12
|
+
<handoff>
|
|
13
|
+
先閱讀 .flow/handoff-context.md,確認有哪些待處理事項。再寫入 .flow/handoff-response.json;即使沒有事項也必須寫出空陣列:
|
|
14
|
+
|
|
15
|
+
```json
|
|
16
|
+
{ "newIssues": [], "dispositions": [] }
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
新增事項格式:{ "kind": "action 或 info", "summary": "具體問題", "evidence": "檔案位置或檢查證據", "targetStage": "plan 或 code" }。
|
|
20
|
+
處置格式:{ "id": "既有事項 ID", "status": "proposed_resolved", "reason": "具體處理理由", "evidence": "檔案、commit 或檢查結果" }。只有「待處理事項」(action)可以處置,而且這個步驟的作者只能用 proposed_resolved;「參考資訊」(info)不要放進 dispositions。
|
|
21
|
+
</handoff>
|
|
22
|
+
|
|
23
|
+
<inputs>
|
|
24
|
+
- .flow/handoff-context.md:待處理事項與參考資訊
|
|
25
|
+
- `git log {{range}}` 與 `git diff {{range}}`:這一步實際做了什麼,處置的理由與證據要以它為準。範圍兩端相同代表這一步沒有產生 commit,這時不要把任何事項標成已處理
|
|
26
|
+
</inputs>
|
|
27
|
+
|
|
28
|
+
<steps>
|
|
29
|
+
1. 讀 .flow/handoff-context.md 與上述 git 範圍,判斷每個待處理事項這一步有沒有真的處理;沒有把握的事項不要處置。
|
|
30
|
+
2. 只寫入 .flow/handoff-response.json,格式如上。
|
|
31
|
+
</steps>
|
|
32
|
+
|
|
33
|
+
<constraints>
|
|
34
|
+
- 只能建立或修改 .flow/handoff-response.json。其他檔案的變更與任何 git commit 都會在你結束後被自動丟棄,不會生效。
|
|
35
|
+
- 不要重做或改善這一步的工作,也不要重新跑完整測試。
|
|
36
|
+
</constraints>
|
|
37
|
+
|
|
38
|
+
<reply_format>
|
|
39
|
+
完成後,回覆的最後必須附上以下 XML 中繼資料(只附一次,標籤名稱不可更改):
|
|
40
|
+
|
|
41
|
+
```xml
|
|
42
|
+
<result>
|
|
43
|
+
<status>done 或 blocked</status>
|
|
44
|
+
<summary>一兩句說明補寫了什麼;blocked 時說明卡在哪裡</summary>
|
|
45
|
+
<files_changed>
|
|
46
|
+
<file>.flow/handoff-response.json</file>
|
|
47
|
+
</files_changed>
|
|
48
|
+
<concerns>沒有就留空</concerns>
|
|
49
|
+
</result>
|
|
50
|
+
```
|
|
51
|
+
</reply_format>
|