agency-orchestrator 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/README.en.md +14 -8
  2. package/README.md +9 -4
  3. package/dist/cli/compose.js +20 -0
  4. package/dist/cli.js +25 -4
  5. package/dist/connectors/api-providers.js +9 -0
  6. package/dist/core/compare.d.ts +4 -3
  7. package/dist/core/compare.js +10 -6
  8. package/dist/core/executor.d.ts +18 -1
  9. package/dist/core/executor.js +149 -51
  10. package/dist/core/parser.js +15 -1
  11. package/dist/core/verify.d.ts +44 -0
  12. package/dist/core/verify.js +130 -0
  13. package/dist/index.d.ts +15 -0
  14. package/dist/index.js +66 -2
  15. package/dist/output/reporter.d.ts +6 -1
  16. package/dist/output/reporter.js +33 -3
  17. package/dist/types.d.ts +18 -0
  18. package/dist/utils/sponsor-guide.d.ts +22 -0
  19. package/dist/utils/sponsor-guide.js +30 -0
  20. package/package.json +5 -4
  21. package/web/server.js +255 -51
  22. package/website/dist/apple-touch-icon.png +0 -0
  23. package/website/dist/assets/Changelog-BIGmNgWM.js +8 -0
  24. package/website/dist/assets/{CreativeLibrary-Bafp2-7y.js → CreativeLibrary-BHqHEQIh.js} +1 -1
  25. package/website/dist/assets/{Docs-DWjEFgqb.js → Docs-C_Bg8DeD.js} +17 -10
  26. package/website/dist/assets/{Experts-ARxT7R0z.js → Experts-ByRxvzGb.js} +1 -1
  27. package/website/dist/assets/{Home-D1fuy-e9.js → Home-UwklCf3w.js} +1 -1
  28. package/website/dist/assets/{Markdown-DIBk11T6.js → Markdown-DG2oIi7J.js} +1 -1
  29. package/website/dist/assets/{NotFound-yaONXr2c.js → NotFound-D8ohn3mE.js} +1 -1
  30. package/website/dist/assets/{PromptStudio-DdCuT1Il.js → PromptStudio-8ZbqqsfO.js} +1 -1
  31. package/website/dist/assets/SiteFooter-CEbbrx2O.js +1 -0
  32. package/website/dist/assets/Sponsors-L7_xxpqq.js +21 -0
  33. package/website/dist/assets/Studio-BHdFIqMb.js +188 -0
  34. package/website/dist/assets/{TutorialDetail-C8bjE0Uh.js → TutorialDetail-BxE_n_aG.js} +1 -1
  35. package/website/dist/assets/{Tutorials-CscSAnuX.js → Tutorials-C2ZzhsJ8.js} +1 -1
  36. package/website/dist/assets/{UsagePanel-CFfVmSwy.js → UsagePanel-pKvnMg0M.js} +1 -1
  37. package/website/dist/assets/{arrow-left-CBsGQe3h.js → arrow-left-y9GCeSb5.js} +1 -1
  38. package/website/dist/assets/{arrow-up-right-LSsttJsJ.js → arrow-up-right-BSpKIm_F.js} +1 -1
  39. package/website/dist/assets/{badge-xc3rfZGN.js → badge-BWe9-A6x.js} +1 -1
  40. package/website/dist/assets/{clock-LLdpHBz2.js → clock-CY07fup8.js} +1 -1
  41. package/website/dist/assets/{copy-nrJtFYdc.js → copy-CDpr96ho.js} +1 -1
  42. package/website/dist/assets/{copy-button-CiQbKrJI.js → copy-button-DmWvJste.js} +1 -1
  43. package/website/dist/assets/{download-DP_pAg1E.js → download-BxGUc2TT.js} +1 -1
  44. package/website/dist/assets/{external-link-CS2XMbHw.js → external-link-qWj52f9i.js} +1 -1
  45. package/website/dist/assets/index-B0DLOSbW.js +97 -0
  46. package/website/dist/assets/index-Bvmb-uHp.css +1 -0
  47. package/website/dist/assets/{mail-Bx74VrG-.js → mail-E8-9lFeI.js} +1 -1
  48. package/website/dist/assets/{search-BH4oWAhq.js → search-ChBwKsLg.js} +1 -1
  49. package/website/dist/assets/{sparkles-jiZMT7tJ.js → sparkles-Df-uhv9v.js} +1 -1
  50. package/website/dist/assets/sponsors-DYxHA2Hi.js +11 -0
  51. package/website/dist/assets/useBackend-CMxh6Cx4.js +30 -0
  52. package/website/dist/assets/{workflow-D0-wMIOl.js → workflow-D3exvF90.js} +1 -1
  53. package/website/dist/assets/workflows-BS7rjs0g.js +1 -0
  54. package/website/dist/favicon.ico +0 -0
  55. package/website/dist/favicon.svg +15 -0
  56. package/website/dist/index.html +11 -2
  57. package/website/dist/logo/ao-app-icon.svg +15 -0
  58. package/website/dist/logo/ao-banner.png +0 -0
  59. package/website/dist/logo/ao-mark-white.svg +6 -0
  60. package/website/dist/logo/ao-mark.svg +13 -0
  61. package/website/dist/og-image.png +0 -0
  62. package/website/dist/sponsors/logo-byteplus-icon.png +0 -0
  63. package/website/dist/sponsors/logo-duoyuanx-icon.png +0 -0
  64. package/website/dist/sponsors/logo-volcengine-icon.png +0 -0
  65. package/workflows//344/270/200/344/272/272/345/205/254/345/217/270-/345/201/232/344/272/247/345/223/201.yaml +112 -0
  66. package/workflows//344/270/200/344/272/272/345/205/254/345/217/270-/345/201/232/345/206/205/345/256/271.yaml +101 -0
  67. package/workflows//344/270/200/344/272/272/345/205/254/345/217/270-/345/201/232/346/212/225/347/240/224.yaml +125 -0
  68. package/workflows//344/270/200/344/272/272/345/205/254/345/217/270/345/205/250/345/221/230/345/244/247/344/274/232.yaml +2 -0
  69. package/website/dist/assets/Changelog-Cd1_ta-F.js +0 -8
  70. package/website/dist/assets/SiteFooter-CuZFXiZH.js +0 -1
  71. package/website/dist/assets/Sponsors-D_4M9Xth.js +0 -21
  72. package/website/dist/assets/Studio-BFYs1jm4.js +0 -182
  73. package/website/dist/assets/index-DawHdnrM.js +0 -97
  74. package/website/dist/assets/index-GFHogxlv.css +0 -1
  75. package/website/dist/assets/sponsors-2Ed3J2D2.js +0 -11
  76. package/website/dist/assets/useBackend-DAjkH7qN.js +0 -30
  77. package/website/dist/assets/workflows-OVJ6BxVu.js +0 -1
package/README.en.md CHANGED
@@ -3,13 +3,15 @@
3
3
  **English** | [中文](./README.md)
4
4
 
5
5
  > **One sentence in, a full plan out — multiple AI roles collaborate automatically.**
6
+ >
7
+ > **It's your one-person company: you're the boss, AI is the team — auto-assembled, asking your sign-off on big calls, delivering against acceptance criteria.**
6
8
 
7
9
  [![CI](https://github.com/jnMetaCode/agency-orchestrator/actions/workflows/ci.yml/badge.svg)](https://github.com/jnMetaCode/agency-orchestrator/actions)
8
10
  [![npm version](https://img.shields.io/npm/v/agency-orchestrator)](https://www.npmjs.com/package/agency-orchestrator)
9
11
  [![License: Apache-2.0](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](./LICENSE)
10
12
  [![PRs Welcome](https://img.shields.io/badge/PRs-welcome-brightgreen.svg)](./CONTRIBUTING.md)
11
13
 
12
- **One sentence → full plan · 216 expert AI roles · Zero-code YAML · 10 LLM providers · key supported (DeepSeek recommended), plus 7 key-free options**
14
+ **One sentence → full plan · 216 expert AI roles · Zero-code YAML · 11 LLM providers · key supported (DeepSeek recommended), plus 7 key-free options**
13
15
 
14
16
  > **Note:** `ao compose --run` auto-detects your language. Both 216 Chinese roles and 184 English roles ([agency-agents](https://github.com/msitarzewski/agency-agents), MIT) are **bundled in the npm package — no extra download needed**. **10 English workflow templates** are ready in `workflows/en/`.
15
17
 
@@ -20,7 +22,8 @@
20
22
  > If you find this useful, please **Star** it — helps others discover the project.
21
23
 
22
24
  <p align="center">
23
- <img src="./demo.gif" alt="ao compose --run demo" width="700">
25
+ <img src="./demo-studio-en.gif" alt="Web Studio: one sentence, AI builds the team" width="820"><br/>
26
+ <em>Web Studio: type one sentence, AI builds a team from 200+ experts and runs it</em>
24
27
  </p>
25
28
 
26
29
  ---
@@ -29,11 +32,6 @@
29
32
 
30
33
  Prefer not to use the command line? Run `ao web` locally and pick experts, run workflows, view outputs, and intervene live — all in a GUI, fully bilingual (EN/中文).
31
34
 
32
- <p align="center">
33
- <img src="./docs/screenshots/studio-roles-en.png" alt="Studio · Build a Team: pick from 200+ experts, AI composes the team" width="800"><br/>
34
- <em>Build a Team: pick from 200+ experts; AI composes and runs the team</em>
35
- </p>
36
-
37
35
  <p align="center">
38
36
  <img src="./docs/screenshots/studio-workflows-en.png" alt="Studio · Workflows: run built-in templates with one click" width="800"><br/>
39
37
  <em>Workflows: run built-in templates with one click, or compare several</em>
@@ -45,10 +43,16 @@ Prefer not to use the command line? Run `ao web` locally and pick experts, run w
45
43
 
46
44
  ## One Sentence, Full Result
47
45
 
46
+ Prefer the command line? One command, one sentence, full result:
47
+
48
48
  ```bash
49
49
  ao compose "I'm a programmer looking to start a side hustle with AI content, target $3K/month, give me a complete plan" --run
50
50
  ```
51
51
 
52
+ <p align="center">
53
+ <img src="./demo.gif" alt="ao compose CLI demo" width="700">
54
+ </p>
55
+
52
56
  5 AI roles collaborate automatically:
53
57
 
54
58
  ```
@@ -191,6 +195,7 @@ steps:
191
195
  - id: summary
192
196
  role: "product/product-manager"
193
197
  task: "Synthesize feedback:\n\n{{tech_report}}\n\n{{design_report}}"
198
+ acceptance: "1. Clear go/no-go verdict 2. List must-fix issues" # optional: injected at prompt tail, used as review yardstick
194
199
  depends_on: [tech_review, design_review]
195
200
  ```
196
201
 
@@ -209,7 +214,7 @@ analyze ──→ tech_review ──→ summary
209
214
  (parallel)
210
215
  ```
211
216
 
212
- ## 10 LLM Providers — 7 Need No API Key
217
+ ## 11 LLM Providers — 7 Need No API Key
213
218
 
214
219
  **Already paying for one of these? You're ready to go:**
215
220
 
@@ -230,6 +235,7 @@ analyze ──→ tech_review ──→ summary
230
235
  | Provider | Config | Env Variable |
231
236
  |----------|--------|-------------|
232
237
  | DeepSeek | `provider: "deepseek"` | `DEEPSEEK_API_KEY` |
238
+ | Volcengine Ark (Doubao / Kimi / GLM · sponsor) | `provider: "volcengine"` | `ARK_API_KEY` |
233
239
  | Claude API | `provider: "claude"` | `ANTHROPIC_API_KEY` |
234
240
  | OpenAI | `provider: "openai"` | `OPENAI_API_KEY` |
235
241
 
package/README.md CHANGED
@@ -3,13 +3,15 @@
3
3
  **中文** | [English](./README.en.md)
4
4
 
5
5
  > **一句话,让多个 AI 角色自动协作,几分钟出完整方案。**
6
+ >
7
+ > **也是你的「一人公司」:你当老板,AI 当团队——自动组队、重大决策请你签字、按验收标准交付。**
6
8
 
7
9
  [![CI](https://github.com/jnMetaCode/agency-orchestrator/actions/workflows/ci.yml/badge.svg)](https://github.com/jnMetaCode/agency-orchestrator/actions)
8
10
  [![npm version](https://img.shields.io/npm/v/agency-orchestrator)](https://www.npmjs.com/package/agency-orchestrator)
9
11
  [![License: Apache-2.0](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](./LICENSE)
10
12
  [![PRs Welcome](https://img.shields.io/badge/PRs-welcome-brightgreen.svg)](./CONTRIBUTING.md)
11
13
 
12
- **一句话出结果 · 216 个专业 AI 角色 · YAML 零代码 · 10 种大模型 · 支持 key(推荐 DeepSeek),也有 7 种免 key 方式**
14
+ **一句话出结果 · 216 个专业 AI 角色 · YAML 零代码 · 11 种大模型 · 支持 key(推荐 DeepSeek),也有 7 种免 key 方式**
13
15
 
14
16
  > 📖 [完整上手教程](https://mp.weixin.qq.com/s/XcGbkMb6TM6NLQiL7ICwbw) — 从安装到实战,10 分钟上手
15
17
  >
@@ -28,6 +30,7 @@
28
30
 
29
31
  不想敲命令行?本地跑一条 `ao web`,浏览器里勾选专家、运行工作流、查看产物、实时介入——全程图形界面,全中英双语。
30
32
 
33
+ > 🆕 **「一人公司」系列模板**:做产品 / 做内容 / 做投研 + 全员大会,关键步骤带验收标准(`acceptance`),投研含老板签字闸门——交付的是可验收的工作成果,不承诺奇迹。
31
34
  > 🆕 **AI 自动组队**:不知道选哪些专家?角色页一句话、不选角色,AI 自动从全部专家里挑人组队并运行。
32
35
  > 🆕 **可视化画布**:工作流可在画布上拖拽节点 / 连线(自动防环)/ 改任务·角色 / 保存,运行时节点按状态实时点亮。
33
36
  > 🆕 **创意库**:内置图像生成提示词库(Nano Banana / Gemini,可搜索 / 分类 / 一键复制)。
@@ -203,6 +206,7 @@ steps:
203
206
  - id: summary
204
207
  role: "product/product-manager"
205
208
  task: "综合反馈输出结论:\n\n{{tech_report}}\n\n{{design_report}}"
209
+ acceptance: "1. 明确给出通过/不通过结论 2. 列出必须解决的问题" # 可选:验收标准,注入 prompt 并作评审依据
206
210
  depends_on: [tech_review, design_review]
207
211
  ```
208
212
 
@@ -221,7 +225,7 @@ analyze ──→ tech_review ──→ summary
221
225
  (并行)
222
226
  ```
223
227
 
224
- ## 10 种 LLM — 7 种不需要 API key
228
+ ## 11 种 LLM — 7 种不需要 API key
225
229
 
226
230
  **你已经有这些会员了吧?直接就能跑:**
227
231
 
@@ -242,10 +246,11 @@ analyze ──→ tech_review ──→ summary
242
246
  | 提供商 | 配置 | 环境变量 |
243
247
  |--------|------|---------|
244
248
  | DeepSeek | `provider: "deepseek"` | `DEEPSEEK_API_KEY` |
249
+ | 火山引擎(豆包 / Kimi / GLM,赞助商) | `provider: "volcengine"` | `ARK_API_KEY` |
245
250
  | Claude API | `provider: "claude"` | `ANTHROPIC_API_KEY` |
246
251
  | OpenAI | `provider: "openai"` | `OPENAI_API_KEY` |
247
252
 
248
- **自定义 API(火山引擎、智谱、月之暗面、硅基流动等 OpenAI 兼容 API):**
253
+ **自定义 API(智谱、月之暗面、硅基流动等 OpenAI 兼容 API):**
249
254
 
250
255
  ```bash
251
256
  ao init --provider openai --model 模型名 \
@@ -591,7 +596,7 @@ ao-output/产品需求评审-2026-03-22/
591
596
 
592
597
  | 项目 | 定位 | 一句话 |
593
598
  |------|------|-------|
594
- | **本项目**(agency-orchestrator) | 🚀 编排引擎 | 一句话 → 216 专家协作,**几分钟出方案**(10 家 LLM / 7 免费) |
599
+ | **本项目**(agency-orchestrator) | 🚀 编排引擎 | 一句话 → 216 专家协作,**几分钟出方案**(11 家 LLM / 7 免费) |
595
600
  | [agency-agents-zh](https://github.com/jnMetaCode/agency-agents-zh) ![](https://img.shields.io/github/stars/jnMetaCode/agency-agents-zh?style=flat&label=⭐) | 🎭 中文角色库 | 216 个**即插即用** AI 专家,含 50 中国原创(小红书 / 抖音 / 飞书 / 钉钉) |
596
601
  | [agency-agents](https://github.com/msitarzewski/agency-agents) | 🎭 英文角色库 | 184 个英文 AI 角色 by [@msitarzewski](https://github.com/msitarzewski) |
597
602
  | [superpowers-zh](https://github.com/jnMetaCode/superpowers-zh) ![](https://img.shields.io/github/stars/jnMetaCode/superpowers-zh?style=flat&label=⭐) | 🧠 工作方法论 | 20 个 skills 教 AI 怎么干活(TDD / 调试 / 代码审查等) |
@@ -193,12 +193,20 @@ ${autoRun ? ' Include specific information from the user\'s description' :
193
193
  Use {{previous_output}} to reference upstream step outputs
194
194
  output: output_variable_name
195
195
  depends_on: [upstream_step_id] # Only add when there's a dependency
196
+ acceptance: | # Optional: verifiable conditions the output must satisfy (strongly recommended on the final step)
197
+ 1. First checkable condition...
198
+ 2. Second checkable condition...
196
199
 
197
200
  # When you need to ask the user something mid-run, use a human_input step (no role, actually pauses for input):
198
201
  - id: ask_step_id
199
202
  type: human_input
200
203
  prompt: "The specific question to ask the user, can reference {{variable_name}} from earlier steps"
201
204
  output: user_answer_variable # the user's answer is injected downstream as this variable
205
+
206
+ # For high-stakes decisions (finance/medical/legal/spending real money), insert an approval gate (pauses until the user signs off):
207
+ - id: approve_step_id
208
+ type: approval
209
+ prompt: "State clearly what the user is approving, can reference {{variables}}"
202
210
  \`\`\`
203
211
 
204
212
  ## Design Principles
@@ -210,6 +218,8 @@ ${autoRun ? ' Include specific information from the user\'s description' :
210
218
  - **Detailed tasks**: Task descriptions should be specific — tell the role what to do and what format to output
211
219
  ${inputsDesignPrinciple}
212
220
  - **Use human_input when the user needs to clarify something — don't have a role "ask" in its task**: if the task genuinely can't proceed without more info from the user (personal preference, choosing between options, a specific detail not in inputs — classic case: registration/enrollment-type requests where you must ask the user's specifics partway through), insert a \`type: human_input\` step to ask them, and feed the answer downstream as its output variable. Writing "ask the user X" inside a regular role step's task does NOT work — the engine won't actually pause, the model will just make up an answer
221
+ - **Acceptance criteria**: give key steps — at minimum the FINAL deliverable step — an \`acceptance:\` field with 2-5 verifiable conditions the output must satisfy. Concrete and checkable ("contains X/Y/Z sections", "every recommendation states its risk"), never vague ("high quality"). It is injected into the step's prompt AND actually enforced: after the step runs, the engine checks the output against each item, and any unmet item triggers one automatic rework round — so every item must be objectively machine-checkable from the output text alone
222
+ - **High-risk tasks need an approval gate**: for finance/investment, medical, legal, or anything spending real money / hard to reverse — insert a \`type: approval\` step before the final recommendation/execution step, so the user signs off before proceeding
213
223
  - **Final deliverable**: The last step must output the final deliverable the user wants (e.g., complete article, complete report), not review comments or suggestions. If there's a review step, it should output the revised final version, not a "list of suggestions"
214
224
  - **Clean final output (IMPORTANT)**: The LAST step's \`task\` MUST end with an explicit instruction to output ONLY the deliverable itself — no preamble/greeting, no "what I changed"/change-log, no formatting notes, no questions to the user, no suggestions to run \`ao\`/other commands, no "shall I continue?" closers. Append a line like: "⚠️ Output only the final deliverable itself — no preamble, no change-log, no meta-commentary, no questions, no tool/command suggestions." (If you genuinely need to ask the user something, use the human_input step above instead — not in the final step)
215
225
 
@@ -297,12 +307,20 @@ ${autoRun ? ' 直接包含用户需求中的具体信息' : ' 使用 {
297
307
  使用 {{previous_output}} 引用上游步骤的输出
298
308
  output: output_variable_name
299
309
  depends_on: [upstream_step_id] # 仅在有依赖时添加
310
+ acceptance: | # 可选:产出必须满足的可核对验收条件(最终交付步强烈建议写)
311
+ 1. 第一条可核对的条件…
312
+ 2. 第二条可核对的条件…
300
313
 
301
314
  # 需要向用户询问/确认信息时,用 human_input 类型的步骤(无 role,运行到这一步会真的暂停等用户输入):
302
315
  - id: ask_step_id
303
316
  type: human_input
304
317
  prompt: "向用户提的具体问题,可用 {{variable_name}} 引用之前的变量"
305
318
  output: user_answer_variable # 用户的回答会作为这个变量注入下游 task
319
+
320
+ # 高风险决策(金融/医疗/法律/花真金白银)前插入 approval 闸门(暂停等用户签字放行):
321
+ - id: approve_step_id
322
+ type: approval
323
+ prompt: "写清楚用户在批准什么,可引用 {{变量}}"
306
324
  \`\`\`
307
325
 
308
326
  ## 设计原则
@@ -314,6 +332,8 @@ ${autoRun ? ' 直接包含用户需求中的具体信息' : ' 使用 {
314
332
  - **任务详细**:task 描述要具体,告诉角色要做什么、输出什么格式
315
333
  ${inputsDesignPrinciple}
316
334
  - **需要用户澄清时用 human_input,不要指望角色在 task 里"提问"**:如果任务本质上需要用户提供额外信息才能继续(如个人偏好、多个方案里选一个、inputs 里没给的具体细节——典型例子是报名/选课/选方案类需求,中途必须问用户具体情况),插入一个 \`type: human_input\` 的步骤向用户提问,把回答作为 output 变量给下游用。普通 role 步骤的 task 里写"请问用户 XXX"是无效的——引擎不会暂停等回答,模型只会自己编一个答案
335
+ - **验收标准**:给关键步骤——至少是最终交付步——写 \`acceptance:\` 字段,列 2-5 条产出必须满足的可核对条件。要具体可查("包含 X/Y/Z 三节""每条建议都标注风险"),不要空话("高质量")。它不只注入该步 prompt——步骤跑完后引擎会**逐条真核验**,未过自动返工一轮。所以每一条都必须是仅凭产出文本就能客观判定的
336
+ - **高风险任务要加签字闸门**:涉及金融/投资、医疗、法律,或花真金白银、难以撤销的操作——在最终建议/执行步骤之前插入 \`type: approval\` 节点,让用户签字放行后才继续(重大决策必须老板拍板)
317
337
  - **最终成品**:最后一个步骤必须输出用户想要的最终成品(如完整文章、完整报告),而不是审查意见或修改建议。如果有审校步骤,审校步骤应该直接输出修改后的定稿,而不是"修改建议列表"
318
338
  - **干净的最终产出(重要)**:最后一个步骤的 \`task\` 结尾必须显式要求"只输出成品本身"——不要开场白/寒暄、不要"我改了什么/复盘/修改说明"、不要排版备注小节、不要向用户提问或请其拍板、不要建议运行 \`ao\` 或其它命令、不要"要我继续吗"之类收尾。请在该 step 的 task 末尾追加一行类似:「⚠️ 只输出最终成品本身:不要开场白、不要复盘或说明、不要向用户提问、不要建议任何命令或后续动作。」(需要问用户时用上面的 human_input 步骤,不要在最终步骤里问)
319
339
 
package/dist/cli.js CHANGED
@@ -30,6 +30,7 @@ import { t, detectLang } from './i18n.js';
30
30
  import { loadEnvFile, writeEnvFile, ensureEnvGitignored } from './utils/env-loader.js';
31
31
  import { parseDuration } from './utils/duration.js';
32
32
  import { defaultOutputDir, defaultWorkflowsDir } from './utils/paths.js';
33
+ import { rotatingSponsors } from './utils/sponsor-guide.js';
33
34
  // Auto-load ./.env (shell env wins; no overwrite)
34
35
  loadEnvFile();
35
36
  // Suppress Node's DEP0190 warning from legitimate shell:true on Windows (.cmd shims).
@@ -131,6 +132,7 @@ async function handleRun() {
131
132
  console.error(' 或: ao run --team <名字> "你的任务" # 用已保存的团队跑新任务');
132
133
  console.error(' --materialize <目录> 把开发步产出的「### 路径 + 代码围栏」文件块落盘成真实项目脚手架');
133
134
  console.error(' --export <格式> 把本次产出导出:docx/pdf/xlsx(给人)或 skill/plan(给编码 agent 执行)');
135
+ console.error(' --no-verify 关闭 acceptance 自动核验(默认:写了 acceptance 的步骤产出后自动核验,未过自动返工一轮)');
134
136
  console.error(' --compare 跑完后再跑单次基线 + 盲评,并排对比多智能体 vs 单次');
135
137
  console.error(' --judge-provider/--judge-model --compare 时指定评审模型(默认用生成模型)');
136
138
  process.exit(1);
@@ -204,6 +206,8 @@ async function handleRun() {
204
206
  const cmp = await compareWorkflowVsBaseline(resolve(filePath), inputs, {
205
207
  outputDir,
206
208
  quiet,
209
+ verify: parseVerifyFlag(),
210
+ signalFlush: true,
207
211
  genOverride: llmOverride,
208
212
  judgeLlm: judgeProvider
209
213
  ? { provider: judgeProvider, model: judgeModel, timeout: 600_000 }
@@ -219,7 +223,9 @@ async function handleRun() {
219
223
  resumeDir: resumeDir ? resolve(resumeDir) : undefined,
220
224
  fromStep,
221
225
  feedback,
226
+ verify: parseVerifyFlag(),
222
227
  llmOverride,
228
+ signalFlush: true,
223
229
  });
224
230
  // --materialize <dir>:把开发步产出的"文件块"落盘成真实项目脚手架
225
231
  const matDir = getArgValue('--materialize');
@@ -501,6 +507,8 @@ async function handleCompose() {
501
507
  }
502
508
  const result = await run(resolve(savedPath), inputs, {
503
509
  quiet: false,
510
+ signalFlush: true,
511
+ verify: parseVerifyFlag(),
504
512
  // 用 compose 时同样的 provider 执行,避免 YAML 里写的 provider 和用户实际可用的不一致
505
513
  // CLI provider 单步调用可能很慢(1-20 分钟),给足超时;用户显式 --timeout 优先
506
514
  llmOverride: {
@@ -658,6 +666,14 @@ function parseTemperatureArg() {
658
666
  }
659
667
  return temp;
660
668
  }
669
+ /** 解析 --verify/--no-verify 三态:true 强制开 / false 强制关 / undefined 按 YAML 顶层 verify(默认开)。 */
670
+ function parseVerifyFlag() {
671
+ if (args.includes('--no-verify'))
672
+ return false;
673
+ if (args.includes('--verify'))
674
+ return true;
675
+ return undefined;
676
+ }
661
677
  const COMPOSE_CLI_PROVIDERS = ['claude-code', 'gemini-cli', 'copilot-cli', 'codex-cli', 'openclaw-cli', 'hermes-cli'];
662
678
  /** R2.1:判断 compose 要用的 provider 是否已有可用凭证。保守——不确定时返回 true(不拦已能跑的配置)。 */
663
679
  function composeProviderHasCredentials(provider, apiKey, baseUrl) {
@@ -688,10 +704,13 @@ function printFirstRunGuide(provider) {
688
704
  L(` ao compose "…" --run --provider claude-code`);
689
705
  }
690
706
  L('');
691
- L(` ② 用「送额度」的中转(几十秒拿 key,一个 key 通 Claude/GPT/Gemini 全家桶):`);
692
- L(` · CCSub 注册送 $5 → https://www.ccsub.net/register?ref=8G5W4JK4`);
693
- L(` · Cubence → https://cubence.com/signup?code=SCW29JP9&source=agency`);
694
- L(` 拿到 key:ao compose "…" --run --provider ccsub --api-key <你的key>`);
707
+ L(` ② 用「送额度」的聚合/中转(几十秒拿 key,一个 key 通 Claude/GPT/Gemini 全家桶):`);
708
+ // 赞助商位规则(src/utils/sponsor-guide.ts):多元探索持有默认 provider 位不占此处;
709
+ // 这里是其余 6 家(旗舰+标准)按天轮换 2 家
710
+ const rots = rotatingSponsors();
711
+ for (const s of rots)
712
+ L(` · ${s.name}${s.bonus ? ` ${s.bonus}` : ''} → ${s.url}`);
713
+ L(` 拿到 key:ao compose "…" --run --provider ${rots[0].providerId} --api-key <你的key>`);
695
714
  L('');
696
715
  L(` ③ 本地免费跑(需先装 Ollama 并拉好模型,建议 70B+):`);
697
716
  L(` ao compose "…" --run --provider ollama --model llama3`);
@@ -784,7 +803,9 @@ async function runWithTeam(teamRef) {
784
803
  console.log(` ${t('compose.auto_running')}\n`);
785
804
  const result = await run(resolve(savedPath), {}, {
786
805
  quiet: args.includes('--quiet') || args.includes('-q'),
806
+ signalFlush: true,
787
807
  watch: args.includes('--watch'),
808
+ verify: parseVerifyFlag(),
788
809
  outputDir: getArgValue('--output') || defaultOutputDir(),
789
810
  llmOverride: {
790
811
  provider,
@@ -15,5 +15,14 @@ export const API_PROVIDERS = [
15
15
  // CCSub(赞助商)—— AI API 中转:一个 key 通 Claude / GPT / Gemini / DeepSeek 全家桶,
16
16
  // 统一端点 www.ccsub.net 同时兼容 Anthropic 与 OpenAI 协议(此处走 OpenAI 兼容 /v1)
17
17
  { id: 'ccsub', envKey: 'CCSUB_API_KEY', envBase: 'CCSUB_BASE_URL', defaultBaseUrl: 'https://www.ccsub.net/v1', defaultModel: 'claude-sonnet-5' },
18
+ // 火山引擎(赞助商)—— 字节跳动火山方舟 Ark:豆包 / Kimi / GLM 等模型。直连走 OpenAI 兼容
19
+ // 主数据面 /api/v3;key 用官方环境变量名 ARK_API_KEY(console.volcengine.com/ark 创建)。
20
+ // 给 Claude Code / Codex 配中转的另一用法见前端 CLI_RELAY_PRESETS(Anthropic 兼容 /api/compatible)。
21
+ { id: 'volcengine', envKey: 'ARK_API_KEY', envBase: 'VOLCENGINE_BASE_URL', defaultBaseUrl: 'https://ark.cn-beijing.volces.com/api/v3', defaultModel: 'doubao-seed-2-1-pro-260628' },
22
+ // 多元探索 DuoyuanX(赞助商)—— 全球 AI 模型 API 聚合与源头直供:一个 key 通 OpenAI /
23
+ // Claude / Gemini / DeepSeek 等数百款模型。OpenAI 兼容端点 duoyuanx.com/v1。
24
+ // 默认模型必须选平台实际上架且已定价的:claude-sonnet-5 未上架(报"价格尚未由管理员设置"),
25
+ // claude-sonnet-4-6 实测可用(2026-07-17 真 key 连通验证)。
26
+ { id: 'duoyuanx', envKey: 'DUOYUANX_API_KEY', envBase: 'DUOYUANX_BASE_URL', defaultBaseUrl: 'https://duoyuanx.com/v1', defaultModel: 'claude-sonnet-4-6' },
18
27
  ];
19
28
  export const API_PROVIDER_MAP = Object.fromEntries(API_PROVIDERS.map((p) => [p.id, p]));
@@ -12,8 +12,9 @@ export interface JudgeScore {
12
12
  }
13
13
  /** 从 judge 回复里抽出 JSON 分数(judge 偶尔会包代码块/加解释,宽松匹配第一个 {...})。 */
14
14
  export declare function parseJudge(raw: string): JudgeScore | null;
15
- /** 单向评审一次:A、B 两份产出对同一任务打分。解析失败重试一次。 */
16
- export declare function judgeOnce(judgeLlm: LLMConfig, taskDesc: string, outA: string, outB: string): Promise<JudgeScore | null>;
15
+ /** 单向评审一次:A、B 两份产出对同一任务打分。解析失败重试一次。
16
+ * acceptance 非空时作为首要评分锚点(工作流声明的验收标准,两份产出用同一把尺)。 */
17
+ export declare function judgeOnce(judgeLlm: LLMConfig, taskDesc: string, outA: string, outB: string, acceptance?: string): Promise<JudgeScore | null>;
17
18
  export interface CompareVerdict {
18
19
  multiScore: number;
19
20
  baseScore: number;
@@ -31,4 +32,4 @@ export declare function aggregateVerdict(j1: JudgeScore, j2: JudgeScore): Compar
31
32
  * 双向盲评对比:对 (多智能体, 基线) 正反各评一次取平均。
32
33
  * judge 解析失败返回 null(调用方决定跳过/重试)。
33
34
  */
34
- export declare function compareOutputs(judgeLlm: LLMConfig, taskDesc: string, multiOutput: string, baselineOutput: string): Promise<CompareVerdict | null>;
35
+ export declare function compareOutputs(judgeLlm: LLMConfig, taskDesc: string, multiOutput: string, baselineOutput: string, acceptance?: string): Promise<CompareVerdict | null>;
@@ -49,14 +49,18 @@ export function parseJudge(raw) {
49
49
  return null;
50
50
  }
51
51
  }
52
- /** 单向评审一次:A、B 两份产出对同一任务打分。解析失败重试一次。 */
53
- export async function judgeOnce(judgeLlm, taskDesc, outA, outB) {
52
+ /** 单向评审一次:A、B 两份产出对同一任务打分。解析失败重试一次。
53
+ * acceptance 非空时作为首要评分锚点(工作流声明的验收标准,两份产出用同一把尺)。 */
54
+ export async function judgeOnce(judgeLlm, taskDesc, outA, outB, acceptance) {
54
55
  const conn = createConnector(judgeLlm);
55
56
  const prompt = [
56
57
  '你是严格、客观的内容质量评审。下面是针对同一任务的两份产出,请对比。',
57
58
  `任务:${taskDesc}`,
59
+ ...(acceptance ? ['', `交付验收标准(首要评判依据,逐条核对两份产出是否满足):\n${acceptance}`] : []),
58
60
  '', '【产出 A】', trunc(outA), '', '【产出 B】', trunc(outB), '',
59
- '评判维度:完整性、具体性、可用性、是否直接可交付。',
61
+ acceptance
62
+ ? '评判维度:验收标准满足度优先,其次完整性、具体性、可用性、是否直接可交付。'
63
+ : '评判维度:完整性、具体性、可用性、是否直接可交付。',
60
64
  '只输出一行 JSON,不要任何额外文字:{"scoreA": 1-10, "scoreB": 1-10, "reason": "一句话理由"}',
61
65
  ].join('\n');
62
66
  for (let attempt = 0; attempt < 2; attempt++) {
@@ -89,9 +93,9 @@ export function aggregateVerdict(j1, j2) {
89
93
  * 双向盲评对比:对 (多智能体, 基线) 正反各评一次取平均。
90
94
  * judge 解析失败返回 null(调用方决定跳过/重试)。
91
95
  */
92
- export async function compareOutputs(judgeLlm, taskDesc, multiOutput, baselineOutput) {
93
- const j1 = await judgeOnce(judgeLlm, taskDesc, multiOutput, baselineOutput); // A=multi, B=base
94
- const j2 = await judgeOnce(judgeLlm, taskDesc, baselineOutput, multiOutput); // A=base, B=multi
96
+ export async function compareOutputs(judgeLlm, taskDesc, multiOutput, baselineOutput, acceptance) {
97
+ const j1 = await judgeOnce(judgeLlm, taskDesc, multiOutput, baselineOutput, acceptance); // A=multi, B=base
98
+ const j2 = await judgeOnce(judgeLlm, taskDesc, baselineOutput, multiOutput, acceptance); // A=base, B=multi
95
99
  if (!j1 || !j2)
96
100
  return null;
97
101
  return aggregateVerdict(j1, j2);
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * DAG 执行引擎 — 核心调度器
3
3
  */
4
- import type { DAGNode, LLMConnector, LLMConfig, WorkflowResult } from '../types.js';
4
+ import type { DAGNode, LLMConnector, LLMConfig, WorkflowResult, StepResult } from '../types.js';
5
5
  import type { DAG } from './dag.js';
6
6
  export interface ExecutorOptions {
7
7
  connector: LLMConnector;
@@ -27,6 +27,23 @@ export interface ExecutorOptions {
27
27
  text: string;
28
28
  previousOutput?: string;
29
29
  };
30
+ /**
31
+ * acceptance 自动核验:写了 acceptance 的步骤产出后自动逐条核对,未过则带着
32
+ * 未满足条目自动返工一轮(复用对话式返工范式)。验收不过是质量信号而非执行错误,
33
+ * 步骤不会因此 failed。产品入口(run())按 CLI flag > YAML 顶层 verify > 默认开
34
+ * 计算后传入;库级直调 executeDAG 不传 = 不核验(向后兼容)。step.verify: false 单步关闭。
35
+ */
36
+ verify?: boolean;
37
+ /**
38
+ * 调用方提供的步骤结果收集数组:executor 增量写入(每步完成即可见),
39
+ * 供 SIGTERM/SIGINT 中断时把已完成步骤落盘成 metadata(否则中断的 run 无痕)。
40
+ */
41
+ stepResultsSink?: StepResult[];
42
+ /**
43
+ * resume 复用步骤在上一次运行档案里的展示字段(agentName/acceptance/verification 等),
44
+ * 由 run() 从旧 metadata 读出传入——续跑产生的新档案才不丢被复用步骤的验收记录。
45
+ */
46
+ restoredStepMeta?: Map<string, Partial<StepResult>>;
30
47
  }
31
48
  export declare function executeDAG(dag: DAG, options: ExecutorOptions): Promise<WorkflowResult>;
32
49
  /**