agency-orchestrator 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.en.md +14 -8
- package/README.md +9 -4
- package/dist/cli/compose.js +20 -0
- package/dist/cli.js +25 -4
- package/dist/connectors/api-providers.js +9 -0
- package/dist/core/compare.d.ts +4 -3
- package/dist/core/compare.js +10 -6
- package/dist/core/executor.d.ts +18 -1
- package/dist/core/executor.js +149 -51
- package/dist/core/parser.js +15 -1
- package/dist/core/verify.d.ts +44 -0
- package/dist/core/verify.js +130 -0
- package/dist/index.d.ts +15 -0
- package/dist/index.js +66 -2
- package/dist/output/reporter.d.ts +6 -1
- package/dist/output/reporter.js +33 -3
- package/dist/types.d.ts +18 -0
- package/dist/utils/sponsor-guide.d.ts +22 -0
- package/dist/utils/sponsor-guide.js +30 -0
- package/package.json +5 -4
- package/web/server.js +255 -51
- package/website/dist/apple-touch-icon.png +0 -0
- package/website/dist/assets/Changelog-BIGmNgWM.js +8 -0
- package/website/dist/assets/{CreativeLibrary-Bafp2-7y.js → CreativeLibrary-BHqHEQIh.js} +1 -1
- package/website/dist/assets/{Docs-DWjEFgqb.js → Docs-C_Bg8DeD.js} +17 -10
- package/website/dist/assets/{Experts-ARxT7R0z.js → Experts-ByRxvzGb.js} +1 -1
- package/website/dist/assets/{Home-D1fuy-e9.js → Home-UwklCf3w.js} +1 -1
- package/website/dist/assets/{Markdown-DIBk11T6.js → Markdown-DG2oIi7J.js} +1 -1
- package/website/dist/assets/{NotFound-yaONXr2c.js → NotFound-D8ohn3mE.js} +1 -1
- package/website/dist/assets/{PromptStudio-DdCuT1Il.js → PromptStudio-8ZbqqsfO.js} +1 -1
- package/website/dist/assets/SiteFooter-CEbbrx2O.js +1 -0
- package/website/dist/assets/Sponsors-L7_xxpqq.js +21 -0
- package/website/dist/assets/Studio-BHdFIqMb.js +188 -0
- package/website/dist/assets/{TutorialDetail-C8bjE0Uh.js → TutorialDetail-BxE_n_aG.js} +1 -1
- package/website/dist/assets/{Tutorials-CscSAnuX.js → Tutorials-C2ZzhsJ8.js} +1 -1
- package/website/dist/assets/{UsagePanel-CFfVmSwy.js → UsagePanel-pKvnMg0M.js} +1 -1
- package/website/dist/assets/{arrow-left-CBsGQe3h.js → arrow-left-y9GCeSb5.js} +1 -1
- package/website/dist/assets/{arrow-up-right-LSsttJsJ.js → arrow-up-right-BSpKIm_F.js} +1 -1
- package/website/dist/assets/{badge-xc3rfZGN.js → badge-BWe9-A6x.js} +1 -1
- package/website/dist/assets/{clock-LLdpHBz2.js → clock-CY07fup8.js} +1 -1
- package/website/dist/assets/{copy-nrJtFYdc.js → copy-CDpr96ho.js} +1 -1
- package/website/dist/assets/{copy-button-CiQbKrJI.js → copy-button-DmWvJste.js} +1 -1
- package/website/dist/assets/{download-DP_pAg1E.js → download-BxGUc2TT.js} +1 -1
- package/website/dist/assets/{external-link-CS2XMbHw.js → external-link-qWj52f9i.js} +1 -1
- package/website/dist/assets/index-B0DLOSbW.js +97 -0
- package/website/dist/assets/index-Bvmb-uHp.css +1 -0
- package/website/dist/assets/{mail-Bx74VrG-.js → mail-E8-9lFeI.js} +1 -1
- package/website/dist/assets/{search-BH4oWAhq.js → search-ChBwKsLg.js} +1 -1
- package/website/dist/assets/{sparkles-jiZMT7tJ.js → sparkles-Df-uhv9v.js} +1 -1
- package/website/dist/assets/sponsors-DYxHA2Hi.js +11 -0
- package/website/dist/assets/useBackend-CMxh6Cx4.js +30 -0
- package/website/dist/assets/{workflow-D0-wMIOl.js → workflow-D3exvF90.js} +1 -1
- package/website/dist/assets/workflows-BS7rjs0g.js +1 -0
- package/website/dist/favicon.ico +0 -0
- package/website/dist/favicon.svg +15 -0
- package/website/dist/index.html +11 -2
- package/website/dist/logo/ao-app-icon.svg +15 -0
- package/website/dist/logo/ao-banner.png +0 -0
- package/website/dist/logo/ao-mark-white.svg +6 -0
- package/website/dist/logo/ao-mark.svg +13 -0
- package/website/dist/og-image.png +0 -0
- package/website/dist/sponsors/logo-byteplus-icon.png +0 -0
- package/website/dist/sponsors/logo-duoyuanx-icon.png +0 -0
- package/website/dist/sponsors/logo-volcengine-icon.png +0 -0
- package/workflows//344/270/200/344/272/272/345/205/254/345/217/270-/345/201/232/344/272/247/345/223/201.yaml +112 -0
- package/workflows//344/270/200/344/272/272/345/205/254/345/217/270-/345/201/232/345/206/205/345/256/271.yaml +101 -0
- package/workflows//344/270/200/344/272/272/345/205/254/345/217/270-/345/201/232/346/212/225/347/240/224.yaml +125 -0
- package/workflows//344/270/200/344/272/272/345/205/254/345/217/270/345/205/250/345/221/230/345/244/247/344/274/232.yaml +2 -0
- package/website/dist/assets/Changelog-Cd1_ta-F.js +0 -8
- package/website/dist/assets/SiteFooter-CuZFXiZH.js +0 -1
- package/website/dist/assets/Sponsors-D_4M9Xth.js +0 -21
- package/website/dist/assets/Studio-BFYs1jm4.js +0 -182
- package/website/dist/assets/index-DawHdnrM.js +0 -97
- package/website/dist/assets/index-GFHogxlv.css +0 -1
- package/website/dist/assets/sponsors-2Ed3J2D2.js +0 -11
- package/website/dist/assets/useBackend-DAjkH7qN.js +0 -30
- package/website/dist/assets/workflows-OVJ6BxVu.js +0 -1
package/README.en.md
CHANGED
|
@@ -3,13 +3,15 @@
|
|
|
3
3
|
**English** | [中文](./README.md)
|
|
4
4
|
|
|
5
5
|
> **One sentence in, a full plan out — multiple AI roles collaborate automatically.**
|
|
6
|
+
>
|
|
7
|
+
> **It's your one-person company: you're the boss, AI is the team — auto-assembled, asking your sign-off on big calls, delivering against acceptance criteria.**
|
|
6
8
|
|
|
7
9
|
[](https://github.com/jnMetaCode/agency-orchestrator/actions)
|
|
8
10
|
[](https://www.npmjs.com/package/agency-orchestrator)
|
|
9
11
|
[](./LICENSE)
|
|
10
12
|
[](./CONTRIBUTING.md)
|
|
11
13
|
|
|
12
|
-
**One sentence → full plan · 216 expert AI roles · Zero-code YAML ·
|
|
14
|
+
**One sentence → full plan · 216 expert AI roles · Zero-code YAML · 11 LLM providers · key supported (DeepSeek recommended), plus 7 key-free options**
|
|
13
15
|
|
|
14
16
|
> **Note:** `ao compose --run` auto-detects your language. Both 216 Chinese roles and 184 English roles ([agency-agents](https://github.com/msitarzewski/agency-agents), MIT) are **bundled in the npm package — no extra download needed**. **10 English workflow templates** are ready in `workflows/en/`.
|
|
15
17
|
|
|
@@ -20,7 +22,8 @@
|
|
|
20
22
|
> If you find this useful, please **Star** it — helps others discover the project.
|
|
21
23
|
|
|
22
24
|
<p align="center">
|
|
23
|
-
<img src="./demo.gif" alt="
|
|
25
|
+
<img src="./demo-studio-en.gif" alt="Web Studio: one sentence, AI builds the team" width="820"><br/>
|
|
26
|
+
<em>Web Studio: type one sentence, AI builds a team from 200+ experts and runs it</em>
|
|
24
27
|
</p>
|
|
25
28
|
|
|
26
29
|
---
|
|
@@ -29,11 +32,6 @@
|
|
|
29
32
|
|
|
30
33
|
Prefer not to use the command line? Run `ao web` locally and pick experts, run workflows, view outputs, and intervene live — all in a GUI, fully bilingual (EN/中文).
|
|
31
34
|
|
|
32
|
-
<p align="center">
|
|
33
|
-
<img src="./docs/screenshots/studio-roles-en.png" alt="Studio · Build a Team: pick from 200+ experts, AI composes the team" width="800"><br/>
|
|
34
|
-
<em>Build a Team: pick from 200+ experts; AI composes and runs the team</em>
|
|
35
|
-
</p>
|
|
36
|
-
|
|
37
35
|
<p align="center">
|
|
38
36
|
<img src="./docs/screenshots/studio-workflows-en.png" alt="Studio · Workflows: run built-in templates with one click" width="800"><br/>
|
|
39
37
|
<em>Workflows: run built-in templates with one click, or compare several</em>
|
|
@@ -45,10 +43,16 @@ Prefer not to use the command line? Run `ao web` locally and pick experts, run w
|
|
|
45
43
|
|
|
46
44
|
## One Sentence, Full Result
|
|
47
45
|
|
|
46
|
+
Prefer the command line? One command, one sentence, full result:
|
|
47
|
+
|
|
48
48
|
```bash
|
|
49
49
|
ao compose "I'm a programmer looking to start a side hustle with AI content, target $3K/month, give me a complete plan" --run
|
|
50
50
|
```
|
|
51
51
|
|
|
52
|
+
<p align="center">
|
|
53
|
+
<img src="./demo.gif" alt="ao compose CLI demo" width="700">
|
|
54
|
+
</p>
|
|
55
|
+
|
|
52
56
|
5 AI roles collaborate automatically:
|
|
53
57
|
|
|
54
58
|
```
|
|
@@ -191,6 +195,7 @@ steps:
|
|
|
191
195
|
- id: summary
|
|
192
196
|
role: "product/product-manager"
|
|
193
197
|
task: "Synthesize feedback:\n\n{{tech_report}}\n\n{{design_report}}"
|
|
198
|
+
acceptance: "1. Clear go/no-go verdict 2. List must-fix issues" # optional: injected at prompt tail, used as review yardstick
|
|
194
199
|
depends_on: [tech_review, design_review]
|
|
195
200
|
```
|
|
196
201
|
|
|
@@ -209,7 +214,7 @@ analyze ──→ tech_review ──→ summary
|
|
|
209
214
|
(parallel)
|
|
210
215
|
```
|
|
211
216
|
|
|
212
|
-
##
|
|
217
|
+
## 11 LLM Providers — 7 Need No API Key
|
|
213
218
|
|
|
214
219
|
**Already paying for one of these? You're ready to go:**
|
|
215
220
|
|
|
@@ -230,6 +235,7 @@ analyze ──→ tech_review ──→ summary
|
|
|
230
235
|
| Provider | Config | Env Variable |
|
|
231
236
|
|----------|--------|-------------|
|
|
232
237
|
| DeepSeek | `provider: "deepseek"` | `DEEPSEEK_API_KEY` |
|
|
238
|
+
| Volcengine Ark (Doubao / Kimi / GLM · sponsor) | `provider: "volcengine"` | `ARK_API_KEY` |
|
|
233
239
|
| Claude API | `provider: "claude"` | `ANTHROPIC_API_KEY` |
|
|
234
240
|
| OpenAI | `provider: "openai"` | `OPENAI_API_KEY` |
|
|
235
241
|
|
package/README.md
CHANGED
|
@@ -3,13 +3,15 @@
|
|
|
3
3
|
**中文** | [English](./README.en.md)
|
|
4
4
|
|
|
5
5
|
> **一句话,让多个 AI 角色自动协作,几分钟出完整方案。**
|
|
6
|
+
>
|
|
7
|
+
> **也是你的「一人公司」:你当老板,AI 当团队——自动组队、重大决策请你签字、按验收标准交付。**
|
|
6
8
|
|
|
7
9
|
[](https://github.com/jnMetaCode/agency-orchestrator/actions)
|
|
8
10
|
[](https://www.npmjs.com/package/agency-orchestrator)
|
|
9
11
|
[](./LICENSE)
|
|
10
12
|
[](./CONTRIBUTING.md)
|
|
11
13
|
|
|
12
|
-
**一句话出结果 · 216 个专业 AI 角色 · YAML 零代码 ·
|
|
14
|
+
**一句话出结果 · 216 个专业 AI 角色 · YAML 零代码 · 11 种大模型 · 支持 key(推荐 DeepSeek),也有 7 种免 key 方式**
|
|
13
15
|
|
|
14
16
|
> 📖 [完整上手教程](https://mp.weixin.qq.com/s/XcGbkMb6TM6NLQiL7ICwbw) — 从安装到实战,10 分钟上手
|
|
15
17
|
>
|
|
@@ -28,6 +30,7 @@
|
|
|
28
30
|
|
|
29
31
|
不想敲命令行?本地跑一条 `ao web`,浏览器里勾选专家、运行工作流、查看产物、实时介入——全程图形界面,全中英双语。
|
|
30
32
|
|
|
33
|
+
> 🆕 **「一人公司」系列模板**:做产品 / 做内容 / 做投研 + 全员大会,关键步骤带验收标准(`acceptance`),投研含老板签字闸门——交付的是可验收的工作成果,不承诺奇迹。
|
|
31
34
|
> 🆕 **AI 自动组队**:不知道选哪些专家?角色页一句话、不选角色,AI 自动从全部专家里挑人组队并运行。
|
|
32
35
|
> 🆕 **可视化画布**:工作流可在画布上拖拽节点 / 连线(自动防环)/ 改任务·角色 / 保存,运行时节点按状态实时点亮。
|
|
33
36
|
> 🆕 **创意库**:内置图像生成提示词库(Nano Banana / Gemini,可搜索 / 分类 / 一键复制)。
|
|
@@ -203,6 +206,7 @@ steps:
|
|
|
203
206
|
- id: summary
|
|
204
207
|
role: "product/product-manager"
|
|
205
208
|
task: "综合反馈输出结论:\n\n{{tech_report}}\n\n{{design_report}}"
|
|
209
|
+
acceptance: "1. 明确给出通过/不通过结论 2. 列出必须解决的问题" # 可选:验收标准,注入 prompt 并作评审依据
|
|
206
210
|
depends_on: [tech_review, design_review]
|
|
207
211
|
```
|
|
208
212
|
|
|
@@ -221,7 +225,7 @@ analyze ──→ tech_review ──→ summary
|
|
|
221
225
|
(并行)
|
|
222
226
|
```
|
|
223
227
|
|
|
224
|
-
##
|
|
228
|
+
## 11 种 LLM — 7 种不需要 API key
|
|
225
229
|
|
|
226
230
|
**你已经有这些会员了吧?直接就能跑:**
|
|
227
231
|
|
|
@@ -242,10 +246,11 @@ analyze ──→ tech_review ──→ summary
|
|
|
242
246
|
| 提供商 | 配置 | 环境变量 |
|
|
243
247
|
|--------|------|---------|
|
|
244
248
|
| DeepSeek | `provider: "deepseek"` | `DEEPSEEK_API_KEY` |
|
|
249
|
+
| 火山引擎(豆包 / Kimi / GLM,赞助商) | `provider: "volcengine"` | `ARK_API_KEY` |
|
|
245
250
|
| Claude API | `provider: "claude"` | `ANTHROPIC_API_KEY` |
|
|
246
251
|
| OpenAI | `provider: "openai"` | `OPENAI_API_KEY` |
|
|
247
252
|
|
|
248
|
-
**自定义 API
|
|
253
|
+
**自定义 API(智谱、月之暗面、硅基流动等 OpenAI 兼容 API):**
|
|
249
254
|
|
|
250
255
|
```bash
|
|
251
256
|
ao init --provider openai --model 模型名 \
|
|
@@ -591,7 +596,7 @@ ao-output/产品需求评审-2026-03-22/
|
|
|
591
596
|
|
|
592
597
|
| 项目 | 定位 | 一句话 |
|
|
593
598
|
|------|------|-------|
|
|
594
|
-
| **本项目**(agency-orchestrator) | 🚀 编排引擎 | 一句话 → 216 专家协作,**几分钟出方案**(
|
|
599
|
+
| **本项目**(agency-orchestrator) | 🚀 编排引擎 | 一句话 → 216 专家协作,**几分钟出方案**(11 家 LLM / 7 免费) |
|
|
595
600
|
| [agency-agents-zh](https://github.com/jnMetaCode/agency-agents-zh)  | 🎭 中文角色库 | 216 个**即插即用** AI 专家,含 50 中国原创(小红书 / 抖音 / 飞书 / 钉钉) |
|
|
596
601
|
| [agency-agents](https://github.com/msitarzewski/agency-agents) | 🎭 英文角色库 | 184 个英文 AI 角色 by [@msitarzewski](https://github.com/msitarzewski) |
|
|
597
602
|
| [superpowers-zh](https://github.com/jnMetaCode/superpowers-zh)  | 🧠 工作方法论 | 20 个 skills 教 AI 怎么干活(TDD / 调试 / 代码审查等) |
|
package/dist/cli/compose.js
CHANGED
|
@@ -193,12 +193,20 @@ ${autoRun ? ' Include specific information from the user\'s description' :
|
|
|
193
193
|
Use {{previous_output}} to reference upstream step outputs
|
|
194
194
|
output: output_variable_name
|
|
195
195
|
depends_on: [upstream_step_id] # Only add when there's a dependency
|
|
196
|
+
acceptance: | # Optional: verifiable conditions the output must satisfy (strongly recommended on the final step)
|
|
197
|
+
1. First checkable condition...
|
|
198
|
+
2. Second checkable condition...
|
|
196
199
|
|
|
197
200
|
# When you need to ask the user something mid-run, use a human_input step (no role, actually pauses for input):
|
|
198
201
|
- id: ask_step_id
|
|
199
202
|
type: human_input
|
|
200
203
|
prompt: "The specific question to ask the user, can reference {{variable_name}} from earlier steps"
|
|
201
204
|
output: user_answer_variable # the user's answer is injected downstream as this variable
|
|
205
|
+
|
|
206
|
+
# For high-stakes decisions (finance/medical/legal/spending real money), insert an approval gate (pauses until the user signs off):
|
|
207
|
+
- id: approve_step_id
|
|
208
|
+
type: approval
|
|
209
|
+
prompt: "State clearly what the user is approving, can reference {{variables}}"
|
|
202
210
|
\`\`\`
|
|
203
211
|
|
|
204
212
|
## Design Principles
|
|
@@ -210,6 +218,8 @@ ${autoRun ? ' Include specific information from the user\'s description' :
|
|
|
210
218
|
- **Detailed tasks**: Task descriptions should be specific — tell the role what to do and what format to output
|
|
211
219
|
${inputsDesignPrinciple}
|
|
212
220
|
- **Use human_input when the user needs to clarify something — don't have a role "ask" in its task**: if the task genuinely can't proceed without more info from the user (personal preference, choosing between options, a specific detail not in inputs — classic case: registration/enrollment-type requests where you must ask the user's specifics partway through), insert a \`type: human_input\` step to ask them, and feed the answer downstream as its output variable. Writing "ask the user X" inside a regular role step's task does NOT work — the engine won't actually pause, the model will just make up an answer
|
|
221
|
+
- **Acceptance criteria**: give key steps — at minimum the FINAL deliverable step — an \`acceptance:\` field with 2-5 verifiable conditions the output must satisfy. Concrete and checkable ("contains X/Y/Z sections", "every recommendation states its risk"), never vague ("high quality"). It is injected into the step's prompt AND actually enforced: after the step runs, the engine checks the output against each item, and any unmet item triggers one automatic rework round — so every item must be objectively machine-checkable from the output text alone
|
|
222
|
+
- **High-risk tasks need an approval gate**: for finance/investment, medical, legal, or anything spending real money / hard to reverse — insert a \`type: approval\` step before the final recommendation/execution step, so the user signs off before proceeding
|
|
213
223
|
- **Final deliverable**: The last step must output the final deliverable the user wants (e.g., complete article, complete report), not review comments or suggestions. If there's a review step, it should output the revised final version, not a "list of suggestions"
|
|
214
224
|
- **Clean final output (IMPORTANT)**: The LAST step's \`task\` MUST end with an explicit instruction to output ONLY the deliverable itself — no preamble/greeting, no "what I changed"/change-log, no formatting notes, no questions to the user, no suggestions to run \`ao\`/other commands, no "shall I continue?" closers. Append a line like: "⚠️ Output only the final deliverable itself — no preamble, no change-log, no meta-commentary, no questions, no tool/command suggestions." (If you genuinely need to ask the user something, use the human_input step above instead — not in the final step)
|
|
215
225
|
|
|
@@ -297,12 +307,20 @@ ${autoRun ? ' 直接包含用户需求中的具体信息' : ' 使用 {
|
|
|
297
307
|
使用 {{previous_output}} 引用上游步骤的输出
|
|
298
308
|
output: output_variable_name
|
|
299
309
|
depends_on: [upstream_step_id] # 仅在有依赖时添加
|
|
310
|
+
acceptance: | # 可选:产出必须满足的可核对验收条件(最终交付步强烈建议写)
|
|
311
|
+
1. 第一条可核对的条件…
|
|
312
|
+
2. 第二条可核对的条件…
|
|
300
313
|
|
|
301
314
|
# 需要向用户询问/确认信息时,用 human_input 类型的步骤(无 role,运行到这一步会真的暂停等用户输入):
|
|
302
315
|
- id: ask_step_id
|
|
303
316
|
type: human_input
|
|
304
317
|
prompt: "向用户提的具体问题,可用 {{variable_name}} 引用之前的变量"
|
|
305
318
|
output: user_answer_variable # 用户的回答会作为这个变量注入下游 task
|
|
319
|
+
|
|
320
|
+
# 高风险决策(金融/医疗/法律/花真金白银)前插入 approval 闸门(暂停等用户签字放行):
|
|
321
|
+
- id: approve_step_id
|
|
322
|
+
type: approval
|
|
323
|
+
prompt: "写清楚用户在批准什么,可引用 {{变量}}"
|
|
306
324
|
\`\`\`
|
|
307
325
|
|
|
308
326
|
## 设计原则
|
|
@@ -314,6 +332,8 @@ ${autoRun ? ' 直接包含用户需求中的具体信息' : ' 使用 {
|
|
|
314
332
|
- **任务详细**:task 描述要具体,告诉角色要做什么、输出什么格式
|
|
315
333
|
${inputsDesignPrinciple}
|
|
316
334
|
- **需要用户澄清时用 human_input,不要指望角色在 task 里"提问"**:如果任务本质上需要用户提供额外信息才能继续(如个人偏好、多个方案里选一个、inputs 里没给的具体细节——典型例子是报名/选课/选方案类需求,中途必须问用户具体情况),插入一个 \`type: human_input\` 的步骤向用户提问,把回答作为 output 变量给下游用。普通 role 步骤的 task 里写"请问用户 XXX"是无效的——引擎不会暂停等回答,模型只会自己编一个答案
|
|
335
|
+
- **验收标准**:给关键步骤——至少是最终交付步——写 \`acceptance:\` 字段,列 2-5 条产出必须满足的可核对条件。要具体可查("包含 X/Y/Z 三节""每条建议都标注风险"),不要空话("高质量")。它不只注入该步 prompt——步骤跑完后引擎会**逐条真核验**,未过自动返工一轮。所以每一条都必须是仅凭产出文本就能客观判定的
|
|
336
|
+
- **高风险任务要加签字闸门**:涉及金融/投资、医疗、法律,或花真金白银、难以撤销的操作——在最终建议/执行步骤之前插入 \`type: approval\` 节点,让用户签字放行后才继续(重大决策必须老板拍板)
|
|
317
337
|
- **最终成品**:最后一个步骤必须输出用户想要的最终成品(如完整文章、完整报告),而不是审查意见或修改建议。如果有审校步骤,审校步骤应该直接输出修改后的定稿,而不是"修改建议列表"
|
|
318
338
|
- **干净的最终产出(重要)**:最后一个步骤的 \`task\` 结尾必须显式要求"只输出成品本身"——不要开场白/寒暄、不要"我改了什么/复盘/修改说明"、不要排版备注小节、不要向用户提问或请其拍板、不要建议运行 \`ao\` 或其它命令、不要"要我继续吗"之类收尾。请在该 step 的 task 末尾追加一行类似:「⚠️ 只输出最终成品本身:不要开场白、不要复盘或说明、不要向用户提问、不要建议任何命令或后续动作。」(需要问用户时用上面的 human_input 步骤,不要在最终步骤里问)
|
|
319
339
|
|
package/dist/cli.js
CHANGED
|
@@ -30,6 +30,7 @@ import { t, detectLang } from './i18n.js';
|
|
|
30
30
|
import { loadEnvFile, writeEnvFile, ensureEnvGitignored } from './utils/env-loader.js';
|
|
31
31
|
import { parseDuration } from './utils/duration.js';
|
|
32
32
|
import { defaultOutputDir, defaultWorkflowsDir } from './utils/paths.js';
|
|
33
|
+
import { rotatingSponsors } from './utils/sponsor-guide.js';
|
|
33
34
|
// Auto-load ./.env (shell env wins; no overwrite)
|
|
34
35
|
loadEnvFile();
|
|
35
36
|
// Suppress Node's DEP0190 warning from legitimate shell:true on Windows (.cmd shims).
|
|
@@ -131,6 +132,7 @@ async function handleRun() {
|
|
|
131
132
|
console.error(' 或: ao run --team <名字> "你的任务" # 用已保存的团队跑新任务');
|
|
132
133
|
console.error(' --materialize <目录> 把开发步产出的「### 路径 + 代码围栏」文件块落盘成真实项目脚手架');
|
|
133
134
|
console.error(' --export <格式> 把本次产出导出:docx/pdf/xlsx(给人)或 skill/plan(给编码 agent 执行)');
|
|
135
|
+
console.error(' --no-verify 关闭 acceptance 自动核验(默认:写了 acceptance 的步骤产出后自动核验,未过自动返工一轮)');
|
|
134
136
|
console.error(' --compare 跑完后再跑单次基线 + 盲评,并排对比多智能体 vs 单次');
|
|
135
137
|
console.error(' --judge-provider/--judge-model --compare 时指定评审模型(默认用生成模型)');
|
|
136
138
|
process.exit(1);
|
|
@@ -204,6 +206,8 @@ async function handleRun() {
|
|
|
204
206
|
const cmp = await compareWorkflowVsBaseline(resolve(filePath), inputs, {
|
|
205
207
|
outputDir,
|
|
206
208
|
quiet,
|
|
209
|
+
verify: parseVerifyFlag(),
|
|
210
|
+
signalFlush: true,
|
|
207
211
|
genOverride: llmOverride,
|
|
208
212
|
judgeLlm: judgeProvider
|
|
209
213
|
? { provider: judgeProvider, model: judgeModel, timeout: 600_000 }
|
|
@@ -219,7 +223,9 @@ async function handleRun() {
|
|
|
219
223
|
resumeDir: resumeDir ? resolve(resumeDir) : undefined,
|
|
220
224
|
fromStep,
|
|
221
225
|
feedback,
|
|
226
|
+
verify: parseVerifyFlag(),
|
|
222
227
|
llmOverride,
|
|
228
|
+
signalFlush: true,
|
|
223
229
|
});
|
|
224
230
|
// --materialize <dir>:把开发步产出的"文件块"落盘成真实项目脚手架
|
|
225
231
|
const matDir = getArgValue('--materialize');
|
|
@@ -501,6 +507,8 @@ async function handleCompose() {
|
|
|
501
507
|
}
|
|
502
508
|
const result = await run(resolve(savedPath), inputs, {
|
|
503
509
|
quiet: false,
|
|
510
|
+
signalFlush: true,
|
|
511
|
+
verify: parseVerifyFlag(),
|
|
504
512
|
// 用 compose 时同样的 provider 执行,避免 YAML 里写的 provider 和用户实际可用的不一致
|
|
505
513
|
// CLI provider 单步调用可能很慢(1-20 分钟),给足超时;用户显式 --timeout 优先
|
|
506
514
|
llmOverride: {
|
|
@@ -658,6 +666,14 @@ function parseTemperatureArg() {
|
|
|
658
666
|
}
|
|
659
667
|
return temp;
|
|
660
668
|
}
|
|
669
|
+
/** 解析 --verify/--no-verify 三态:true 强制开 / false 强制关 / undefined 按 YAML 顶层 verify(默认开)。 */
|
|
670
|
+
function parseVerifyFlag() {
|
|
671
|
+
if (args.includes('--no-verify'))
|
|
672
|
+
return false;
|
|
673
|
+
if (args.includes('--verify'))
|
|
674
|
+
return true;
|
|
675
|
+
return undefined;
|
|
676
|
+
}
|
|
661
677
|
const COMPOSE_CLI_PROVIDERS = ['claude-code', 'gemini-cli', 'copilot-cli', 'codex-cli', 'openclaw-cli', 'hermes-cli'];
|
|
662
678
|
/** R2.1:判断 compose 要用的 provider 是否已有可用凭证。保守——不确定时返回 true(不拦已能跑的配置)。 */
|
|
663
679
|
function composeProviderHasCredentials(provider, apiKey, baseUrl) {
|
|
@@ -688,10 +704,13 @@ function printFirstRunGuide(provider) {
|
|
|
688
704
|
L(` ao compose "…" --run --provider claude-code`);
|
|
689
705
|
}
|
|
690
706
|
L('');
|
|
691
|
-
L(` ②
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
|
|
707
|
+
L(` ② 用「送额度」的聚合/中转(几十秒拿 key,一个 key 通 Claude/GPT/Gemini 全家桶):`);
|
|
708
|
+
// 赞助商位规则(src/utils/sponsor-guide.ts):多元探索持有默认 provider 位不占此处;
|
|
709
|
+
// 这里是其余 6 家(旗舰+标准)按天轮换 2 家
|
|
710
|
+
const rots = rotatingSponsors();
|
|
711
|
+
for (const s of rots)
|
|
712
|
+
L(` · ${s.name}${s.bonus ? ` ${s.bonus}` : ''} → ${s.url}`);
|
|
713
|
+
L(` 拿到 key:ao compose "…" --run --provider ${rots[0].providerId} --api-key <你的key>`);
|
|
695
714
|
L('');
|
|
696
715
|
L(` ③ 本地免费跑(需先装 Ollama 并拉好模型,建议 70B+):`);
|
|
697
716
|
L(` ao compose "…" --run --provider ollama --model llama3`);
|
|
@@ -784,7 +803,9 @@ async function runWithTeam(teamRef) {
|
|
|
784
803
|
console.log(` ${t('compose.auto_running')}\n`);
|
|
785
804
|
const result = await run(resolve(savedPath), {}, {
|
|
786
805
|
quiet: args.includes('--quiet') || args.includes('-q'),
|
|
806
|
+
signalFlush: true,
|
|
787
807
|
watch: args.includes('--watch'),
|
|
808
|
+
verify: parseVerifyFlag(),
|
|
788
809
|
outputDir: getArgValue('--output') || defaultOutputDir(),
|
|
789
810
|
llmOverride: {
|
|
790
811
|
provider,
|
|
@@ -15,5 +15,14 @@ export const API_PROVIDERS = [
|
|
|
15
15
|
// CCSub(赞助商)—— AI API 中转:一个 key 通 Claude / GPT / Gemini / DeepSeek 全家桶,
|
|
16
16
|
// 统一端点 www.ccsub.net 同时兼容 Anthropic 与 OpenAI 协议(此处走 OpenAI 兼容 /v1)
|
|
17
17
|
{ id: 'ccsub', envKey: 'CCSUB_API_KEY', envBase: 'CCSUB_BASE_URL', defaultBaseUrl: 'https://www.ccsub.net/v1', defaultModel: 'claude-sonnet-5' },
|
|
18
|
+
// 火山引擎(赞助商)—— 字节跳动火山方舟 Ark:豆包 / Kimi / GLM 等模型。直连走 OpenAI 兼容
|
|
19
|
+
// 主数据面 /api/v3;key 用官方环境变量名 ARK_API_KEY(console.volcengine.com/ark 创建)。
|
|
20
|
+
// 给 Claude Code / Codex 配中转的另一用法见前端 CLI_RELAY_PRESETS(Anthropic 兼容 /api/compatible)。
|
|
21
|
+
{ id: 'volcengine', envKey: 'ARK_API_KEY', envBase: 'VOLCENGINE_BASE_URL', defaultBaseUrl: 'https://ark.cn-beijing.volces.com/api/v3', defaultModel: 'doubao-seed-2-1-pro-260628' },
|
|
22
|
+
// 多元探索 DuoyuanX(赞助商)—— 全球 AI 模型 API 聚合与源头直供:一个 key 通 OpenAI /
|
|
23
|
+
// Claude / Gemini / DeepSeek 等数百款模型。OpenAI 兼容端点 duoyuanx.com/v1。
|
|
24
|
+
// 默认模型必须选平台实际上架且已定价的:claude-sonnet-5 未上架(报"价格尚未由管理员设置"),
|
|
25
|
+
// claude-sonnet-4-6 实测可用(2026-07-17 真 key 连通验证)。
|
|
26
|
+
{ id: 'duoyuanx', envKey: 'DUOYUANX_API_KEY', envBase: 'DUOYUANX_BASE_URL', defaultBaseUrl: 'https://duoyuanx.com/v1', defaultModel: 'claude-sonnet-4-6' },
|
|
18
27
|
];
|
|
19
28
|
export const API_PROVIDER_MAP = Object.fromEntries(API_PROVIDERS.map((p) => [p.id, p]));
|
package/dist/core/compare.d.ts
CHANGED
|
@@ -12,8 +12,9 @@ export interface JudgeScore {
|
|
|
12
12
|
}
|
|
13
13
|
/** 从 judge 回复里抽出 JSON 分数(judge 偶尔会包代码块/加解释,宽松匹配第一个 {...})。 */
|
|
14
14
|
export declare function parseJudge(raw: string): JudgeScore | null;
|
|
15
|
-
/** 单向评审一次:A、B 两份产出对同一任务打分。解析失败重试一次。
|
|
16
|
-
|
|
15
|
+
/** 单向评审一次:A、B 两份产出对同一任务打分。解析失败重试一次。
|
|
16
|
+
* acceptance 非空时作为首要评分锚点(工作流声明的验收标准,两份产出用同一把尺)。 */
|
|
17
|
+
export declare function judgeOnce(judgeLlm: LLMConfig, taskDesc: string, outA: string, outB: string, acceptance?: string): Promise<JudgeScore | null>;
|
|
17
18
|
export interface CompareVerdict {
|
|
18
19
|
multiScore: number;
|
|
19
20
|
baseScore: number;
|
|
@@ -31,4 +32,4 @@ export declare function aggregateVerdict(j1: JudgeScore, j2: JudgeScore): Compar
|
|
|
31
32
|
* 双向盲评对比:对 (多智能体, 基线) 正反各评一次取平均。
|
|
32
33
|
* judge 解析失败返回 null(调用方决定跳过/重试)。
|
|
33
34
|
*/
|
|
34
|
-
export declare function compareOutputs(judgeLlm: LLMConfig, taskDesc: string, multiOutput: string, baselineOutput: string): Promise<CompareVerdict | null>;
|
|
35
|
+
export declare function compareOutputs(judgeLlm: LLMConfig, taskDesc: string, multiOutput: string, baselineOutput: string, acceptance?: string): Promise<CompareVerdict | null>;
|
package/dist/core/compare.js
CHANGED
|
@@ -49,14 +49,18 @@ export function parseJudge(raw) {
|
|
|
49
49
|
return null;
|
|
50
50
|
}
|
|
51
51
|
}
|
|
52
|
-
/** 单向评审一次:A、B 两份产出对同一任务打分。解析失败重试一次。
|
|
53
|
-
|
|
52
|
+
/** 单向评审一次:A、B 两份产出对同一任务打分。解析失败重试一次。
|
|
53
|
+
* acceptance 非空时作为首要评分锚点(工作流声明的验收标准,两份产出用同一把尺)。 */
|
|
54
|
+
export async function judgeOnce(judgeLlm, taskDesc, outA, outB, acceptance) {
|
|
54
55
|
const conn = createConnector(judgeLlm);
|
|
55
56
|
const prompt = [
|
|
56
57
|
'你是严格、客观的内容质量评审。下面是针对同一任务的两份产出,请对比。',
|
|
57
58
|
`任务:${taskDesc}`,
|
|
59
|
+
...(acceptance ? ['', `交付验收标准(首要评判依据,逐条核对两份产出是否满足):\n${acceptance}`] : []),
|
|
58
60
|
'', '【产出 A】', trunc(outA), '', '【产出 B】', trunc(outB), '',
|
|
59
|
-
|
|
61
|
+
acceptance
|
|
62
|
+
? '评判维度:验收标准满足度优先,其次完整性、具体性、可用性、是否直接可交付。'
|
|
63
|
+
: '评判维度:完整性、具体性、可用性、是否直接可交付。',
|
|
60
64
|
'只输出一行 JSON,不要任何额外文字:{"scoreA": 1-10, "scoreB": 1-10, "reason": "一句话理由"}',
|
|
61
65
|
].join('\n');
|
|
62
66
|
for (let attempt = 0; attempt < 2; attempt++) {
|
|
@@ -89,9 +93,9 @@ export function aggregateVerdict(j1, j2) {
|
|
|
89
93
|
* 双向盲评对比:对 (多智能体, 基线) 正反各评一次取平均。
|
|
90
94
|
* judge 解析失败返回 null(调用方决定跳过/重试)。
|
|
91
95
|
*/
|
|
92
|
-
export async function compareOutputs(judgeLlm, taskDesc, multiOutput, baselineOutput) {
|
|
93
|
-
const j1 = await judgeOnce(judgeLlm, taskDesc, multiOutput, baselineOutput); // A=multi, B=base
|
|
94
|
-
const j2 = await judgeOnce(judgeLlm, taskDesc, baselineOutput, multiOutput); // A=base, B=multi
|
|
96
|
+
export async function compareOutputs(judgeLlm, taskDesc, multiOutput, baselineOutput, acceptance) {
|
|
97
|
+
const j1 = await judgeOnce(judgeLlm, taskDesc, multiOutput, baselineOutput, acceptance); // A=multi, B=base
|
|
98
|
+
const j2 = await judgeOnce(judgeLlm, taskDesc, baselineOutput, multiOutput, acceptance); // A=base, B=multi
|
|
95
99
|
if (!j1 || !j2)
|
|
96
100
|
return null;
|
|
97
101
|
return aggregateVerdict(j1, j2);
|
package/dist/core/executor.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* DAG 执行引擎 — 核心调度器
|
|
3
3
|
*/
|
|
4
|
-
import type { DAGNode, LLMConnector, LLMConfig, WorkflowResult } from '../types.js';
|
|
4
|
+
import type { DAGNode, LLMConnector, LLMConfig, WorkflowResult, StepResult } from '../types.js';
|
|
5
5
|
import type { DAG } from './dag.js';
|
|
6
6
|
export interface ExecutorOptions {
|
|
7
7
|
connector: LLMConnector;
|
|
@@ -27,6 +27,23 @@ export interface ExecutorOptions {
|
|
|
27
27
|
text: string;
|
|
28
28
|
previousOutput?: string;
|
|
29
29
|
};
|
|
30
|
+
/**
|
|
31
|
+
* acceptance 自动核验:写了 acceptance 的步骤产出后自动逐条核对,未过则带着
|
|
32
|
+
* 未满足条目自动返工一轮(复用对话式返工范式)。验收不过是质量信号而非执行错误,
|
|
33
|
+
* 步骤不会因此 failed。产品入口(run())按 CLI flag > YAML 顶层 verify > 默认开
|
|
34
|
+
* 计算后传入;库级直调 executeDAG 不传 = 不核验(向后兼容)。step.verify: false 单步关闭。
|
|
35
|
+
*/
|
|
36
|
+
verify?: boolean;
|
|
37
|
+
/**
|
|
38
|
+
* 调用方提供的步骤结果收集数组:executor 增量写入(每步完成即可见),
|
|
39
|
+
* 供 SIGTERM/SIGINT 中断时把已完成步骤落盘成 metadata(否则中断的 run 无痕)。
|
|
40
|
+
*/
|
|
41
|
+
stepResultsSink?: StepResult[];
|
|
42
|
+
/**
|
|
43
|
+
* resume 复用步骤在上一次运行档案里的展示字段(agentName/acceptance/verification 等),
|
|
44
|
+
* 由 run() 从旧 metadata 读出传入——续跑产生的新档案才不丢被复用步骤的验收记录。
|
|
45
|
+
*/
|
|
46
|
+
restoredStepMeta?: Map<string, Partial<StepResult>>;
|
|
30
47
|
}
|
|
31
48
|
export declare function executeDAG(dag: DAG, options: ExecutorOptions): Promise<WorkflowResult>;
|
|
32
49
|
/**
|