flavor-code 1.4.4-beta.2 → 1.4.5-beta.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -3
- package/README.zh-CN.md +43 -3
- package/dist/agent/expert-agents.d.ts +35 -0
- package/dist/agent/expert-generator.d.ts +14 -0
- package/dist/agent/expert-templates.d.ts +26 -0
- package/dist/agent/planner.d.ts +2 -0
- package/dist/{app-C5I2TH7I.js → app-LJV3GEXU.js} +805 -398
- package/dist/{chunk-IKGT4MDT.js → chunk-4GKKXT5V.js} +1 -0
- package/dist/{chunk-Y7EH4NVO.js → chunk-55YUJKT4.js} +1 -1
- package/dist/{chunk-X6U6BO5D.js → chunk-CI6IHATI.js} +7 -3
- package/dist/{chunk-U4XTQ5PU.js → chunk-G3LKW5SY.js} +4536 -3367
- package/dist/{chunk-5BDKRA6U.js → chunk-MVBRIM53.js} +1 -1
- package/dist/{chunk-36IDJRLQ.js → chunk-SAAIDKTM.js} +46 -9
- package/dist/{chunk-JQM2L2YM.js → chunk-SW7VZDFS.js} +1 -1
- package/dist/{chunk-ONRY3C2L.js → chunk-TZ7PI4NZ.js} +1 -1
- package/dist/{claude-ink-F554I3GJ.js → claude-ink-5X2DRKN6.js} +1 -1
- package/dist/{cli-V3BLZ5UW.js → cli-4NWD3ZN3.js} +66 -14
- package/dist/cli-main.js +19 -19
- package/dist/desktop/main.js +9336 -8125
- package/dist/desktop-renderer/assets/{index-CjC0Ibiq.js → index-CBjfgCRz.js} +3 -3
- package/dist/desktop-renderer/assets/{interactive-terminal-IjIT3Xoa.js → interactive-terminal-x8YYSqF6.js} +1 -1
- package/dist/desktop-renderer/index.html +1 -1
- package/dist/{doctor-SEL32R7M.js → doctor-77UMJ6TK.js} +3 -3
- package/dist/evolution/preferences.d.ts +52 -0
- package/dist/evolve/loader.d.ts +2 -0
- package/dist/evolve/service.d.ts +6 -3
- package/dist/evolve/store.d.ts +55 -2
- package/dist/evolve/verification.d.ts +30 -0
- package/dist/harness/local.d.ts +4 -2
- package/dist/{load-RHWXNUCS.js → load-JINXGR3N.js} +2 -2
- package/dist/{manager-IYBNNKUP.js → manager-UTQT466W.js} +2 -2
- package/dist/memory/coordinator.d.ts +1 -1
- package/dist/memory/review.d.ts +13 -0
- package/dist/memory/store.d.ts +1 -0
- package/dist/memory/types.d.ts +2 -0
- package/dist/{production-TD7UWWXW.js → production-J7ZWR7VL.js} +6 -6
- package/dist/sdk/index.js +6 -6
- package/dist/session/store.d.ts +1 -0
- package/dist/{store-MZZERW7K.js → store-Y4V5TLXE.js} +2 -2
- package/dist/ui/commands.d.ts +21 -2
- package/dist/ui/session.d.ts +8 -0
- package/package.json +1 -1
- package//346/212/200/346/234/257/346/226/271/346/241/210/346/212/245/345/221/212.md +97 -6
package/README.md
CHANGED
|
@@ -41,7 +41,7 @@ Flavor Code connects to OpenAI, Anthropic, or compatible services and works with
|
|
|
41
41
|
| 🌿 | **Git-native workflows** | `/commit` drafts a Conventional-Commits message for staged changes and commits after confirmation; `/review` audits uncommitted changes; the read-only `GitHistory` tool explains when and why code changed |
|
|
42
42
|
| 🎨 | **E2E requirement-to-delivery** | From a rough requirement or a design export to a delivered product: PRD, interactive prototype, visual implementation, API integration, autonomous acceptance, and scored delivery (Electron only) |
|
|
43
43
|
| 🌐 | **Agent-driven built-in browser** | The Electron desktop embeds an interactive browser: the agent navigates pages and acts on elements by snapshot reference with `BrowserSnapshot`/`BrowserAct` and friends, every action is annotated in real time via a visual overlay, navigation is guarded against SSRF, and snapshots redact passwords, tokens, and other sensitive fields (Electron only) |
|
|
44
|
-
| 🔁 | **
|
|
44
|
+
| 🔁 | **Evaluable self-improvement trial** | Explicit preferences take effect, inferred preferences enter a tracked trial and can be dropped; fix plugins require hash-bound sandbox verification and tests before activation, with rollback (`/evolve`) |
|
|
45
45
|
| 🛡️ | **Clear permission boundaries** | Independent control over read, write, Shell, network, and destructive actions; Docker supported |
|
|
46
46
|
|
|
47
47
|
## Quick Start
|
|
@@ -188,6 +188,7 @@ Common commands:
|
|
|
188
188
|
| `/init` | Generate or update `FLAVOR.md` |
|
|
189
189
|
| `/doctor` | Diagnose the local runtime, configuration, tools, plugins, and npm access |
|
|
190
190
|
| `/model` | View or switch main/sub-agent models |
|
|
191
|
+
| `/agent` | List, create, or run expert agents |
|
|
191
192
|
| `/permissions` | Switch permission modes |
|
|
192
193
|
| `/tasks` | View task plans and sub-agent status |
|
|
193
194
|
| `/compact` | Manually compact long session context |
|
|
@@ -201,10 +202,12 @@ Common commands:
|
|
|
201
202
|
| `/commit [hint]` | Draft a Conventional-Commits message for staged changes and commit after confirmation |
|
|
202
203
|
| `/review [focus]` | Review uncommitted changes for bugs and risks before committing |
|
|
203
204
|
| `/explain <symbol \| file.ts#symbol> [focus]` | Explain a symbol for newcomers using the code graph, real source and git history (interactive picker on ambiguity) |
|
|
204
|
-
| `/evolve
|
|
205
|
+
| `/evolve status` | Inspect task trends, tool failures, proposed rules, and learned preferences; use `rule list/accept` to review model rules, `preference list/drop/restore` for preferences, and `verify → test → reload` for implemented fix plugins |
|
|
205
206
|
| `/pals`, `/chat`, `/co-work` | Discover and collaborate with other local CLI instances |
|
|
206
207
|
| `/audit` | View tool failure audits |
|
|
207
208
|
|
|
209
|
+
Self-improvement starts during ordinary conversations with the default memory settings. Say “From now on, explain changes in Chinese,” then inspect `/evolve preference list`; after another task, inspect `/evolve trends 2`. Use `/evolve preference drop <id>` to stop an unwanted preference. Inferred preferences stay proposed until another independent task supports them and only enter related tasks during trial. See section 50.6 of the technical report for a full check.
|
|
210
|
+
|
|
208
211
|
You can submit steering or queue follow-ups while a run is in progress; once the current model response finishes, the task picks up new instructions at safe boundaries.
|
|
209
212
|
While a run is active, Enter queues the text for the next turn; `/steer <message>` changes the current turn. Type `/queue` to inspect all queued messages, use Up/Down to select one, Enter to move it back to the draft, or `d` to cancel it. Escape closes the queue; outside that view it restores the latest queued message for editing.
|
|
210
213
|
|
|
@@ -323,6 +326,40 @@ flavor mcp disable docs
|
|
|
323
326
|
|
|
324
327
|
A Skill is a `SKILL.md` with YAML frontmatter, placed in `.flavor/skills/<name>/` or `~/.flavor-code/skills/<name>/`. Flavor loads skills progressively based on the task, and you can invoke one explicitly with `/<skill-name>`. Skill bodies support `$ARGUMENTS`, `$ARGUMENTS[N]`, and `$N` substitutions. A running composite Skill can load a dependency through the read-only `Skill` tool; plugin-qualified names such as `superharness:test-driven-development` resolve to discovered skills.
|
|
325
328
|
|
|
329
|
+
### Expert agents
|
|
330
|
+
|
|
331
|
+
Create a project role from a preset without writing Markdown:
|
|
332
|
+
|
|
333
|
+
```text
|
|
334
|
+
/agent templates
|
|
335
|
+
/agent create reviewer
|
|
336
|
+
/agent create api-reviewer reviewer Check API compatibility and regressions
|
|
337
|
+
/agent create worker implementer Implement the assigned module
|
|
338
|
+
/agent create test-writer Add unit tests for authentication
|
|
339
|
+
/agent create db-auditor --read-only Review database migration risks
|
|
340
|
+
/agent list
|
|
341
|
+
/agent api-reviewer inspect authentication
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
Agent names and responsibilities are not limited to the three presets. `reviewer` (read-only review), `explorer` (read-only code tracing), and `implementer` (edits under existing approval rules) are instant shortcuts. `/agent create <name> <description>` uses the configured main model to generate a role-specific workflow, deliverables, boundaries, tools, and permission; add `--read-only` to require a read-only role. A review-only description is read-only automatically. If generation or validation fails, no file is written. The command writes `.flavor/agents/<name>.md` without overwriting an existing file. You can edit the generated file later.
|
|
345
|
+
|
|
346
|
+
You can also place a role definition manually in `.flavor/agents/<name>.md` or `~/.flavor-code/agents/<name>.md`. Project definitions take precedence over global ones. The filename must match `name`:
|
|
347
|
+
|
|
348
|
+
```markdown
|
|
349
|
+
---
|
|
350
|
+
name: reviewer
|
|
351
|
+
description: Review code for correctness and regression risks
|
|
352
|
+
model: openai:gpt-5-mini
|
|
353
|
+
tools: [Read, Glob, Grep, LspFindRefs, TaskOutput]
|
|
354
|
+
permission: readOnly
|
|
355
|
+
maxIterations: 30
|
|
356
|
+
---
|
|
357
|
+
|
|
358
|
+
Check edge cases, error handling, and test gaps. Report concrete files and lines by severity.
|
|
359
|
+
```
|
|
360
|
+
|
|
361
|
+
Use `/agent list` to inspect roles and `/agent reviewer inspect authentication` to run one directly. The main agent can select a role in a `Task` node with `"agent": "reviewer"`. `model`, `tools`, and `maxIterations` are optional; they default to the subagent model, all available subagent tools, and the global subagent iteration limit. `permission` defaults to `standard` and follows existing subagent permissions. `readOnly` exposes only read tools and enforces read-only mode at runtime; Shell and write tools are unavailable. A `Task` node's `files` field schedules conflicting writes; it does not sandbox file access.
|
|
362
|
+
|
|
326
363
|
Plugins live in `.flavor/plugins/` and can register commands, tools, hooks, Skill roots, and model adapters. Official plugins can be installed with the plugin manager: `npx --yes @flavor-code/plugin-manager`. `additionalContext` returned by `SessionStart` and `UserPromptSubmit` hooks is added to the current task context, enabling reliable project-level engineering policy injection. Plugin loads record a content fingerprint plus declared capabilities. Worker/vm isolation is available through the embedding API's `pluginSandbox: true` option; the compatibility default remains in-process because bundled and existing plugins use Node.js APIs that the isolated runtime does not yet mediate.
|
|
327
364
|
|
|
328
365
|
When the `flavor-island` plugin is loaded, Flavor also starts a Flavor Island local control channel: a loopback-only IPC service (Windows named pipe, or Unix socket on macOS/Linux) secured by a random token. A host app (such as the Flavor Island desktop) can use it to abort, steer, or send follow-ups to a running session, and desktop hosts can also bring their window into focus. The channel's endpoint, token, and capability list are exposed to the host plugin via hook event context (`islandControlEndpoint`/`islandControlToken`/`islandControlCapabilities`); model-call duration and token usage, the current session title (desktop custom task name, falling back to the session preview), plus the final task summary and deliverables, are reported through hook events so the host can show live status and a result overview.
|
|
@@ -350,7 +387,7 @@ Project runtime data lives under `.flavor/`:
|
|
|
350
387
|
└── plugins/ # Project plugins
|
|
351
388
|
```
|
|
352
389
|
|
|
353
|
-
Long-term memory distinguishes user preferences, behavioral feedback, project conventions, and external references.
|
|
390
|
+
Long-term memory distinguishes user preferences, behavioral feedback, project conventions, and external references. Automatically extracted project facts and external references enter a review inbox that survives restarts; a high model score alone cannot save them. Review cards show whether an exact user quote supports a candidate. Use `Ctrl+Y` to save or `Ctrl+N` to ignore; on macOS, `Command+Y` and `Command+N` also work when the terminal passes those keys through. `/memory` shows pending and decision counts. Explicit `/remember` still saves directly. Secrets, tokens, and raw tool output are rejected.
|
|
354
391
|
|
|
355
392
|
Image prompts support PNG, JPEG, and WebP, with a 5 MiB per-image maximum and up to 5 images per prompt. The desktop app supports picking or drag-and-drop; CLI clipboard images currently work on Windows and macOS. Standard CLI paste prefers clipboard text and creates an image attachment only when no usable text flavor is available; use `/paste-image` when both flavors exist and the image is intended.
|
|
356
393
|
|
|
@@ -422,8 +459,11 @@ Run evaluations:
|
|
|
422
459
|
|
|
423
460
|
```bash
|
|
424
461
|
flavor eval eval.json --output report.json
|
|
462
|
+
flavor eval eval.json --baseline ../baseline-worktree --output comparison.json
|
|
425
463
|
```
|
|
426
464
|
|
|
465
|
+
The second command runs the same prompt and checks in the spec's candidate workspace and a separate baseline workspace. Both are modified by the agent, so use disposable test copies. Compact results go to the candidate project's `.flavor/evolve/comparisons.jsonl` and are shown by `/evolve comparisons`. Ordinary task outcomes, preference exposures, and attributable feedback go to `.flavor/evolve/outcome-events.jsonl` and are shown by `/evolve outcomes`. These local files are Git-ignored and do not copy full conversations.
|
|
466
|
+
|
|
427
467
|
Design constraints for RPC, traces, replay, eval, session trees, and Docker are in the [control-plane spec](./docs/specs/2026-07-29-control-plane-sandbox-vscode.md).
|
|
428
468
|
|
|
429
469
|
## Development
|
package/README.zh-CN.md
CHANGED
|
@@ -41,7 +41,7 @@ Flavor Code 接入 OpenAI、Anthropic 或兼容服务,在受控工作区内使
|
|
|
41
41
|
| 🌿 | **Git 原生工作流** | `/commit` 为暂存改动生成 Conventional Commits 提交信息并确认提交;`/review` 审查未提交改动;只读 `GitHistory` 工具回答“这段代码为什么是这样” |
|
|
42
42
|
| 🎨 | **E2E 需求到交付** | 从粗需求或设计稿到可交付产品:PRD、交互原型、视觉还原、接口联调、自主验收与评分交付(仅 Electron) |
|
|
43
43
|
| 🌐 | **Agent 内置浏览器** | Electron 桌面端内置可交互浏览器:Agent 通过 `BrowserSnapshot`/`BrowserAct` 等工具导航页面、按元素引用点击与输入,操作以可视化覆盖层实时标注;导航受 SSRF 防护约束,快照自动脱敏密码、token 等敏感字段(仅 Electron) |
|
|
44
|
-
| 🔁 |
|
|
44
|
+
| 🔁 | **可评价的自进化试用** | 明确偏好直接生效、推断偏好先试用并记录反馈,可查看和 drop;修复插件按内容哈希完成沙箱验证与测试后才能启用,失败可回滚(`/evolve`) |
|
|
45
45
|
| 🛡️ | **明确的权限边界** | 分别控制读、写、Shell、网络和破坏性操作,也可使用 Docker |
|
|
46
46
|
|
|
47
47
|
## 快速开始
|
|
@@ -188,6 +188,7 @@ OAuth PKCE 的运行时行为与配置约定见 [PKCE 规范](./docs/specs/pkce-
|
|
|
188
188
|
| `/init` | 生成或更新 `FLAVOR.md` |
|
|
189
189
|
| `/doctor` | 诊断本地运行时、配置、工具、插件和 npm 连通性 |
|
|
190
190
|
| `/model` | 查看或切换主/子 Agent 模型 |
|
|
191
|
+
| `/agent` | 查看、快速创建或运行专家 Agent |
|
|
191
192
|
| `/permissions` | 切换权限模式 |
|
|
192
193
|
| `/tasks` | 查看任务计划和子 Agent 状态 |
|
|
193
194
|
| `/compact` | 手动压缩长会话上下文 |
|
|
@@ -201,10 +202,12 @@ OAuth PKCE 的运行时行为与配置约定见 [PKCE 规范](./docs/specs/pkce-
|
|
|
201
202
|
| `/commit [hint]` | 为暂存改动生成 Conventional Commits 提交信息,确认后提交 |
|
|
202
203
|
| `/review [focus]` | 提交前审查未提交改动的缺陷与风险 |
|
|
203
204
|
| `/explain <符号 \| file.ts#符号> [关注点]` | 面向新人讲解一个符号:结合代码图、真实源码与 Git 历史,歧义时弹卡片选择符号 |
|
|
204
|
-
| `/evolve
|
|
205
|
+
| `/evolve status` | 查看普通任务与循环任务的趋势、工具故障建议、待审核规则和偏好状态;`preference list/drop/restore` 管理偏好,`rule list/accept` 审核模型规则,`verify → test → reload` 启用非空修复插件 |
|
|
205
206
|
| `/pals`、`/chat`、`/co-work` | 发现并协作其他本机 CLI 实例 |
|
|
206
207
|
| `/audit` | 查看工具失败审计 |
|
|
207
208
|
|
|
209
|
+
自进化可直接从普通对话开始,无需单独打开开关:输入「以后请用中文解释」,完成一轮后用 `/evolve preference list` 查看 `active`;再执行普通任务,用 `/evolve trends 2` 查看运行记录。不合适时用 `/evolve preference drop <id>` 停用。自动推断的偏好先显示 `proposed`,需要另一个独立任务的用户原话支持才会进入 `canary`。完整验证步骤见《技术方案报告》第 50.6 节。
|
|
210
|
+
|
|
208
211
|
运行中可以提交 steering 或排队 follow-up;当前模型响应结束后,任务会在安全边界处接收新指令。
|
|
209
212
|
运行中按 Enter 会把输入排到下一轮;输入 `/steer <内容>` 会影响当前回合。输入 `/queue` 可查看全部待发送消息,用上下键选中,按 Enter 移回输入框编辑,或按 `d` 取消。Esc 关闭队列视图;在普通输入界面则会取回最后一条待发送消息供编辑。
|
|
210
213
|
|
|
@@ -323,6 +326,40 @@ flavor mcp disable docs
|
|
|
323
326
|
|
|
324
327
|
Skill 是带有 YAML 头信息的 `SKILL.md`,放在 `.flavor/skills/<name>/` 或 `~/.flavor-code/skills/<name>/`。Flavor 会按任务渐进加载,也支持通过 `/<skill-name>` 显式调用。Skill 正文支持 `$ARGUMENTS`、`$ARGUMENTS[N]` 和 `$N` 参数占位符;运行中的组合 Skill 可以使用只读 `Skill` 工具继续加载依赖 Skill,插件限定名称(如 `superharness:test-driven-development`)会安全解析到已发现的 Skill。
|
|
325
328
|
|
|
329
|
+
### 专家 Agent
|
|
330
|
+
|
|
331
|
+
用预设一条命令创建项目角色,无需手写 Markdown:
|
|
332
|
+
|
|
333
|
+
```text
|
|
334
|
+
/agent templates
|
|
335
|
+
/agent create reviewer
|
|
336
|
+
/agent create api-reviewer reviewer 检查 API 兼容性和回归风险
|
|
337
|
+
/agent create worker implementer 处理指定模块的实现任务
|
|
338
|
+
/agent create test-writer 补齐认证模块的单元测试
|
|
339
|
+
/agent create db-auditor --read-only 审查数据库迁移风险
|
|
340
|
+
/agent list
|
|
341
|
+
/agent api-reviewer 检查认证接口
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
Agent 名称和职责不受这三个预设限制。`reviewer`(只读审查)、`explorer`(只读探索)和 `implementer`(按现有审批规则修改代码)是无需调用模型的快捷模板。`/agent create <名称> <职责描述>` 会调用当前主模型,生成针对该职责的工作步骤、交付内容、边界、工具和权限;加 `--read-only` 可强制只读,纯审查职责也会自动设为只读。生成或校验失败时不会写入文件。命令会生成 `.flavor/agents/<名称>.md`,已存在的文件不会被覆盖。生成后可直接运行,也可编辑文件细调。
|
|
345
|
+
|
|
346
|
+
角色定义也可以手动放在项目的 `.flavor/agents/<name>.md` 或全局的 `~/.flavor-code/agents/<name>.md`;同名时项目定义优先。定义文件的名称必须与 `name` 一致。格式示例:
|
|
347
|
+
|
|
348
|
+
```markdown
|
|
349
|
+
---
|
|
350
|
+
name: reviewer
|
|
351
|
+
description: Review code for correctness and regression risks
|
|
352
|
+
model: openai:gpt-5-mini
|
|
353
|
+
tools: [Read, Glob, Grep, LspFindRefs, TaskOutput]
|
|
354
|
+
permission: readOnly
|
|
355
|
+
maxIterations: 30
|
|
356
|
+
---
|
|
357
|
+
|
|
358
|
+
检查边界条件、错误处理和测试缺口,按严重程度报告具体文件与行号。
|
|
359
|
+
```
|
|
360
|
+
|
|
361
|
+
使用 `/agent list` 查看角色,使用 `/agent reviewer 检查认证模块` 单独运行。主 Agent 也能在 `Task` 的节点里设置 `"agent": "reviewer"`。`model`、`tools`、`maxIterations` 可省略,分别使用子 Agent 模型、所有可用的子 Agent 工具和全局子 Agent 迭代上限。`permission` 默认为 `standard`,遵守现有子 Agent 权限;`readOnly` 仅提供只读工具并在运行时使用只读权限,Shell 和写入工具不可用。`Task` 节点的 `files` 仍用于并行写冲突调度,不是文件写入沙箱。
|
|
362
|
+
|
|
326
363
|
插件放在 `.flavor/plugins/`,可以注册命令、工具、Hook、Skill 根目录和模型适配器。官方插件可通过插件管理器安装:`npx --yes @flavor-code/plugin-manager`。`SessionStart` 与 `UserPromptSubmit` Hook 返回的 `additionalContext` 会进入当前任务上下文,可用于注入项目级工程规则。插件加载会记录内容指纹与声明的能力。可通过嵌入 API 的 `pluginSandbox: true` 启用 Worker/vm 隔离;由于内置插件和已有插件依赖沙箱尚未代理的 Node.js API,当前兼容默认值仍为进程内运行。
|
|
327
364
|
|
|
328
365
|
加载 `flavor-island` 插件时,Flavor 会额外启动一个 Flavor Island 本地控制通道:这是一个只监听本机回环的 IPC 服务(Windows 使用 named pipe,macOS/Linux 使用 Unix socket),通过随机 token 认证。宿主应用(如 Flavor Island 桌面端)可以借此对运行中的会话执行中止、steering、follow-up 等操作,桌面端还支持把窗口带到前台(focus)。通道的地址、token 与能力列表会通过 Hook 事件上下文(`islandControlEndpoint`/`islandControlToken`/`islandControlCapabilities`)提供给宿主插件,模型调用的耗时与 token 用量、当前会话标题(桌面端自定义任务名,缺省时回退到会话 preview 摘要)、任务结束时的摘要和交付文件也会随 Hook 事件上报,方便宿主展示运行状态与结果概览。
|
|
@@ -350,7 +387,7 @@ Skill 是带有 YAML 头信息的 `SKILL.md`,放在 `.flavor/skills/<name>/`
|
|
|
350
387
|
└── plugins/ # 项目插件
|
|
351
388
|
```
|
|
352
389
|
|
|
353
|
-
|
|
390
|
+
长期记忆会区分用户偏好、行为反馈、项目约定和外部引用。自动提取的项目约定和外部引用会进入可跨会话保留的审核队列,模型自评分再高也不会直接写入;卡片会标明是否有可核对的用户原话。使用 `Ctrl+Y` 保存或 `Ctrl+N` 忽略;macOS 终端传递按键时也支持 `Command+Y` 和 `Command+N`。`/memory` 显示待审核数量与采纳、忽略统计。明确的 `/remember` 仍直接保存。密钥、Token 和原始工具输出不会作为候选保存。
|
|
354
391
|
|
|
355
392
|
图片提示支持 PNG、JPEG 和 WebP,单图最大 5 MiB、每次最多 5 张。桌面端支持选择或拖放;CLI 剪贴板图片目前支持 Windows 和 macOS。CLI 标准粘贴会优先使用剪贴板文字,只有没有可用文字时才把剪贴板图像添加为附件;剪贴板同时含有文字和图片但需要图片时,使用 `/paste-image`。
|
|
356
393
|
|
|
@@ -422,8 +459,11 @@ flavor --mode rpc --workspace . --trace .flavor/traces/run.jsonl
|
|
|
422
459
|
|
|
423
460
|
```bash
|
|
424
461
|
flavor eval eval.json --output report.json
|
|
462
|
+
flavor eval eval.json --baseline ../baseline-worktree --output comparison.json
|
|
425
463
|
```
|
|
426
464
|
|
|
465
|
+
第二条命令把 `eval.json` 的 `workspace` 作为候选工作区,在独立的基线工作区运行同一提示和验证命令;两个工作区都会被 Agent 修改,应使用可丢弃的测试副本。精简结果保存在候选项目的 `.flavor/evolve/comparisons.jsonl`,可用 `/evolve comparisons` 查看。普通任务的偏好暴露、完成结果和可归因反馈写入 `.flavor/evolve/outcome-events.jsonl`,可用 `/evolve outcomes` 查看;这些本地文件已被 Git 忽略,不复制完整对话。只有脱敏并愿意共享的固定评测题才适合提交到仓库。
|
|
466
|
+
|
|
427
467
|
RPC、trace、replay、eval、会话树与 Docker 的设计约束见 [控制面规范](./docs/specs/2026-07-29-control-plane-sandbox-vscode.md)。
|
|
428
468
|
|
|
429
469
|
## 开发
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import { type ExpertAgentCreationKind } from "./expert-templates.js";
|
|
3
|
+
declare const AgentFrontmatterSchema: z.ZodObject<{
|
|
4
|
+
name: z.ZodString;
|
|
5
|
+
description: z.ZodString;
|
|
6
|
+
model: z.ZodOptional<z.ZodString>;
|
|
7
|
+
tools: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
8
|
+
permission: z.ZodDefault<z.ZodEnum<{
|
|
9
|
+
readOnly: "readOnly";
|
|
10
|
+
standard: "standard";
|
|
11
|
+
}>>;
|
|
12
|
+
maxIterations: z.ZodOptional<z.ZodNumber>;
|
|
13
|
+
}, z.core.$strict>;
|
|
14
|
+
export type ExpertAgent = z.infer<typeof AgentFrontmatterSchema> & {
|
|
15
|
+
instructions: string;
|
|
16
|
+
source: "global" | "project";
|
|
17
|
+
path: string;
|
|
18
|
+
};
|
|
19
|
+
export interface ExpertAgentDiagnostic {
|
|
20
|
+
path: string;
|
|
21
|
+
message: string;
|
|
22
|
+
}
|
|
23
|
+
export type GeneratedExpertAgent = Pick<ExpertAgent, "description" | "permission" | "tools" | "maxIterations" | "instructions">;
|
|
24
|
+
/** Project definitions override global definitions with the same name. */
|
|
25
|
+
export declare class ExpertAgentRegistry {
|
|
26
|
+
#private;
|
|
27
|
+
constructor(home: string, workspace: string);
|
|
28
|
+
get diagnostics(): readonly ExpertAgentDiagnostic[];
|
|
29
|
+
assertAvailable(name: string): Promise<void>;
|
|
30
|
+
create(name: string, kind: ExpertAgentCreationKind, description?: string, readOnly?: boolean): Promise<ExpertAgent>;
|
|
31
|
+
createGenerated(name: string, generated: GeneratedExpertAgent): Promise<ExpertAgent>;
|
|
32
|
+
discover(): Promise<readonly ExpertAgent[]>;
|
|
33
|
+
get(name: string): Promise<ExpertAgent>;
|
|
34
|
+
}
|
|
35
|
+
export {};
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import type { GeneratedExpertAgent } from "./expert-agents.js";
|
|
2
|
+
import type { ModelRegistry } from "../models/registry.js";
|
|
3
|
+
import type { ToolDefinition } from "../tools/types.js";
|
|
4
|
+
export interface GenerateExpertAgentOptions {
|
|
5
|
+
registry: ModelRegistry;
|
|
6
|
+
modelId: string;
|
|
7
|
+
name: string;
|
|
8
|
+
request: string;
|
|
9
|
+
tools: readonly ToolDefinition<unknown>[];
|
|
10
|
+
forceReadOnly?: boolean;
|
|
11
|
+
signal?: AbortSignal;
|
|
12
|
+
}
|
|
13
|
+
/** Creation is only successful after a substantive, runnable definition passes validation. */
|
|
14
|
+
export declare function generateExpertAgent(options: GenerateExpertAgentOptions): Promise<GeneratedExpertAgent>;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
export declare const EXPERT_AGENT_TEMPLATES: {
|
|
2
|
+
readonly reviewer: {
|
|
3
|
+
readonly description: "Review code for correctness, regressions, and missing tests";
|
|
4
|
+
readonly permission: "readOnly";
|
|
5
|
+
readonly tools: readonly ["Read", "Glob", "Grep", "TaskOutput"];
|
|
6
|
+
readonly maxIterations: 30;
|
|
7
|
+
readonly instructions: "Inspect the relevant code and its callers. Report actionable findings with file paths and line numbers, ordered by severity. Explain the failure mode and any missing coverage. Do not modify files.";
|
|
8
|
+
};
|
|
9
|
+
readonly explorer: {
|
|
10
|
+
readonly description: "Trace code paths and locate the files needed for a task";
|
|
11
|
+
readonly permission: "readOnly";
|
|
12
|
+
readonly tools: readonly ["Read", "Glob", "Grep", "TaskOutput"];
|
|
13
|
+
readonly maxIterations: 30;
|
|
14
|
+
readonly instructions: "Find the entry points, relevant files, call paths, and existing tests. Return a concise map of what you found with file paths. State uncertainties that need inspection. Do not modify files.";
|
|
15
|
+
};
|
|
16
|
+
readonly implementer: {
|
|
17
|
+
readonly description: "Make scoped code changes and verify them";
|
|
18
|
+
readonly permission: "standard";
|
|
19
|
+
readonly tools: readonly ["Read", "Glob", "Grep", "Write", "Edit", "ApplyPatch", "Shell", "TaskOutput"];
|
|
20
|
+
readonly maxIterations: 100;
|
|
21
|
+
readonly instructions: "Implement the assigned change within the requested scope. Inspect related code first, preserve unrelated work, run focused verification, and report changed files, test results, and remaining risks.";
|
|
22
|
+
};
|
|
23
|
+
};
|
|
24
|
+
export type ExpertAgentTemplateName = keyof typeof EXPERT_AGENT_TEMPLATES;
|
|
25
|
+
export type ExpertAgentCreationKind = ExpertAgentTemplateName | "custom";
|
|
26
|
+
export declare const EXPERT_AGENT_TEMPLATE_NAMES: ExpertAgentTemplateName[];
|
package/dist/agent/planner.d.ts
CHANGED
|
@@ -3,6 +3,7 @@ import type { HookBus } from "../hooks/bus.js";
|
|
|
3
3
|
export declare const TaskNodeSchema: z.ZodObject<{
|
|
4
4
|
id: z.ZodString;
|
|
5
5
|
description: z.ZodString;
|
|
6
|
+
agent: z.ZodOptional<z.ZodString>;
|
|
6
7
|
dependencies: z.ZodArray<z.ZodString>;
|
|
7
8
|
expectedOutputs: z.ZodArray<z.ZodString>;
|
|
8
9
|
verification: z.ZodArray<z.ZodString>;
|
|
@@ -12,6 +13,7 @@ export declare const TaskGraphSchema: z.ZodObject<{
|
|
|
12
13
|
nodes: z.ZodArray<z.ZodObject<{
|
|
13
14
|
id: z.ZodString;
|
|
14
15
|
description: z.ZodString;
|
|
16
|
+
agent: z.ZodOptional<z.ZodString>;
|
|
15
17
|
dependencies: z.ZodArray<z.ZodString>;
|
|
16
18
|
expectedOutputs: z.ZodArray<z.ZodString>;
|
|
17
19
|
verification: z.ZodArray<z.ZodString>;
|