@genee/omp-opsx-addon 0.7.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +441 -71
- package/index.ts +1030 -208
- package/lib/agent-defs.ts +55 -82
- package/lib/complexity-router.ts +336 -0
- package/lib/concurrency-control.ts +1246 -0
- package/lib/concurrency-multiproc.worker.ts +376 -0
- package/lib/concurrency-state.ts +397 -0
- package/lib/concurrency-tuner.ts +773 -0
- package/lib/concurrency-writer.ts +894 -0
- package/lib/continuity-guard.ts +117 -0
- package/lib/direct-fetchers.ts +18 -3
- package/lib/family-filter.ts +115 -2
- package/lib/model-roles.ts +41 -39
- package/lib/model-selector.ts +300 -36
- package/lib/model-speed.ts +189 -0
- package/lib/pick-model-render.ts +414 -0
- package/lib/pipe-core.ts +325 -98
- package/lib/pipe-push.ts +136 -5
- package/lib/provider-variants.ts +154 -0
- package/lib/selection-filters.ts +42 -1
- package/lib/stall-detector.ts +185 -0
- package/lib/system-prompt.ts +17 -9
- package/lib/tiers-data.ts +40 -5
- package/lib/tiers-updater.ts +51 -8
- package/lib/unified-config.ts +732 -27
- package/lib/usage-poller.ts +44 -1
- package/lib/usage-render.ts +57 -4
- package/lib/usage-widget.ts +18 -4
- package/package.json +1 -1
- package/skills/opsx-orchestration-protocol/SKILL.md +87 -0
package/lib/agent-defs.ts
CHANGED
|
@@ -2,10 +2,14 @@ export const ADDON_VERSION = '0.1.0';
|
|
|
2
2
|
|
|
3
3
|
const MARKER = '<!-- @genee/omp-opsx-addon -->';
|
|
4
4
|
|
|
5
|
+
/** 硬栅栏 marker:CODER_MD 栅栏小节标题与派发浓缩句都必须含它,供正/负断言与防漂移定位(change: opsx-scope-fence)。 */
|
|
6
|
+
export const SCOPE_FENCE_TAG = '【无关状态栅栏】';
|
|
7
|
+
|
|
5
8
|
const CODE_REVIEWER_MD = (model?: string) => `---
|
|
6
9
|
name: code-reviewer
|
|
7
10
|
description: OpenSpec 代码审查 agent,审查代码实现并执行全局验证
|
|
8
|
-
tools: read, bash, glob, grep
|
|
11
|
+
tools: read, bash, glob, grep
|
|
12
|
+
autoloadSkills: opsx-orchestration-protocol${model ? `\nmodel: "${model}"` : ''}
|
|
9
13
|
---
|
|
10
14
|
|
|
11
15
|
你是一位资深代码审查员。你根据 OpenSpec 提案和设计文档审查代码实现,并在审阅的同时执行最后一次全局验证。
|
|
@@ -14,12 +18,13 @@ tools: read, bash, glob, grep${model ? `\nmodel: "${model}"` : ''}
|
|
|
14
18
|
1. 从 prompt 或上下文确定要审查的变更
|
|
15
19
|
2. 读取 proposal、design 和 tasks 文件
|
|
16
20
|
3. 运行 \`git diff HEAD~1\` 或 \`git diff --stat\` 查看改动
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
-
|
|
21
|
+
|
|
22
|
+
## 技能
|
|
23
|
+
- 共享编排协议(scratchpad 四区共享缓存与角色读写矩阵、supersede 权威语义、结论上报去向、报告契约)由技能 \`opsx-orchestration-protocol\` 承载;若它未自动加载,用 Skill tool 读取 \`skill://opsx-orchestration-protocol\` 并遵循其流程。
|
|
24
|
+
|
|
25
|
+
## 审查范围(scratchpad 涉及文件)
|
|
26
|
+
- 以「调研与设计」+「实现探索」两区累计的涉及文件 \`path:line\` 集合为代码审查与增量验证范围:Tier 1(typecheck/lint)与 Tier 3(E2E,仅终审轮)聚焦涉及文件改动面;Tier 2 按改动分级选范围(见「全局验证」节)。
|
|
20
27
|
- 无 scratchpad.md 时回退:从 proposal/design/tasks 确定审查范围。
|
|
21
|
-
- 识别 \`- [R<n> coder] supersede:\` 条目:以紧随其后的新结论为权威,旧 \`path:line\` 不再作为验证范围;只读不改写。
|
|
22
|
-
- 发现 scratchpad.md 结论与代码现状不符时,作为 P0/P1 审查发现发回,不直接改写。
|
|
23
28
|
|
|
24
29
|
## 审查维度
|
|
25
30
|
1. **规范遵循**:是否精确实现了 proposal/design/tasks 定义的内容?有无缺失或偏离?
|
|
@@ -29,9 +34,12 @@ tools: read, bash, glob, grep${model ? `\nmodel: "${model}"` : ''}
|
|
|
29
34
|
|
|
30
35
|
## 全局验证(交付前最后一道质量门)
|
|
31
36
|
coder 已在其改动面完成自验证;你的验证是 maker-checker 的兜底确认,与代码审查同轮执行:
|
|
32
|
-
|
|
33
|
-
-
|
|
34
|
-
-
|
|
37
|
+
先定级,再按级别选范围:
|
|
38
|
+
- **微改级**(纯配置 / 展示文案 / 诊断输出 / 单文件可逆,且未命中下方全流程级任一项):Tier 1 只跑改动面 typecheck/lint(无需依赖安装);Tier 2 只跑与改动面相关的测试文件,不跑全量套件。
|
|
39
|
+
- **全流程级**(跨文件语义 / 选择行为 / 排序 / 分区 / modelRoles 写入契约):Tier 1 含依赖安装与完整 typecheck/lint;Tier 2 跑全量测试套件。
|
|
40
|
+
- **Tier 1 – 环境完整性**:依赖安装(仅全流程级)、typecheck、lint
|
|
41
|
+
- **Tier 2 – 冒烟测试**:微改级 → 相关测试文件;全流程级 → 项目现有测试套件(\`bun test\`/\`npm test\` 等)
|
|
42
|
+
- **Tier 3 – E2E(仅终审轮,一次)**:开发过程中的中间轮一律不跑。只有本轮代码审查与 Tier 1 / Tier 2 均无 P0/P1(即本轮通过、无需发回修复)的**终审轮**,才在最后跑一次 E2E;本轮已发回修复(含 P0/P1)时记 N/A,留给后续轮次。变更不涉及用户可见行为时 N/A;因 E2E 自身失败而发回修复的,下一轮终审复核 E2E。
|
|
35
43
|
|
|
36
44
|
## 输出格式
|
|
37
45
|
\`\`\`markdown
|
|
@@ -55,23 +63,20 @@ coder 已在其改动面完成自验证;你的验证是 maker-checker 的兜
|
|
|
55
63
|
### 全局验证
|
|
56
64
|
- [x] / - [ ] Tier 1 环境完整性: <通过/失败项>
|
|
57
65
|
- [x] / - [ ] Tier 2 冒烟测试: <通过/失败项>
|
|
58
|
-
- [x] / - [ ] Tier 3 E2E
|
|
66
|
+
- [x] / - [ ] Tier 3 E2E(仅终审轮;中间轮记 N/A): <通过/失败或 N/A>
|
|
59
67
|
- 验证失败项(按严重度):<P0/P1 已并入关键问题;P2+ 告警,不阻塞通过>
|
|
60
68
|
\`\`\`
|
|
61
69
|
|
|
62
70
|
## 结论与上报
|
|
63
|
-
|
|
64
|
-
- **APPROVE** → 报告通过
|
|
65
|
-
- **REQUEST_CHANGES + P0/P1** → 返回问题列表,由主 agent 委派 coder 修复
|
|
66
|
-
- **REQUEST_CHANGES + 仅 P2+** → 视为通过,建议可选
|
|
67
|
-
- **BLOCKED** → 说明原因,由用户决策
|
|
71
|
+
- 结论语义与上报去向见技能 \`opsx-orchestration-protocol\` 的「评审闭环与轮次上限」节(审查员只报告、不改代码)。
|
|
68
72
|
- 全局验证失败项按严重度并入结论:P0/P1 → 关键问题(必须修复);P2 级验证告警 → 改进建议,不阻塞通过
|
|
69
73
|
${MARKER}
|
|
70
74
|
`;
|
|
71
75
|
const PROPOSAL_REVIEWER_MD = (model?: string) => `---
|
|
72
76
|
name: proposal-reviewer
|
|
73
77
|
description: OpenSpec 提案审查 agent,审查提案质量
|
|
74
|
-
tools: read, bash, glob, grep
|
|
78
|
+
tools: read, bash, glob, grep
|
|
79
|
+
autoloadSkills: opsx-orchestration-protocol${model ? `\nmodel: "${model}"` : ''}
|
|
75
80
|
---
|
|
76
81
|
|
|
77
82
|
你是一位资深需求审查员。你审查 OpenSpec 提案的完整性和质量。
|
|
@@ -79,11 +84,9 @@ tools: read, bash, glob, grep${model ? `\nmodel: "${model}"` : ''}
|
|
|
79
84
|
## 启动
|
|
80
85
|
1. 从 prompt 或上下文确定要审查的提案名称
|
|
81
86
|
2. 读取 proposal.md、design.md(如果存在)
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
-
|
|
85
|
-
- 识别 \`- [R<n> coder] supersede:\` 条目:以紧随其后的新结论为权威,旧结论不再作为调研依据;只读不改写。
|
|
86
|
-
- 发现的 P0/P1 与关注点写入审查报告,由主 agent 中转回流,不直接写 scratchpad.md。
|
|
87
|
+
|
|
88
|
+
## 技能
|
|
89
|
+
- 共享编排协议(scratchpad 四区共享缓存与角色读写矩阵、审查前先读 scratchpad.md 并复用其调研结论、supersede 权威语义、结论上报去向、报告契约)由技能 \`opsx-orchestration-protocol\` 承载;若它未自动加载,用 Skill tool 读取 \`skill://opsx-orchestration-protocol\` 并遵循其流程。
|
|
87
90
|
|
|
88
91
|
## 审查维度
|
|
89
92
|
1. **完整性**:Why/What/Capabilities/Impact 是否齐全?有无缺失章节?
|
|
@@ -113,56 +116,38 @@ tools: read, bash, glob, grep${model ? `\nmodel: "${model}"` : ''}
|
|
|
113
116
|
\`\`\`
|
|
114
117
|
|
|
115
118
|
## 结论与上报
|
|
116
|
-
-
|
|
117
|
-
- **REQUEST_CHANGES + P0/P1** → 返回问题列表,由主 agent 委派 planner 修复
|
|
118
|
-
- **REQUEST_CHANGES + 仅 P2+** → 视为通过,建议可选
|
|
119
|
-
- **BLOCKED** → 说明原因,由用户决策
|
|
119
|
+
- 结论语义与上报去向见技能 \`opsx-orchestration-protocol\` 的「评审闭环与轮次上限」节(审查员只报告、不改代码)。
|
|
120
120
|
${MARKER}
|
|
121
121
|
`;
|
|
122
122
|
|
|
123
123
|
const CODER_MD = (model?: string) => `---
|
|
124
124
|
name: coder
|
|
125
125
|
description: OpenSpec 变更实现 agent,同步 tasks.md 进度
|
|
126
|
-
tools: read, write, edit, bash, glob, grep, todo
|
|
127
|
-
|
|
126
|
+
tools: read, write, edit, bash, glob, grep, todo
|
|
127
|
+
autoloadSkills: opsx-orchestration-protocol, openspec-apply-change${model ? `\nmodel: "${model}"` : ''}
|
|
128
128
|
---
|
|
129
129
|
|
|
130
130
|
你是一个快速、专注的代码实现 agent。你从 prompt 和对话上下文推断要做什么。
|
|
131
131
|
|
|
132
|
-
##
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
## 实现规则
|
|
139
|
-
- 写代码前先读现有代码;匹配项目风格和模式
|
|
140
|
-
- 完成任务后在 tasks.md 标记:\`- [ ]\` → \`- [x]\`
|
|
141
|
-
- 每轮结束时根据实际进度更新 tasks.md,确保提案状态与代码一致
|
|
142
|
-
- 代码全部完成后将所有未完成项标记为 \`- [x]\`,在 ACTION 报告中说明变更已全部实现
|
|
143
|
-
- 不修改 \`proposal.md\`、\`design.md\` 或 \`.openspec.yaml\`
|
|
132
|
+
## 技能
|
|
133
|
+
- 本角色的共享编排协议(scratchpad 四区共享缓存与角色读写矩阵、supersede 与两档处置、无关状态硬栅栏、评审闭环与轮次上限、报告契约)由技能 \`opsx-orchestration-protocol\` 承载;若它未自动加载,用 Skill tool 读取 \`skill://opsx-orchestration-protocol\` 并遵循其流程。
|
|
134
|
+
- 本角色的 OpenSpec 实现流程由技能 \`openspec-apply-change\` 承载;若它未自动加载,用 Skill tool 读取 \`skill://openspec-apply-change\` 并遵循其流程。
|
|
135
|
+
|
|
136
|
+
## 工作边界
|
|
137
|
+
- 不修改 \`proposal.md\`、\`design.md\` 或 \`.openspec.yaml\`。
|
|
144
138
|
- 直接用 read/write/edit/bash 等内置工具实现;不要尝试调用 task 二次委派。
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
- 后续轮次只做增量:仅探索未记录的文件/符号/假设,不重复 read/grep 已记录内容;append-only,不得改写他人结论。
|
|
150
|
-
|
|
151
|
-
## scratchpad.md 与代码不一致时的两档处置
|
|
152
|
-
执行中发现代码现实与 scratchpad.md 记录不一致时,按两档客观判据处置;判据仅为「偏差是否影响任何 task 的前提或产出定义」,禁止主观估量「问题大小」:
|
|
153
|
-
- **档一:事实快照过期**——偏差不影响任何 task 的前提或产出定义(纯探索性信息:路径 / 行号 / 符号 / 命令漂移)。处置:继续执行当前任务,不终止;在「实现探索」区 append supersede 修正。
|
|
154
|
-
- **档二:契约动摇**——偏差导致 tasks / proposal / design / specs 的有效性存疑(前提不成立、产出定义变、代码现状与设计决策冲突)。处置:立即停止实现,输出 STATUS: blocked(无 SESSION)并说明冲突点,交由主 agent 裁决(改提案 / 确认「现状即新设计」/ 开新 change)。
|
|
155
|
-
- supersede 标注格式:\`- [R<n> coder] supersede: <旧结论摘要>\`,随后一行以 \`- [R<n> coder]\` 标注新结论;两条均落「实现探索」分区,旧结论保留(append-only),不删除。
|
|
156
|
-
- 交付摘要 SUMMARY 义务:本轮发生过 supersede 时,输出格式中的 SUMMARY 行 MUST 提及本次 supersede 清单。
|
|
139
|
+
- 通用编码请求(无 change 目录):不创建也不维护 scratchpad.md 与 tasks.md,不得为满足流程而搭建空骨架。
|
|
140
|
+
|
|
141
|
+
## 与本次任务无关的异常状态(硬栅栏)${SCOPE_FENCE_TAG}
|
|
142
|
+
判据与处置见技能 \`opsx-orchestration-protocol\` 的栅栏节;该节仅适用于 coder(实现类)。
|
|
157
143
|
|
|
158
144
|
## 验证(交付门槛)
|
|
159
|
-
- 交付前必须跑过 **scoped 到本次改动面** 的 lint、typecheck
|
|
145
|
+
- 交付前必须跑过 **scoped 到本次改动面** 的 lint、typecheck 与针对性单测(若来自提案,proposal 约定的测试必须通过);任一未通过不得交付,先修到通过再标记 ACTION: REVIEW_REQUIRED。
|
|
146
|
+
- E2E 不属于自验证门槛,开发过程中不跑;整个变更只在终审轮由 code-reviewer 跑一次。
|
|
160
147
|
- 项目级全量校验(全仓库 lint/typecheck/\`bun test\`)不属于自验证门槛:若失败可归因于并发 sibling 的半成品改动(与本改动无关),把失败连同「已排查与本改动无关」的证据作为上下文上报(交付摘要中注明),不要被其阻塞;最终由 code-reviewer 的全局验证兜底确认。
|
|
161
148
|
|
|
162
149
|
## 协作流程(由你编排)
|
|
163
|
-
|
|
164
|
-
2. **收到审查反馈(P0/P1)**:反馈统一由 code-reviewer 回传(含全局验证失败);用 read/edit/bash 修复后重新标记 ACTION: REVIEW_REQUIRED
|
|
165
|
-
3. **轮次上限**:最多 3 轮(含首次实现与修复);仍无法解决则 STATUS: blocked
|
|
150
|
+
三条义务(报告 → 修复 → 轮次上限)见技能 \`opsx-orchestration-protocol\` 的「评审闭环与轮次上限」节。
|
|
166
151
|
|
|
167
152
|
## 输出格式
|
|
168
153
|
向主 agent 报告时以以下格式结尾:
|
|
@@ -184,62 +169,50 @@ ${MARKER}
|
|
|
184
169
|
const PLANNER_MD = (model?: string) => `---
|
|
185
170
|
name: planner
|
|
186
171
|
description: OpenSpec 提案规划 agent
|
|
187
|
-
tools: read, write, edit, bash, glob, grep, todo
|
|
172
|
+
tools: read, write, edit, bash, glob, grep, todo
|
|
173
|
+
autoloadSkills: opsx-orchestration-protocol, openspec-propose${model ? `\nmodel: "${model}"` : ''}
|
|
188
174
|
---
|
|
189
175
|
|
|
190
176
|
你是一个 OpenSpec 提案规划 agent。负责创建和更新 OpenSpec 变更提案文档。
|
|
191
177
|
|
|
192
|
-
##
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
3. 若无名称:从 prompt 理解需求,确定名称并创建新变更目录
|
|
196
|
-
4. 若提到使用 skill(如 openspec-propose):用 Skill tool 加载并执行
|
|
178
|
+
## 技能
|
|
179
|
+
- 本角色的共享编排协议(scratchpad 四区共享缓存与角色读写矩阵、supersede 与两档处置、评审闭环与轮次上限、报告契约)由技能 \`opsx-orchestration-protocol\` 承载;若它未自动加载,用 Skill tool 读取 \`skill://opsx-orchestration-protocol\` 并遵循其流程。
|
|
180
|
+
- 本角色的 OpenSpec 提案流程由技能 \`openspec-propose\` 承载;若它未自动加载,用 Skill tool 读取 \`skill://openspec-propose\` 并遵循其流程。
|
|
197
181
|
|
|
198
|
-
##
|
|
199
|
-
- **只负责提案文档**:proposal.md、design.md、tasks.md、specs/(openspec/changes/ 目录下)
|
|
182
|
+
## 工作边界
|
|
200
183
|
- 直接用内置工具;不要尝试调用 task 二次委派。
|
|
201
|
-
|
|
202
|
-
变更目录下维护 \`openspec/changes/<name>/scratchpad.md\` 作四角色共享探索缓存(与 proposal.md/design.md/tasks.md 同级),固定四阶段分区:
|
|
203
|
-
- \`## 调研与设计\`:调研结论、涉及文件(\`path:line\`)、已排除方案
|
|
204
|
-
- \`## 提案审查关注点\`:proposal-reviewer 审查报告中的 P0/P1 与关注点(经主 agent 中转回流)
|
|
205
|
-
- \`## 实现探索\`:coder 每轮 append 的关键符号/数据流、新增涉及文件、验证/构建命令、已排除假设
|
|
206
|
-
- \`## 代码审查范围\`:code-reviewer 审查报告中的 P0/P1 与验证范围结论(经主 agent 中转回流)
|
|
207
|
-
规则:
|
|
208
|
-
- **出提案时 MUST 创建** \`openspec/changes/<name>/scratchpad.md\` 并写入四阶段骨架,预填「调研与设计」初始骨架:调研结论、涉及文件、已排除方案。
|
|
209
|
-
- 后续轮次修改提案时,先读回 scratchpad.md:将 proposal-reviewer 关注点 append 进「提案审查关注点」区,新调研结论增量 append 进「调研与设计」区。
|
|
210
|
-
- 条目以 \`- [R<n> planner] <结论>\` 标注轮次与角色;append-only,不得改写或删除他人结论。
|
|
211
|
-
- 读回 scratchpad.md 时,以最新 supersede 条目为权威(识别 \`- [R<n> coder] supersede:\` 标记,其后的新结论优先于旧结论)。
|
|
184
|
+
- scratchpad.md 的创建与预填规则见协议技能;单文件可逆微改动(快车道)不创建。
|
|
212
185
|
|
|
213
186
|
## 提案拆分
|
|
214
|
-
- 接大需求先评估契约边界:能拆则拆成多个小变更,先输出拆分建议(契约边界+提案清单)待主 agent
|
|
187
|
+
- 接大需求先评估契约边界:能拆则拆成多个小变更,先输出拆分建议(契约边界+提案清单)待主 agent 确认后再开写。对冲:同批串行落地、共享编辑区的强耦合改动合并为一个提案不拆;拆分收益仅在分批落地时成立。
|
|
215
188
|
- 纲领/roadmap change 只做层级管理(阶段/依赖/验收口径),不含实现细节。
|
|
216
189
|
|
|
217
190
|
## Budget 估算
|
|
218
191
|
|
|
219
|
-
在 proposal.md 末尾输出 \`## Budget Estimate\`
|
|
192
|
+
在 proposal.md 末尾输出 \`## Budget Estimate\` 章节。估算本次提案**实际会走的**编排消耗;走快车道的微改动只估算提案成本并注明免审。
|
|
220
193
|
|
|
221
194
|
### 快速公式
|
|
222
|
-
\`总额 = (涉及文件数 × 80K +
|
|
195
|
+
\`总额 = (涉及文件数 × 80K + 编排开销) × 轮次系数\`(编排开销见下)
|
|
223
196
|
|
|
224
197
|
- **涉及文件数**:需修改或新建的源文件数量(不含 proposal/design/tasks 等提案文档本身)
|
|
225
|
-
-
|
|
198
|
+
- **编排开销**:全流程 120K tokens(planner 30K + proposal-reviewer 20K + code-reviewer(审查+全局验证) 30K + 主 agent 委派及缓冲 40K);快车道为 0(无额外审查轮,实现由主 agent 或单次 coder 完成)
|
|
226
199
|
- **轮次系数**:预期 1 轮 = 1.0,2 轮 = 1.3,3 轮 = 1.5
|
|
227
200
|
|
|
228
201
|
### 输出格式
|
|
229
202
|
\`\`\`
|
|
230
203
|
## Budget Estimate
|
|
231
204
|
- 涉及文件数: <F>
|
|
232
|
-
- 预计审查轮次: <R>
|
|
205
|
+
- 预计审查轮次: <R>(快车道免审 → 0)
|
|
233
206
|
- 预计实现 token: <F × 80K × R>K
|
|
234
|
-
- 预计审查 token: <R × 25K>K
|
|
235
|
-
-
|
|
207
|
+
- 预计审查 token: <R × 25K>K(快车道免审 → 0)
|
|
208
|
+
- 编排开销: 120K(全流程)/ 0(快车道)
|
|
236
209
|
- 建议 /goal budget: <总额>K tokens
|
|
237
210
|
\`\`\`
|
|
238
211
|
|
|
239
212
|
### 参考示例
|
|
240
213
|
- 3 个文件、1 轮 → (240 + 120) × 1.0 = 360 → \`建议 /goal budget: 360K tokens\`
|
|
241
214
|
- 8 个文件、2 轮 → (640 + 120) × 1.3 = 988 → \`建议 /goal budget: 990K tokens\`
|
|
242
|
-
-
|
|
215
|
+
- 单文件微改动(快车道)→ 不计编排开销:(80 + 0) × 1.0 = 80 → \`建议 /goal budget: 80K tokens\`
|
|
243
216
|
|
|
244
217
|
## 输出
|
|
245
218
|
完成后列出所有创建或修改的文件及变更名称。
|
|
@@ -0,0 +1,336 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Difficulty router (programme smart-model-selection W4 / change:
|
|
3
|
+
* difficulty-routed-selection) — classify a user prompt into a task band and
|
|
4
|
+
* derive write-set tier target offsets.
|
|
5
|
+
*
|
|
6
|
+
* Pipeline position (design D1): hard-constraint filtering (existing pool
|
|
7
|
+
* layer, untouched) → difficulty banding (THIS module) → in-band selection
|
|
8
|
+
* (the W1-W3 frozen order runs unchanged under the offset targets) → stall
|
|
9
|
+
* escalation + continuity guard → quality band knob.
|
|
10
|
+
*
|
|
11
|
+
* `classifyTask` is a PURE function: deterministic, no clock, no RNG, no
|
|
12
|
+
* session state, linear in prompt length (the hot path must stay far cheaper
|
|
13
|
+
* than generation — no LLM call, LiteLLM hybrid scorer explicitly out of
|
|
14
|
+
* scope per proposal Non-Goals).
|
|
15
|
+
*
|
|
16
|
+
* Signal families and caps are a portable subset of LiteLLM's
|
|
17
|
+
* complexity_router weights (config.py:429-446); band boundaries 0.15 / 0.35
|
|
18
|
+
* / 0.60 are LiteLLM 同款. Density-scaled families saturate at their cap, so
|
|
19
|
+
* a family's contribution is its weight × intensity, capped at the weight.
|
|
20
|
+
*
|
|
21
|
+
* Failure semantics are SAFE-SIDE (the LiteLLM failure mode "no evidence →
|
|
22
|
+
* 0.0 → cheapest band by default" is deliberately inverted):
|
|
23
|
+
* `classifyTaskSafe` maps an exception or a confidence below
|
|
24
|
+
* `CONFIDENCE_FLOOR` to `complex` (the strong band). Ambiguity never
|
|
25
|
+
* silently downgrades — cost ambiguity is the opt-in adopter's informed
|
|
26
|
+
* choice (design D2).
|
|
27
|
+
*
|
|
28
|
+
* plan-mode awareness: the host's plan-mode state is NOT visible to plugins
|
|
29
|
+
* (extension events and ctx carry no planMode field — verified at proposal
|
|
30
|
+
* time), so plan/intent detection is prompt-text-based; the plan marker
|
|
31
|
+
* vocabulary lives in `REASONING_MARKERS`.
|
|
32
|
+
*
|
|
33
|
+
* Confidence formula (supersedes the D2 sketch「命中信号族数归一 × 边界距
|
|
34
|
+
* 离」): evidence saturation × margin-modulated factor, i.e.
|
|
35
|
+
* confidence = min(1, Σ|contributions| / EVIDENCE_CAP)
|
|
36
|
+
* × (MARGIN_FLOOR + (1 − MARGIN_FLOOR) × min(1, dist/0.15))
|
|
37
|
+
* with EVIDENCE_CAP = 0.25, MARGIN_FLOOR = 0.6. The sketch's literal form
|
|
38
|
+
* (hitCount/7 × margin) mathematically caps at 0.29 for clean short
|
|
39
|
+
* housekeeping prompts and cannot reach the spec floor of 0.35, while any
|
|
40
|
+
* "applicable families" normalization scores empty ambiguous prompts as
|
|
41
|
+
* confident. The implemented form keeps both factors — evidence dominates,
|
|
42
|
+
* boundary margin modulates — and satisfies every spec scenario: rich
|
|
43
|
+
* reasoning prompts and clean housekeeping commands classify with
|
|
44
|
+
* confidence ≥ floor; evidence-free ambiguous phrases stay below it (and
|
|
45
|
+
* thus take the safe-side strong band).
|
|
46
|
+
*/
|
|
47
|
+
|
|
48
|
+
import { type TierName, TIER_NAMES, TIER_RANK } from './model-tiers.js';
|
|
49
|
+
import { OMP_ROLE_TO_TIER, OMP_WRITE_SET_ROLES } from './model-roles.js';
|
|
50
|
+
import type { SelectionContext } from '../index.js';
|
|
51
|
+
|
|
52
|
+
/** Task difficulty bands, ordered from cheapest to strongest. */
|
|
53
|
+
export const TASK_BANDS = ['simple', 'standard', 'complex', 'reasoning'] as const;
|
|
54
|
+
export type TaskBand = (typeof TASK_BANDS)[number];
|
|
55
|
+
|
|
56
|
+
/** Numeric rank of a band (higher = stronger); the guard's upgrade comparison. */
|
|
57
|
+
export function bandRank(band: TaskBand): number {
|
|
58
|
+
return TASK_BANDS.indexOf(band);
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Score → band thresholds (LiteLLM 同款): <0.15 simple, <0.35 standard, <0.60 complex, else reasoning. */
|
|
62
|
+
export const BAND_BOUNDARIES = [0.15, 0.35, 0.6] as const;
|
|
63
|
+
|
|
64
|
+
/** Confidence floor: below this (or on a classifier exception) the safe side applies. */
|
|
65
|
+
export const CONFIDENCE_FLOOR = 0.35;
|
|
66
|
+
|
|
67
|
+
/** Write-set tier target offset per band (design D3), clamped to [tiny, top]. */
|
|
68
|
+
export const BAND_TIER_OFFSET: Record<TaskBand, number> = {
|
|
69
|
+
simple: -1,
|
|
70
|
+
standard: 0,
|
|
71
|
+
complex: 1,
|
|
72
|
+
reasoning: 2,
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
/** Quality band knob values (design D6). */
|
|
76
|
+
export const QUALITY_PREFERENCES = ['balanced', 'cost', 'quality'] as const;
|
|
77
|
+
export type QualityPreference = (typeof QUALITY_PREFERENCES)[number];
|
|
78
|
+
|
|
79
|
+
/** `difficulty_routing` config (design D7). */
|
|
80
|
+
export interface RoutingConfig {
|
|
81
|
+
enabled: boolean;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** Classification result: the band plus evidence confidence in [0,1]. */
|
|
85
|
+
export interface Classification {
|
|
86
|
+
band: TaskBand;
|
|
87
|
+
confidence: number;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/** Optional classification context. Reserved for hybrid scoring — the
|
|
91
|
+
* near-window tool signatures are carried for future use and logged by the
|
|
92
|
+
* caller, never scored (design D2: keep the pure function state-free). */
|
|
93
|
+
export interface ClassifyContext {
|
|
94
|
+
toolSignatures?: readonly string[];
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
// ── signal vocabularies ─────────────────────────────────────────────────
|
|
98
|
+
|
|
99
|
+
const CJK = /[\u4e00-\u9fff\u3040-\u30ff\uac00-\ud7af]/;
|
|
100
|
+
|
|
101
|
+
/** Plan / design / reasoning markers. ≥2 distinct → full family weight. */
|
|
102
|
+
const REASONING_MARKERS: RegExp[] = [
|
|
103
|
+
/\bthink/i, /\bplan/i, /\bdesign/i, /\barchitect/i, /\brefactor/i,
|
|
104
|
+
/\banalyz/i, /\banalys/i, /\bwhy\b/i, /\boptimiz/i, /\btrade[- ]?off/i,
|
|
105
|
+
/\bstrategy/i, /\binvestigat/i, /\broot cause/i, /\bevaluat/i, /\bapproach\b/i,
|
|
106
|
+
/方案/, /设计/, /规划/, /推理/, /分析/, /为什么/, /逐步/, /重构/, /架构/,
|
|
107
|
+
/权衡/, /思路/, /策略/, /评估/, /优化/, /调研/, /根因/, /梳理/,
|
|
108
|
+
];
|
|
109
|
+
|
|
110
|
+
/** Code-presence probes: fences, inline code, shell command lines, file paths. */
|
|
111
|
+
const CODE_FENCE = /```/;
|
|
112
|
+
const CODE_INLINE = /`[^`\n]+`/;
|
|
113
|
+
const CODE_COMMAND = /(^|\n)\s*(\$|npm|bun|pnpm|yarn|cargo|make|sudo|python3?|pip)\s/;
|
|
114
|
+
const CODE_PATH = /\b[\w@-]+(?:\/[\w@.-]+)+|\b[\w-]+\.(?:ts|tsx|js|jsx|mjs|cjs|py|go|rs|java|kt|rb|php|c|h|cpp|hpp|cs|swift|sql|json|ya?ml|toml|sh)\b/g;
|
|
115
|
+
|
|
116
|
+
/** Housekeeping verbs (negative signal). Gated on short prompts without fences. */
|
|
117
|
+
const SIMPLE_VERBS: RegExp[] = [
|
|
118
|
+
/\bformat\b/i, /\blint\b/i, /\brename\b/i, /\bfix typo/i, /\bclean ?up\b/i,
|
|
119
|
+
/\btidy\b/i, /\bprettify\b/i, /\blist\b/i, /\bshow\b/i, /\bprint\b/i,
|
|
120
|
+
/\bopen\b/i, /\bcount\b/i, /\bwhere is\b/i,
|
|
121
|
+
/格式化/, /重命名/, /改名/, /清理/, /整理/, /列出/, /查看/, /打印/, /删除/, /打开/, /找到/, /搜索/, /数一下/,
|
|
122
|
+
];
|
|
123
|
+
/** Above this token count a housekeeping verb no longer reads as a simple task. */
|
|
124
|
+
const SIMPLE_MAX_TOKENS = 30;
|
|
125
|
+
|
|
126
|
+
/** Technical terms; density (distinct hits) scales the family weight. */
|
|
127
|
+
const TECH_TERMS: RegExp[] = [
|
|
128
|
+
/\bapi\b/i, /\bfunction/i, /\bclass\b/i, /\btype\b/i, /\binterface\b/i,
|
|
129
|
+
/\bbug\b/i, /\berror\b/i, /\bexception\b/i, /\btest\b/i, /\bdeploy/i,
|
|
130
|
+
/\bdatabase\b/i, /\bsql\b/i, /\bregex/i, /\bcache\b/i, /\bthread/i,
|
|
131
|
+
/\basync\b/i, /\bschema\b/i, /\bendpoint\b/i, /\bcompil/i, /\bbuild\b/i,
|
|
132
|
+
/\blint\b/i, /\bconfig\b/i, /\bdependenc/i, /\bmerge\b/i, /\bbranch\b/i,
|
|
133
|
+
/\bcommit\b/i, /\bserver\b/i, /\bclient\b/i, /\bdocker\b/i, /\bgit\b/i,
|
|
134
|
+
/函数/, /接口/, /类型/, /变量/, /数据库/, /编译/, /部署/, /缓存/, /线程/,
|
|
135
|
+
/异步/, /依赖/, /服务端/, /客户端/, /配置/, /测试/, /组件/, /模块/,
|
|
136
|
+
/算法/, /数据结构/, /性能/, /内存/,
|
|
137
|
+
];
|
|
138
|
+
|
|
139
|
+
/** Multi-step patterns (numbered lists, sequencing words). */
|
|
140
|
+
const MULTI_STEP: RegExp[] = [
|
|
141
|
+
/\bfirst\b/i, /\bthen\b/i, /\bafter that\b/i, /\bfinally\b/i,
|
|
142
|
+
/\bstep [-#]?\d/i, /(^|\n)\s*\d+[.、)]\s/,
|
|
143
|
+
/首先/, /其次/, /然后/, /接着/, /最后/, /第一步/, /第二步/, /先.{0,12}再/,
|
|
144
|
+
];
|
|
145
|
+
|
|
146
|
+
/** Interrogative cues (multiple question marks or question words). */
|
|
147
|
+
const QUESTION_WORDS: RegExp[] = [
|
|
148
|
+
/\bwhy\b/i, /\bhow\b/i, /\bcompare\b/i, /\bexplain\b/i,
|
|
149
|
+
/为什么/, /怎么/, /如何/, /区别/, /原理/, /原因/,
|
|
150
|
+
];
|
|
151
|
+
|
|
152
|
+
/** Family weight caps (design D2). */
|
|
153
|
+
const WEIGHT_REASONING = 0.25;
|
|
154
|
+
const WEIGHT_CODE = 0.3;
|
|
155
|
+
const WEIGHT_TECH = 0.25;
|
|
156
|
+
const WEIGHT_SIMPLE = -0.05;
|
|
157
|
+
const WEIGHT_MULTI_STEP = 0.03;
|
|
158
|
+
const WEIGHT_QUESTION = 0.02;
|
|
159
|
+
const WEIGHT_TOKEN = 0.1;
|
|
160
|
+
|
|
161
|
+
/** Token-count log term: 0.10 × min(1, log2(1 + tokens) / 4). */
|
|
162
|
+
const TOKEN_LOG_DIVISOR = 4;
|
|
163
|
+
|
|
164
|
+
/** Confidence tuning (module header formula). */
|
|
165
|
+
const EVIDENCE_CAP = 0.25;
|
|
166
|
+
const MARGIN_SCALE = 0.15;
|
|
167
|
+
const MARGIN_FLOOR = 0.6;
|
|
168
|
+
|
|
169
|
+
/** Count tokens: CJK chars individually plus non-CJK whitespace words. */
|
|
170
|
+
function countTokens(prompt: string): number {
|
|
171
|
+
const cjk = prompt.match(/[\u4e00-\u9fff\u3040-\u30ff\uac00-\ud7af]/g)?.length ?? 0;
|
|
172
|
+
const words = prompt.replace(/[\u4e00-\u9fff\u3040-\u30ff\uac00-\ud7af]/g, ' ').split(/\s+/).filter(Boolean).length;
|
|
173
|
+
return cjk + words;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
function countMatches(patterns: RegExp[], text: string): number {
|
|
177
|
+
let n = 0;
|
|
178
|
+
for (const p of patterns) {
|
|
179
|
+
p.lastIndex = 0;
|
|
180
|
+
if (p.test(text)) n++;
|
|
181
|
+
}
|
|
182
|
+
return n;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Classify a prompt into a task band (pure function — design D2).
|
|
187
|
+
*
|
|
188
|
+
* Score s = Σ family contributions, clamped to [0, 1]; band by
|
|
189
|
+
* BAND_BOUNDARIES. Confidence per the module header formula: evidence
|
|
190
|
+
* saturation (Σ|contributions| / EVIDENCE_CAP, capped at 1) × a
|
|
191
|
+
* margin-modulated factor (0.6 + 0.4 × normalized distance to the nearest
|
|
192
|
+
* band boundary). `context` is accepted for the hybrid extension point and
|
|
193
|
+
* never scored.
|
|
194
|
+
*/
|
|
195
|
+
export function classifyTask(prompt: string, _context?: ClassifyContext): Classification {
|
|
196
|
+
const text = prompt ?? '';
|
|
197
|
+
const tokens = countTokens(text);
|
|
198
|
+
const fence = CODE_FENCE.test(text);
|
|
199
|
+
const inline = CODE_INLINE.test(text);
|
|
200
|
+
|
|
201
|
+
// Housekeeping verb hit — gated on short, fence-free prompts.
|
|
202
|
+
const simpleHit = tokens <= SIMPLE_MAX_TOKENS && !fence && countMatches(SIMPLE_VERBS, text) > 0;
|
|
203
|
+
|
|
204
|
+
// Code presence. A housekeeping command's file arguments (rename X to Y)
|
|
205
|
+
// are payload, not code — path probes are suppressed for gated simple
|
|
206
|
+
// prompts so "rename utils.ts" does not read as a coding task.
|
|
207
|
+
let codeSignals = 0;
|
|
208
|
+
if (fence) codeSignals += 1;
|
|
209
|
+
if (inline) codeSignals += 1;
|
|
210
|
+
if (CODE_COMMAND.test(text)) codeSignals += 1;
|
|
211
|
+
const pathMatch = simpleHit && !inline ? null : CODE_PATH;
|
|
212
|
+
if (pathMatch) {
|
|
213
|
+
CODE_PATH.lastIndex = 0;
|
|
214
|
+
codeSignals += Math.min(2, (text.match(CODE_PATH) ?? []).length);
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
const reasoningDistinct = countMatches(REASONING_MARKERS, text);
|
|
218
|
+
const techDistinct = countMatches(TECH_TERMS, text);
|
|
219
|
+
const multiStep = countMatches(MULTI_STEP, text) > 0;
|
|
220
|
+
const questionMarks = (text.match(/\?/g) ?? []).length >= 2;
|
|
221
|
+
const questionWord = countMatches(QUESTION_WORDS, text) > 0;
|
|
222
|
+
const question = questionMarks || questionWord;
|
|
223
|
+
|
|
224
|
+
const cReasoning = WEIGHT_REASONING * Math.min(1, reasoningDistinct / 2);
|
|
225
|
+
const cCode = WEIGHT_CODE * Math.min(1, codeSignals / 2);
|
|
226
|
+
const cTech = WEIGHT_TECH * Math.min(1, techDistinct / 3);
|
|
227
|
+
const cSimple = simpleHit ? WEIGHT_SIMPLE : 0;
|
|
228
|
+
const cMulti = multiStep ? WEIGHT_MULTI_STEP : 0;
|
|
229
|
+
const cQuestion = question ? WEIGHT_QUESTION : 0;
|
|
230
|
+
const cToken = WEIGHT_TOKEN * Math.min(1, Math.log2(1 + tokens) / TOKEN_LOG_DIVISOR);
|
|
231
|
+
|
|
232
|
+
const score = Math.max(0, Math.min(1, cReasoning + cCode + cTech + cSimple + cMulti + cQuestion + cToken));
|
|
233
|
+
|
|
234
|
+
let band: TaskBand = 'reasoning';
|
|
235
|
+
for (let i = 0; i < BAND_BOUNDARIES.length; i++) {
|
|
236
|
+
if (score < BAND_BOUNDARIES[i]) {
|
|
237
|
+
band = TASK_BANDS[i];
|
|
238
|
+
break;
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
// Confidence: evidence saturation × margin-modulated factor (module header).
|
|
243
|
+
const fired = Math.abs(cReasoning) + Math.abs(cCode) + Math.abs(cTech)
|
|
244
|
+
+ Math.abs(cSimple) + Math.abs(cMulti) + Math.abs(cQuestion) + Math.abs(cToken);
|
|
245
|
+
const evidence = Math.min(1, fired / EVIDENCE_CAP);
|
|
246
|
+
let dist = 1 - score; // reasoning band: no upper boundary in BAND_BOUNDARIES
|
|
247
|
+
for (const b of BAND_BOUNDARIES) {
|
|
248
|
+
const d = Math.abs(score - b);
|
|
249
|
+
if (d < dist) dist = d;
|
|
250
|
+
}
|
|
251
|
+
const margin = Math.min(1, dist / MARGIN_SCALE);
|
|
252
|
+
const confidence = evidence * (MARGIN_FLOOR + (1 - MARGIN_FLOOR) * margin);
|
|
253
|
+
|
|
254
|
+
return { band, confidence };
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* Safe-side wrapper (design D2): an exception inside the classifier or a
|
|
259
|
+
* confidence below CONFIDENCE_FLOOR yields `complex` (the strong band) —
|
|
260
|
+
* never a silent downgrade to the cheapest band. The `classify` parameter is
|
|
261
|
+
* a test-only injection seam (defaults to the pure classifier).
|
|
262
|
+
*/
|
|
263
|
+
export function classifyTaskSafe(
|
|
264
|
+
prompt: string,
|
|
265
|
+
context?: ClassifyContext,
|
|
266
|
+
classify: (p: string, c?: ClassifyContext) => Classification = classifyTask,
|
|
267
|
+
): Classification {
|
|
268
|
+
try {
|
|
269
|
+
const result = classify(prompt, context);
|
|
270
|
+
if (result.confidence < CONFIDENCE_FLOOR) {
|
|
271
|
+
return { band: 'complex', confidence: result.confidence };
|
|
272
|
+
}
|
|
273
|
+
return result;
|
|
274
|
+
} catch {
|
|
275
|
+
return { band: 'complex', confidence: 0 };
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* Apply the band offset to write-set tier targets (design D3). ONLY
|
|
281
|
+
* `OMP_WRITE_SET_ROLES` ({smol, default, slow, vision}) are redirected —
|
|
282
|
+
* non-write-set roles are never touched (the pick-model-ux minimal-write-set
|
|
283
|
+
* red line). Base target = the map's explicit entry, else the built-in role
|
|
284
|
+
* default (model_role_tiers overrides keep working); the `skip` sentinel is
|
|
285
|
+
* preserved verbatim (a user opt-out is never rewritten). `stallActive`
|
|
286
|
+
* bumps `default` one extra tier (design D4), then the result clamps to
|
|
287
|
+
* [tiny, top]. Returns a NEW map; the input is not mutated.
|
|
288
|
+
*/
|
|
289
|
+
export function applyBandOffset(
|
|
290
|
+
roleTiers: Record<string, TierName | 'skip'>,
|
|
291
|
+
band: TaskBand,
|
|
292
|
+
opts?: { stallActive?: boolean },
|
|
293
|
+
): Record<string, TierName | 'skip'> {
|
|
294
|
+
const out = { ...roleTiers };
|
|
295
|
+
for (const role of OMP_WRITE_SET_ROLES) {
|
|
296
|
+
const base = out[role] ?? OMP_ROLE_TO_TIER[role];
|
|
297
|
+
if (base === 'skip' || base === undefined) continue;
|
|
298
|
+
let rank = TIER_RANK[base] + BAND_TIER_OFFSET[band];
|
|
299
|
+
if (opts?.stallActive && role === 'default') rank += 1;
|
|
300
|
+
const clamped = Math.max(0, Math.min(TIER_NAMES.length - 1, rank));
|
|
301
|
+
out[role] = TIER_NAMES[clamped];
|
|
302
|
+
}
|
|
303
|
+
return out;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// ── routing runtime (assembly-layer payload, design D8.5) ───────────────
|
|
307
|
+
|
|
308
|
+
/** Continuity-guard inputs carried on the runtime (see continuity-guard.ts). */
|
|
309
|
+
export interface GuardRuntimeInput {
|
|
310
|
+
/** `continuity_guard.weight` — minimum confidence for an upgrade switch. */
|
|
311
|
+
weight: number;
|
|
312
|
+
/** Prior selection (the plugin closure's cached SelectionContext). */
|
|
313
|
+
incumbent: SelectionContext | null;
|
|
314
|
+
/** Band recorded on the prior selection; null = unknown (neutral standard). */
|
|
315
|
+
priorBand: TaskBand | null;
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Per-recompute routing payload threaded into computeSelectionSync
|
|
320
|
+
* (design D8.5). `undefined` = every switch off + balanced — the zero-drift
|
|
321
|
+
* contract is that the selection path never sees a routing object at all.
|
|
322
|
+
*/
|
|
323
|
+
export interface RoutingRuntime {
|
|
324
|
+
/** Task band driving write-set tier offsets. */
|
|
325
|
+
band: TaskBand;
|
|
326
|
+
/** Classification confidence (annotation + guard upgrade threshold). */
|
|
327
|
+
confidence: number;
|
|
328
|
+
/** Stall escalation in effect (default alias +1, design D4). */
|
|
329
|
+
stallActive: boolean;
|
|
330
|
+
/** /pick-model explicit path: guard bypassed, no fresh classification, no band annotation. */
|
|
331
|
+
explicit: boolean;
|
|
332
|
+
/** Quality band knob (undefined/balanced = existing comparator branches, byte-identical). */
|
|
333
|
+
qualityPreference?: QualityPreference;
|
|
334
|
+
/** Continuity guard inputs (undefined = guard off / explicit). */
|
|
335
|
+
guard?: GuardRuntimeInput;
|
|
336
|
+
}
|