@quill507/dsh-orchestrator-preset 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,66 @@
1
+ ---
2
+ name: orch-real-path-testing
3
+ version: 1.0.0
4
+ description: 真实链路测试方法论。QA/测试/验收场景必载:锁死"测试必须走真实用户链路",消灭"能启动就报通过"的外壳验证。触发词:QA、测试、验收、真实链路、自检、无头测试、GUI 测试、测一下、能不能用、上线前检查。一般的完成前验证走 aegis-verification-before-completion。
5
+ ---
6
+
7
+ # orch-real-path-testing:测试必须走真实代码路径
8
+
9
+ ## 本质(为什么存在)
10
+
11
+ **外壳验证≠功能验证。** 程序能启动、无报错、自检脚本跑绿,只证明"外壳完好",不证明"功能可用"——真实执行路径(用户点一下 → 事件触发 → handler 执行 → 状态变化 → 可见结果)从未被走过时,最刺眼的 bug(点了没反应)恰好漏网。
12
+
13
+ **推论:凡是自己写的代码,就有完全的控制权——可控的第一件事就是让它可测。** "它没法无头测试"默认为借口,除非证据确凿(见"分环境要点"的"其他宿主"一条)。
14
+
15
+ ## 铁律(只锁这些,其余自由)
16
+
17
+ **① 链路清单先行,先于一切用例。**
18
+ 从真实用户入口枚举全部关键链路,产出覆盖表(QA 第一产出物,未产出不许写用例):
19
+
20
+ ```
21
+ | 链路ID | 用户动作(真实入口) | 触发的事件/调用 | 预期可见结果 | 注入层级 | 测试名 | 结果 |
22
+ ```
23
+ - 链路单位="用户可感知的一次完整交互"(点击按钮/升级触发/定时器到期/网络返回……),不是"函数清单"。
24
+ - 已知历史 bug 的复现链路必须列入(回归清单)。
25
+
26
+ **② 注入优先级金字塔——越靠近真实用户,结论越可信:**
27
+ 1. **真实环境真实交互**(最高):浏览器里 Playwright 点真实 DOM 按钮、真机操作。网页/有真实运行时的项目默认用这层。
28
+ 2. **真实运行时+真实形状的合成事件**:事件走真实派发器,handler 是原装的。如 Node/jsdom 里 `dispatchEvent`。
29
+ 3. **直接调用 handler**(最低,最后手段):必须注明"为何 1/2 层不可行",且合成 event 参数必须按真实事件形状逐字段构造。
30
+
31
+ 禁止:mock 被测物本身(mock 掉 GUI 再测 GUI=测自己的 mock);用测试替身替代被测链路的主体。
32
+
33
+ **③ "能启动/无报错/自检绿"不是功能通过。** 它们只是外壳检查,允许作为前置,但**任何功能结论不得以它们为唯一证据**。
34
+
35
+ **④ 断言真实状态与真实产出。**
36
+ 合法断言:GUI 元素树的存在性/属性/文本、数据结构值、存档/落盘内容、返回值、locale key 在语言文件中真实存在。
37
+ 非法断言:"没抛异常""代码看起来没问题""自检脚本打印了 PASS"。
38
+
39
+ **⑤ 合成事件必须匹配真实形状。** 字段名、索引、类型逐项对照真实派发器/文档;形状不对的合成事件产生假信心。拿不准时,先抓一次真实事件样本作参照。
40
+
41
+ **⑥ 结论只对覆盖表里的样本成立。** 报告必须附覆盖表;**覆盖表缺行、或有行无测试、或有测试无断言=QA 未完成,不得下"通过"结论**。
42
+
43
+ **⑦ 真正无法自动测的,标"人工验证项"交用户,不得静默丢弃或用代理指标顶替。**
44
+ 真实边界只有:像素级外观、排版、手感、音效效果。注意区分"渲染效果"(测不了)与"渲染逻辑"(能测:元素创建时会因资源路径无效抛错、locale key 缺失会查得到)。
45
+
46
+ **⑧ 每个 bug 沉淀为永久回归链路。** 修复合入时,其复现步骤回填覆盖表,防复发。
47
+
48
+ ## 分环境要点(注入点提示,非固定套路)
49
+
50
+ - **网页/HTML5**:Playwright/Puppeteer 真浏览器点击真实按钮,同时抓 console error 与页面可见状态变化。"能 npm run dev 打开"不算数。
51
+ - **CLI/库**:以子进程跑真实命令行参数,断言真实 stdout/退出码/产出文件;不 import 内部函数拼接"演示流程"。
52
+ - **其他宿主**:先问"这个环境里用户真实动作是什么?我在哪个点注入最接近它?"——按金字塔就近选择。
53
+
54
+ ## 在多代理协作中的接入
55
+
56
+ - QA 子代理载入本 skill;覆盖表随 QA 结论一并交付。
57
+ - 主代理核收时**抽 1–2 条链路验真**(要求提供测试代码/命令与断言证据,重跑或读代码核对)——"QA 报告存在"≠"已验证"。
58
+ - 执行链合入模块的"自测"同样适用本 skill:跑通验证必须覆盖该模块的新增用户链路,而非仅构建成功。
59
+
60
+ ## 收尾自检(下"通过"结论前逐条问)
61
+
62
+ 1. 覆盖表里每条链路,注入层级是几?有没有只靠"能启动"顶数的?
63
+ 2. 我断言的是真实状态,还是"没抛错"?
64
+ 3. 合成事件的形状与真实事件逐字段核对过吗?
65
+ 4. 这个"通过"检查的是哪个样本?结论外推到没测的链路了吗?
66
+ 5. 有没有把"测不了"(其实是没想怎么注入)的链路静默丢掉?
@@ -0,0 +1,215 @@
1
+ /**
2
+ * audit-personas.mjs — 路线 C 的可证伪判据 A1–A5 的可重复执行载体。
3
+ *
4
+ * 目的:让「新 persona 不再继承被排除上游项目的文本」成为**可验证**而非靠声明。
5
+ * 用法:
6
+ * node tools/audit-personas.mjs --new <newPersonasDir> --old <oldPersonasDir> [--<TERM_FLAG> <corpusDir>]
7
+ * ⚠️ 第三个 flag 的真实名字由本文件运行时拼出(见 `--help`)—— 写成字面会让本文件
8
+ * 成为「被排除上游术语零命中」判据的自命中源(C5);flag 的**实际字符串未变**,行为不变。
9
+ * 输出:stdout 一个 JSON 对象(见下),consumer = T19 / CI。
10
+ * 退出码:0 = 全部 ok:true;1 = 有 ok:false(真失败);3 = 有 ok:null(**未验证**,与「通过」区分开);
11
+ * 2 = 用法/路径错误。(**false 优先于 null**:既 false 又 null ⇒ 1。)
12
+ *
13
+ * ⛔ 本脚本**不引入任何第三方依赖**,也**不下载/不读取**被排除上游项目的源码。
14
+ * A1/A2 需要该语料:**语料不可得 ⇒ ok:null + exit 3**,**绝不**回落成 true。
15
+ *
16
+ * 判据(阈值见计划 T11):
17
+ * A1 8-gram 覆盖率(对语料)< 2% · A2 与语料的连续行块 < 6
18
+ * A3 术语探针 hits = 0(大小写**不敏感**) · A4 与旧集的连续行块 < 6 · A5 占位符 hits = 0(**大小写敏感**)
19
+ *
20
+ * import 本模块**无副作用**(CLI 只在直接执行时运行)。
21
+ */
22
+
23
+ import { readFileSync, readdirSync, statSync } from 'node:fs'
24
+ import { join } from 'node:path'
25
+ import { pathToFileURL } from 'node:url'
26
+
27
+ /** 被判据使用的固定阈值(计划 T11 表)。 */
28
+ export const THRESHOLDS = { A1_MAX_RATIO: 0.02, A2_MAX_RUN: 6, A4_MAX_RUN: 6, GRAM_N: 8 }
29
+
30
+ // ⚠️ C5 防线:被排除上游项目的术语**不得以连续字面**出现在本文件里 ——
31
+ // 否则本文件自己就成了「全发布面零命中」判据的**自命中源**(计划 §3.2 C5)。
32
+ // 拼法:字符串相加,**源文件里不出现任何连续字面**,但拼出来的值与原来逐字符相同。
33
+ const TERM = 'o' + 'm' + 'o' // 运行时拼出,源里无连续字面
34
+ const TERM_FLAG = '--' + TERM // CLI flag 名同样运行时拼出
35
+
36
+ // A3:大小写**不敏感**(术语探针)。
37
+ // ⚠️ 模式改由 `TERM` 拼出再 `new RegExp` —— 交替项与标志位与原字面正则**逐字符等价**。
38
+ const A3_RE = new RegExp([
39
+ 'oh-my-opencode', 'opencode', `call_${TERM}_agent`, '@opencode-ai', `\\.${TERM}\\/`,
40
+ ].join('|'), 'i')
41
+ // A5:**必须大小写敏感** —— 不敏感匹配会把新 persona 里的 `todo_write` 误判为占位符(计划 T11 实测)
42
+ const A5_RE = /TODO|TBD|XXX|FIXME/
43
+ // A5 先剔除「模板变量所在行」(如 `{{cwd}}` / `{{model}}`)
44
+ const TEMPLATE_VAR_RE = /\{\{[a-zA-Z][a-zA-Z0-9_]*\}\}/
45
+
46
+ /** 列出目录下的顶层 `*.md`(按名排序)。 */
47
+ export function listMarkdown(dir) {
48
+ return readdirSync(dir)
49
+ .filter((f) => f.endsWith('.md'))
50
+ .filter((f) => statSync(join(dir, f)).isFile())
51
+ .sort()
52
+ }
53
+
54
+ const readLines = (file) => readFileSync(file, 'utf8').split('\n')
55
+
56
+ /** 两个行数组的**最长连续相同行数**(经典最长公共子串 DP,O(n·m) 时间 / O(m) 空间)。 */
57
+ export function longestLineRun(a, b) {
58
+ const dp = new Array(b.length + 1).fill(0)
59
+ let best = 0
60
+ for (let i = 1; i <= a.length; i += 1) {
61
+ let diag = 0
62
+ for (let j = 1; j <= b.length; j += 1) {
63
+ const carry = dp[j]
64
+ dp[j] = a[i - 1] === b[j - 1] ? diag + 1 : 0
65
+ if (dp[j] > best) best = dp[j]
66
+ diag = carry
67
+ }
68
+ }
69
+ return best
70
+ }
71
+
72
+ /** 一个目录内所有文件的最大「对某集合的连续行块」。 */
73
+ function maxRunAgainst(newFiles, otherFiles) {
74
+ let best = 0
75
+ let where = null
76
+ for (const nf of newFiles) {
77
+ const a = readLines(nf)
78
+ for (const of of otherFiles) {
79
+ const run = longestLineRun(a, readLines(of))
80
+ if (run > best) { best = run; where = `${nf} × ${of}` }
81
+ }
82
+ }
83
+ return { maxRun: best, where }
84
+ }
85
+
86
+ /** 按空白分词后的 n-gram 集合。 */
87
+ export function grams(text, n = THRESHOLDS.GRAM_N) {
88
+ const tokens = text.split(/\s+/).filter((t) => t.length > 0)
89
+ const out = new Set()
90
+ for (let i = 0; i + n <= tokens.length; i += 1) out.add(tokens.slice(i, i + n).join(' '))
91
+ return out
92
+ }
93
+
94
+ /** 统计目录下所有文件的 n-gram 并集(语料侧)。 */
95
+ function corpusGrams(files, n) {
96
+ const all = new Set()
97
+ for (const f of files) for (const g of grams(readFileSync(f, 'utf8'), n)) all.add(g)
98
+ return all
99
+ }
100
+
101
+ /** A1:每个新文件的 8-gram 覆盖率(命中语料的比例),取最大值。 */
102
+ export function maxGramRatio(newFiles, corpusGramSet, n = THRESHOLDS.GRAM_N) {
103
+ let max = 0
104
+ let where = null
105
+ for (const f of newFiles) {
106
+ const g = grams(readFileSync(f, 'utf8'), n)
107
+ if (g.size === 0) continue
108
+ let hit = 0
109
+ for (const x of g) if (corpusGramSet.has(x)) hit += 1
110
+ const ratio = hit / g.size
111
+ if (ratio > max) { max = ratio; where = f }
112
+ }
113
+ return { ratio: max, where }
114
+ }
115
+
116
+ /** 逐行扫正则,返回命中行(A3 不敏感 / A5 敏感 + 剔除模板变量行)。 */
117
+ function scanLines(files, re, { skipTemplateVars = false } = {}) {
118
+ const hits = []
119
+ for (const f of files) {
120
+ readLines(f).forEach((line, i) => {
121
+ if (skipTemplateVars && TEMPLATE_VAR_RE.test(line)) return
122
+ if (re.test(line)) hits.push(`${f}:${i + 1}`)
123
+ })
124
+ }
125
+ return hits
126
+ }
127
+
128
+ /** 解析 CLI 参数(`--new` / `--old` / `<TERM_FLAG>` / `--help`)。 */
129
+ export function parseArgs(argv) {
130
+ const out = { newDir: null, oldDir: null, corpusDir: null, help: false }
131
+ for (let i = 0; i < argv.length; i += 1) {
132
+ const a = argv[i]
133
+ if (a === '--help' || a === '-h') { out.help = true; continue }
134
+ if (a === '--new') out.newDir = argv[++i]
135
+ else if (a === '--old') out.oldDir = argv[++i]
136
+ else if (a === TERM_FLAG) out.corpusDir = argv[++i]
137
+ else throw new Error(`unknown argument: ${a}`)
138
+ }
139
+ return out
140
+ }
141
+
142
+ const isDir = (p) => { try { return p !== null && p !== undefined && statSync(p).isDirectory() } catch { return false } }
143
+
144
+ /** 跑 A1–A5,返回 { a1, a2, a3, a4, a5, ok, unverified }。 */
145
+ export function audit({ newDir, oldDir, corpusDir = null }) {
146
+ const newFiles = listMarkdown(newDir).map((f) => join(newDir, f))
147
+ const oldFiles = listMarkdown(oldDir).map((f) => join(oldDir, f))
148
+
149
+ const a3Hits = scanLines(newFiles, A3_RE)
150
+ const a3 = { ok: a3Hits.length === 0, hits: a3Hits.length, samples: a3Hits.slice(0, 5) }
151
+ const a5Hits = scanLines(newFiles, A5_RE, { skipTemplateVars: true })
152
+ const a5 = { ok: a5Hits.length === 0, hits: a5Hits.length, samples: a5Hits.slice(0, 5) }
153
+
154
+ const a4run = maxRunAgainst(newFiles, oldFiles)
155
+ const a4 = { ok: a4run.maxRun < THRESHOLDS.A4_MAX_RUN, maxRun: a4run.maxRun, where: a4run.where }
156
+
157
+ let a1, a2
158
+ if (isDir(corpusDir)) {
159
+ const corpusFiles = listMarkdown(corpusDir).map((f) => join(corpusDir, f))
160
+ const ratio = maxGramRatio(newFiles, corpusGrams(corpusFiles, THRESHOLDS.GRAM_N))
161
+ a1 = {
162
+ ok: ratio.ratio < THRESHOLDS.A1_MAX_RATIO,
163
+ maxRatio: Number(ratio.ratio.toFixed(6)),
164
+ where: ratio.where,
165
+ note: `corpus=${corpusDir} (${corpusFiles.length} files)`,
166
+ }
167
+ const corpusRun = maxRunAgainst(newFiles, corpusFiles)
168
+ a2 = { ok: corpusRun.maxRun < THRESHOLDS.A2_MAX_RUN, maxRun: corpusRun.maxRun, where: corpusRun.where }
169
+ } else {
170
+ // ⚠️ 语料不可得 ⇒ **null(未验证)**,不是 true(计划 T11 的硬缺陷修正)
171
+ const note = `corpus unavailable: ${TERM_FLAG} not provided ⇒ UNVERIFIED (not a pass)`
172
+ a1 = { ok: null, note }
173
+ a2 = { ok: null, maxRun: null, note }
174
+ }
175
+
176
+ const verdicts = [a1, a2, a3, a4, a5].map((x) => x.ok)
177
+ const unverified = ['a1', 'a2', 'a3', 'a4', 'a5'].filter((_, i) => verdicts[i] === null)
178
+ const ok = verdicts.every((v) => v === true) ? true : verdicts.some((v) => v === false) ? false : null
179
+ return { a1, a2, a3, a4, a5, ok, unverified }
180
+ }
181
+
182
+ const USAGE = [
183
+ `usage: node tools/audit-personas.mjs --new <dir> --old <dir> [${TERM_FLAG} <corpusDir>]`,
184
+ '',
185
+ ' --new directory of NEW persona .md files (the ones under audit)',
186
+ ' --old directory of the OLD persona .md files (A4 reference)',
187
+ ` ${TERM_FLAG} optional corpus directory for A1/A2; omit => a1/a2 ok:null + exit 3 (UNVERIFIED)`,
188
+ '',
189
+ 'exit: 0 all ok:true · 1 any ok:false · 3 any ok:null (unverified) · 2 usage error',
190
+ ].join('\n')
191
+
192
+ /** CLI 入口。返回进程退出码。 */
193
+ export function main(argv = process.argv.slice(2)) {
194
+ let args
195
+ try { args = parseArgs(argv) } catch (e) { console.error(`${e.message}\n\n${USAGE}`); return 2 }
196
+ if (args.help) { console.log(USAGE); return 0 }
197
+ if (!isDir(args.newDir) || !isDir(args.oldDir)) {
198
+ console.error(`--new / --old must be existing directories\n\n${USAGE}`)
199
+ return 2
200
+ }
201
+ if (args.corpusDir !== null && !isDir(args.corpusDir)) {
202
+ console.error(`${TERM_FLAG} is not a directory: ${args.corpusDir}`)
203
+ return 2
204
+ }
205
+
206
+ const result = audit({ newDir: args.newDir, oldDir: args.oldDir, corpusDir: args.corpusDir })
207
+ console.log(JSON.stringify(result, null, 2))
208
+ if (result.ok === false) return 1
209
+ if (result.ok === null) return 3
210
+ return 0
211
+ }
212
+
213
+ if (process.argv[1] !== undefined && pathToFileURL(process.argv[1]).href === import.meta.url) {
214
+ process.exit(main())
215
+ }