@geoly-ai/skills-hub 0.3.6 → 0.3.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +6 -2
- package/scripts/submission/promotion-file.mjs +246 -0
- package/scripts/submission/run-gates.mjs +250 -0
- package/scripts/submission/scan-text.mjs +363 -0
- package/scripts/submission/structural-gates.mjs +336 -0
- package/src/cli.mjs +7 -0
- package/src/commands/publish.mjs +400 -0
- package/src/publish/github.mjs +223 -0
- package/src/publish/payload.mjs +241 -0
- package/src/publish/remote.mjs +813 -0
- package/src/publish/token.mjs +244 -0
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// 不可见字符 / bidi / 同形字扫描 —— 06-submission.md §8 第 5 条。
|
|
3
|
+
//
|
|
4
|
+
// §8 的原话:「有没有 Unicode 混淆、零宽字符、双向控制符、同形字?
|
|
5
|
+
// (**审查视图必须高亮不可见字符与 bidi,不能靠肉眼**)」。
|
|
6
|
+
//
|
|
7
|
+
// 🔴 **这一条不能只写进人工清单。** 它点名说了「不能靠肉眼」——
|
|
8
|
+
// 一条要求人去发现零宽字符的清单,等于没有这条。所以做成工具:
|
|
9
|
+
// 能确定是恶意形状的**直接拒**,判不准的**报出来给人看**。
|
|
10
|
+
//
|
|
11
|
+
// ── 拒 vs 报,判据是什么 ────────────────────────────────────────────────
|
|
12
|
+
// **拒**:在 skill 载荷里**没有任何正当用途**、而攻击价值明确的:
|
|
13
|
+
// · bidi 覆盖 / 嵌入 / 隔离控制符(U+202A–202E、U+2066–2069)——
|
|
14
|
+
// Trojan Source(CVE-2021-42574)那一类:**渲染出来的顺序与字节顺序不同**,
|
|
15
|
+
// 人读到的和 agent 读到的可以是两段不同的指令。
|
|
16
|
+
// · 零宽字符(U+200B/200C/200D/2060/FEFF)—— 用来切碎关键词躲过人眼与
|
|
17
|
+
// 字符串匹配,也用来在 `description` 里藏东西。
|
|
18
|
+
// ⚠️ U+200C/200D 在波斯语、印地语等文字里**有正当排版用途**。
|
|
19
|
+
// 这里仍然拒,因为 01-artifacts §4.1 已经把路径限成 ASCII,
|
|
20
|
+
// 而正文用到这两个字符的投稿可以走人工豁免 —— 拒错的代价是一次沟通,
|
|
21
|
+
// 放过的代价是一条看不见的指令。
|
|
22
|
+
//
|
|
23
|
+
// **报**:有正当用途、但值得人看一眼的:
|
|
24
|
+
// · 同一个词里混用拉丁 + 西里尔/希腊字母(`раypal` 那一类同形字);
|
|
25
|
+
// · 其余 Cf 类(格式控制)字符。
|
|
26
|
+
//
|
|
27
|
+
// 🔴 **不执行载荷**(§5):这里只 `readFileSync` + 遍历码点。
|
|
28
|
+
|
|
29
|
+
import { readFileSync, readdirSync, realpathSync, existsSync } from 'node:fs';
|
|
30
|
+
import { fileURLToPath } from 'node:url';
|
|
31
|
+
import { join, relative } from 'node:path';
|
|
32
|
+
|
|
33
|
+
class ScanError extends Error {
|
|
34
|
+
constructor(code, msg) { super(msg); this.name = 'ScanError'; this.code = code; }
|
|
35
|
+
}
|
|
36
|
+
const bad = (code, msg) => { throw new ScanError(code, msg); };
|
|
37
|
+
|
|
38
|
+
/** 直接拒的码点 → 说明。 */
|
|
39
|
+
const FORBIDDEN = new Map([
|
|
40
|
+
[0x202a, 'LRE 从左至右嵌入'], [0x202b, 'RLE 从右至左嵌入'],
|
|
41
|
+
[0x202c, 'PDF 弹出方向格式'], [0x202d, 'LRO 从左至右覆盖'],
|
|
42
|
+
[0x202e, 'RLO 从右至左覆盖'],
|
|
43
|
+
[0x2066, 'LRI 从左至右隔离'], [0x2067, 'RLI 从右至左隔离'],
|
|
44
|
+
[0x2068, 'FSI 首字符强定向隔离'], [0x2069, 'PDI 弹出方向隔离'],
|
|
45
|
+
[0x200b, 'ZWSP 零宽空格'], [0x200c, 'ZWNJ 零宽不连字'],
|
|
46
|
+
[0x200d, 'ZWJ 零宽连字'], [0x2060, 'WJ word joiner'],
|
|
47
|
+
[0xfeff, 'BOM / 零宽不换行空格'],
|
|
48
|
+
// 🔴 NUL 单列。它让 `file(1)` 判整个文件为二进制、`grep -I` 整文件跳过 ——
|
|
49
|
+
// 「这个文件里没有那段话」于是变成一句静默的谎。本仓库两次踩过
|
|
50
|
+
// (src/pack.mjs、test/scan-text.test.mjs 自己)。
|
|
51
|
+
[0x0000, 'NUL —— 会让整个文件被当成二进制而被跳过'],
|
|
52
|
+
]);
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* 只报告的码点:**按 Unicode 属性判**,不是手写一张表。
|
|
56
|
+
*
|
|
57
|
+
* 🔴 手写表必漏(Codex 2026-08-31 点名了一串:标签字符 U+E0020–E007F、
|
|
58
|
+
* U+2061–2064、U+206A–206F、U+180E、U+FFF9–FFFB、变体选择符
|
|
59
|
+
* U+FE00–FE0F 与 U+E0100–E01EF、U+034F CGJ……)。这些都能隐形携带内容、
|
|
60
|
+
* 改变字形或破坏字符串匹配。**枚举打不过属性。**
|
|
61
|
+
*
|
|
62
|
+
* `Default_Ignorable_Code_Point` 正好就是「不该被渲染出来」那一类;
|
|
63
|
+
* `Cf`(格式控制)覆盖其余。两者取并集,减去已经在 FORBIDDEN 里的。
|
|
64
|
+
*/
|
|
65
|
+
const RE_INVISIBLE = /[\p{Default_Ignorable_Code_Point}\p{Cf}]/u;
|
|
66
|
+
// 组合字符:能堆叠、能零宽地插在词中间(U+034F 是典型)
|
|
67
|
+
const RE_MARK = /\p{M}/u;
|
|
68
|
+
|
|
69
|
+
const NAMED = new Map([
|
|
70
|
+
[0x200e, 'LRM 从左至右标记'], [0x200f, 'RLM 从右至左标记'],
|
|
71
|
+
[0x061c, 'ALM 阿拉伯字母标记'], [0x00ad, 'SHY 软连字符'],
|
|
72
|
+
[0x034f, 'CGJ 组合字位连接符'], [0x180e, 'MONGOLIAN VOWEL SEPARATOR'],
|
|
73
|
+
[0xe0001, 'LANGUAGE TAG'],
|
|
74
|
+
]);
|
|
75
|
+
const describe = (cp) => {
|
|
76
|
+
const n = NAMED.get(cp);
|
|
77
|
+
if (n !== undefined) return n;
|
|
78
|
+
if (cp >= 0xfe00 && cp <= 0xfe0f) return '变体选择符 VS1–VS16';
|
|
79
|
+
if (cp >= 0xe0100 && cp <= 0xe01ef) return '变体选择符补充区';
|
|
80
|
+
if (cp >= 0xe0020 && cp <= 0xe007f) return '标签字符(能隐形携带整段文本)';
|
|
81
|
+
return '不可见 / 格式控制字符';
|
|
82
|
+
};
|
|
83
|
+
|
|
84
|
+
// 同形字:只判「拉丁 × 西里尔/希腊」。
|
|
85
|
+
// 🔴 **不判「汉字 × 拉丁」** —— 中文文档里 `SKILL.md 的写法` 到处都是,
|
|
86
|
+
// 把它算成可疑等于让这个工具的输出没人看。
|
|
87
|
+
const SCRIPTS = [
|
|
88
|
+
['latin', /[A-Za-zÀ-ɏ]/],
|
|
89
|
+
['cyrillic', /[Ѐ-ӿԀ-ԯ]/],
|
|
90
|
+
['greek', /[Ͱ-Ͽἀ-]/],
|
|
91
|
+
];
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* 扫一段文本。
|
|
95
|
+
* @returns {{forbidden:Array, suspicious:Array, mixedScript:Array}}
|
|
96
|
+
*/
|
|
97
|
+
export function scanText(text, where = '<text>') {
|
|
98
|
+
const forbidden = [];
|
|
99
|
+
const suspicious = [];
|
|
100
|
+
let line = 1;
|
|
101
|
+
let col = 1;
|
|
102
|
+
for (const ch of text) {
|
|
103
|
+
const cp = ch.codePointAt(0);
|
|
104
|
+
if (ch === '\n') { line++; col = 1; continue; }
|
|
105
|
+
const f = FORBIDDEN.get(cp);
|
|
106
|
+
if (f !== undefined) {
|
|
107
|
+
forbidden.push({ where, line, col, cp, name: f });
|
|
108
|
+
} else if (RE_INVISIBLE.test(ch) || RE_MARK.test(ch)) {
|
|
109
|
+
// 组合字符本身合法(重音、声调),但**零宽**地插在词里能拆开匹配 ——
|
|
110
|
+
// 报出来让人看一眼,不拦。
|
|
111
|
+
if (RE_INVISIBLE.test(ch) || cp === 0x034f) {
|
|
112
|
+
suspicious.push({ where, line, col, cp, name: describe(cp) });
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
col++;
|
|
116
|
+
}
|
|
117
|
+
return {
|
|
118
|
+
forbidden, suspicious,
|
|
119
|
+
mixedScript: findMixedScript(text, where),
|
|
120
|
+
compatibility: findCompatibilityForms(text, where),
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* 同一个「词」里混用拉丁与西里尔/希腊。
|
|
126
|
+
*
|
|
127
|
+
* 🔴 **词里要允许组合字符**(Codex 2026-08-31):用 `\p{L}+` 的话,
|
|
128
|
+
* `p\u034Fа\u034Fypal` 渲染出来是一个词,却被 CGJ 拆成三个**纯脚本**的词,
|
|
129
|
+
* 一处都不报。判据必须和**渲染成一个词**对齐,所以是 `[\p{L}\p{M}]+`。
|
|
130
|
+
*
|
|
131
|
+
* 🔴 **列按码点数**,与 `scanText` 一致。`m.index` 是 UTF-16 码元偏移 ——
|
|
132
|
+
* 前面有 emoji 时两者会报出不同的列,读的人对不上。
|
|
133
|
+
*/
|
|
134
|
+
export function findMixedScript(text, where = '<text>') {
|
|
135
|
+
const out = [];
|
|
136
|
+
let line = 1;
|
|
137
|
+
for (const raw of text.split('\n')) {
|
|
138
|
+
for (const m of raw.matchAll(/[\p{L}\p{M}]+/gu)) {
|
|
139
|
+
const word = m[0];
|
|
140
|
+
const kinds = SCRIPTS.filter(([, re]) => re.test(word)).map(([k]) => k);
|
|
141
|
+
if (kinds.length > 1) {
|
|
142
|
+
out.push({ where, line, col: [...raw.slice(0, m.index)].length + 1, word, scripts: kinds });
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
line++;
|
|
146
|
+
}
|
|
147
|
+
return out;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* 兼容等价形式:原文与 NFKC / 大小写折叠之后**不一样**的词。
|
|
152
|
+
*
|
|
153
|
+
* 🔴 同形字不止「换一个脚本的字母」这一种形状(Codex 2026-08-31):
|
|
154
|
+
* `K`(U+212A KELVIN SIGN)折叠后是 `k`、全角 `shell` NFKC 后是 `shell`、
|
|
155
|
+
* `fi` 展开成 `fi`。这些在 `capability` 之类的关键词上尤其要紧 ——
|
|
156
|
+
* 人眼读到的是 `shell`,字符串匹配读到的不是。
|
|
157
|
+
*
|
|
158
|
+
* ⚠️ 只**报告**:中日韩正文里 NFKC 会动到的字很多,拒会误伤一大片。
|
|
159
|
+
*/
|
|
160
|
+
export function findCompatibilityForms(text, where = '<text>') {
|
|
161
|
+
const out = [];
|
|
162
|
+
let line = 1;
|
|
163
|
+
for (const raw of text.split('\n')) {
|
|
164
|
+
// 🔴 词里要带上 `\p{So}` / `\p{Sk}`:`ⓢⓗⓔⓛⓛ` 属于 So,
|
|
165
|
+
// 用 `[\p{L}\p{M}\p{N}]+` 的话它整个不成词,一处都不报(Codex 2026-08-31)。
|
|
166
|
+
for (const m of raw.matchAll(/[\p{L}\p{M}\p{N}\p{So}\p{Sk}]+/gu)) {
|
|
167
|
+
const word = m[0];
|
|
168
|
+
const nfkc = word.normalize('NFKC');
|
|
169
|
+
// 🔴 **NFKC 的差异要在小写化之前判。** 先 toLowerCase 的话,
|
|
170
|
+
// `K`(U+212A KELVIN)与 `k` 会先变成同一个,差异就没了。
|
|
171
|
+
const folded = nfkc.toLowerCase();
|
|
172
|
+
if (nfkc !== word && /^[\x21-\x7e]+$/.test(folded)) {
|
|
173
|
+
out.push({ where, line, col: [...raw.slice(0, m.index)].length + 1, word, folded });
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
line++;
|
|
177
|
+
}
|
|
178
|
+
return out;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// ── 扫一棵树 ───────────────────────────────────────────────────────────────
|
|
182
|
+
|
|
183
|
+
/**
|
|
184
|
+
* 🔴🔴 **扫所有文件,不按扩展名白名单,也不按 NUL 跳过。**
|
|
185
|
+
*
|
|
186
|
+
* 第一版两条都错,而且**两条各自就是一个绕过**(Codex 2026-08-31):
|
|
187
|
+
* · 白名单只收 `.md/.txt/...` —— 载荷里的 `.sh` / `.py` / `.js` 一个都不扫,
|
|
188
|
+
* 未知扩展名同理。而人要读的恰恰包括那些脚本。
|
|
189
|
+
* · 「含 NUL 当二进制跳过」—— 于是 **`NUL` + `U+202E` 放进一个 `.md`,
|
|
190
|
+
* 整道门直接失效**。用一个「像二进制」的信号去决定要不要检查,
|
|
191
|
+
* 等于把开关交给被检的一方。
|
|
192
|
+
*
|
|
193
|
+
* 现在的判据是**能不能按 UTF-8 严格解码**:
|
|
194
|
+
* · 能 → 它就是给人读的文本,扫(NUL 本身已在 FORBIDDEN 里,会被报出来);
|
|
195
|
+
* · 不能 → 真二进制(图片之类),跳过。
|
|
196
|
+
* 一个刻意构造成合法 UTF-8 的「二进制」文件会被扫,那没有坏处。
|
|
197
|
+
*/
|
|
198
|
+
// 名字看着就是给人读的 —— 这些**必须**是合法 UTF-8,否则按绕过处理。
|
|
199
|
+
const TEXTY = /\.(md|markdown|txt|json|ya?ml|toml|csv|sh|bash|zsh|py|rb|pl|js|mjs|cjs|ts|ps1)$/i;
|
|
200
|
+
|
|
201
|
+
export function scanTree(root) {
|
|
202
|
+
if (!existsSync(root)) bad('E_SCAN_INPUT', `${root} 不存在`);
|
|
203
|
+
// 🔴 `ignoreBOM: true` —— 默认的 TextDecoder 会**吃掉**文件开头的 BOM,
|
|
204
|
+
// 于是 U+FEFF 永远进不了 FORBIDDEN,那一格形同虚设(Codex 2026-08-31)。
|
|
205
|
+
const decoder = new TextDecoder('utf-8', { fatal: true, ignoreBOM: true });
|
|
206
|
+
const all = {
|
|
207
|
+
files: 0, skippedBinary: 0, forbidden: [], suspicious: [], mixedScript: [], compatibility: [],
|
|
208
|
+
};
|
|
209
|
+
const walk = (dir) => {
|
|
210
|
+
// 🔴 `lstat`,不是 `stat` —— `stat` 跟随符号链接,独立调用时能扫出根目录之外
|
|
211
|
+
// (而载荷里本来就不许有 symlink,见 01-artifacts §5)。
|
|
212
|
+
for (const e of readdirSync(dir, { withFileTypes: true }).sort((a, b) => (a.name < b.name ? -1 : 1))) {
|
|
213
|
+
const full = join(dir, e.name);
|
|
214
|
+
if (e.isSymbolicLink()) {
|
|
215
|
+
all.forbidden.push({
|
|
216
|
+
where: relative(root, full), line: 1, col: 1, cp: -1,
|
|
217
|
+
name: '符号链接 —— 载荷里不允许(01-artifacts §5),且会让扫描越出根目录',
|
|
218
|
+
});
|
|
219
|
+
continue;
|
|
220
|
+
}
|
|
221
|
+
if (e.isDirectory()) { walk(full); continue; }
|
|
222
|
+
if (!e.isFile()) continue;
|
|
223
|
+
const rel = relative(root, full);
|
|
224
|
+
let text;
|
|
225
|
+
try {
|
|
226
|
+
text = decoder.decode(readFileSync(full));
|
|
227
|
+
} catch {
|
|
228
|
+
// 🔴🔴 **「不是合法 UTF-8 就跳过」本身又是一个绕过**(Codex 2026-08-31):
|
|
229
|
+
// 在一个塞了 RLO 的 .md 里多放一个非法字节,整文件就不扫了。
|
|
230
|
+
// 所以按文件名分两路:
|
|
231
|
+
// · 名字就是给人读的(.md / .sh / .json…)→ **直接拒**。
|
|
232
|
+
// 一个不是合法 UTF-8 的 SKILL.md 不是「二进制资源」,是规避。
|
|
233
|
+
// · 其余(图片之类)→ 跳过,但**计数并报出来**,不静默。
|
|
234
|
+
if (TEXTY.test(e.name)) {
|
|
235
|
+
all.forbidden.push({
|
|
236
|
+
where: rel, line: 1, col: 1, cp: -1,
|
|
237
|
+
name: '不是合法 UTF-8 —— 文本文件必须能解码,'
|
|
238
|
+
+ '否则「跳过不扫」就成了藏 bidi / 零宽字符的地方',
|
|
239
|
+
});
|
|
240
|
+
} else {
|
|
241
|
+
all.skippedBinary++;
|
|
242
|
+
all.suspicious.push({
|
|
243
|
+
where: rel, line: 1, col: 1, cp: -1,
|
|
244
|
+
name: '非 UTF-8 的二进制文件,**没有**逐字符扫过 —— 请人确认它确实是资源文件',
|
|
245
|
+
});
|
|
246
|
+
}
|
|
247
|
+
continue;
|
|
248
|
+
}
|
|
249
|
+
const r = scanText(text, rel);
|
|
250
|
+
all.files++;
|
|
251
|
+
all.forbidden.push(...r.forbidden);
|
|
252
|
+
all.suspicious.push(...r.suspicious);
|
|
253
|
+
all.mixedScript.push(...r.mixedScript);
|
|
254
|
+
all.compatibility.push(...r.compatibility);
|
|
255
|
+
}
|
|
256
|
+
};
|
|
257
|
+
walk(root);
|
|
258
|
+
return all;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/**
|
|
262
|
+
* 排版。
|
|
263
|
+
*
|
|
264
|
+
* 🔴 **注解格式是 `::error file=…,line=…,col=…::消息`**,不是
|
|
265
|
+
* `::error::file:line:col 消息`(Codex 2026-08-31)。后者 GitHub 认,
|
|
266
|
+
* 但只当成一条**没有位置**的普通注解 —— 它不会挂到文件的那一行上,
|
|
267
|
+
* 而「挂到那一行」正是 §8 第 5 条要的「高亮」。
|
|
268
|
+
* ⚠️ 消息里的 `,` 与 `:` 不需要转义,但**换行要**(`%0A`)。
|
|
269
|
+
*/
|
|
270
|
+
export function format(r, { annotate = false } = {}) {
|
|
271
|
+
const lines = [];
|
|
272
|
+
const esc = (s) => s.replace(/%/g, '%25').replace(/\r/g, '%0D').replace(/\n/g, '%0A');
|
|
273
|
+
const tag = (kind, { where, line, col }, msg) => (annotate
|
|
274
|
+
? `::${kind} file=${where},line=${line},col=${col}::${esc(msg)}`
|
|
275
|
+
: `${kind === 'error' ? '🔴' : '⚠️'} ${where}:${line}:${col} ${msg}`);
|
|
276
|
+
const u = (cp) => `U+${cp.toString(16).toUpperCase().padStart(4, '0')}`;
|
|
277
|
+
|
|
278
|
+
for (const f of r.forbidden) {
|
|
279
|
+
lines.push(tag('error', f, f.cp === -1
|
|
280
|
+
? f.name
|
|
281
|
+
: `有 ${u(f.cp)}(${f.name})—— 载荷里不允许出现,`
|
|
282
|
+
+ '它能让人读到的与 agent 读到的不是同一段文字'));
|
|
283
|
+
}
|
|
284
|
+
for (const s of r.suspicious) {
|
|
285
|
+
lines.push(tag('warning', s, `有 ${u(s.cp)}(${s.name})—— 请人确认`));
|
|
286
|
+
}
|
|
287
|
+
for (const m of r.mixedScript) {
|
|
288
|
+
lines.push(tag('warning', m,
|
|
289
|
+
`「${m.word}」同一个词里混用了 ${m.scripts.join(' + ')} —— 同形字的典型形状`));
|
|
290
|
+
}
|
|
291
|
+
for (const c of r.compatibility ?? []) {
|
|
292
|
+
lines.push(tag('warning', c,
|
|
293
|
+
`「${c.word}」在 NFKC + 大小写折叠之后是「${c.folded}」—— `
|
|
294
|
+
+ '人眼读到的与字符串匹配读到的不是一个东西'));
|
|
295
|
+
}
|
|
296
|
+
return lines;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
// ── CLI ────────────────────────────────────────────────────────────────────
|
|
300
|
+
|
|
301
|
+
const KNOWN = ['submissions', 'annotate'];
|
|
302
|
+
|
|
303
|
+
function parseArgs(argv) {
|
|
304
|
+
const o = {};
|
|
305
|
+
for (let i = 0; i < argv.length; i++) {
|
|
306
|
+
const a = argv[i];
|
|
307
|
+
if (!a.startsWith('--')) bad('E_SCAN_INPUT', `不认得参数 ${a}`);
|
|
308
|
+
const eq = a.indexOf('=');
|
|
309
|
+
const name = eq === -1 ? a : a.slice(0, eq);
|
|
310
|
+
const key = name.slice(2);
|
|
311
|
+
if (!KNOWN.includes(key)) bad('E_SCAN_INPUT', `不认识的选项 ${name}`);
|
|
312
|
+
if (key in o) bad('E_SCAN_INPUT', `${name} 给了不止一次`);
|
|
313
|
+
if (key === 'annotate') { o[key] = 'true'; continue; }
|
|
314
|
+
const val = eq === -1 ? argv[++i] : a.slice(eq + 1);
|
|
315
|
+
if (val === undefined) bad('E_SCAN_INPUT', `${name} 需要一个值`);
|
|
316
|
+
o[key] = val;
|
|
317
|
+
}
|
|
318
|
+
return o;
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
export function main(argv) {
|
|
322
|
+
const o = parseArgs(argv);
|
|
323
|
+
const root = o.submissions ?? 'submissions';
|
|
324
|
+
if (!existsSync(root)) {
|
|
325
|
+
process.stderr.write(`没有待扫的投稿(${root} 不存在)。\n`);
|
|
326
|
+
return 0;
|
|
327
|
+
}
|
|
328
|
+
const r = scanTree(root);
|
|
329
|
+
const annotate = o.annotate === 'true';
|
|
330
|
+
for (const l of format(r, { annotate })) process.stderr.write(`${l}\n`);
|
|
331
|
+
|
|
332
|
+
if (r.forbidden.length) {
|
|
333
|
+
process.stderr.write(
|
|
334
|
+
`\n🔴 ${r.files} 个文本文件里有 ${r.forbidden.length} 处不允许出现的不可见 / bidi 字符。\n`
|
|
335
|
+
+ ' §8 第 5 条要求「审查视图必须高亮不可见字符与 bidi,不能靠肉眼」——\n'
|
|
336
|
+
+ ' 一条要求人去发现零宽字符的清单等于没有这条,所以这里直接拒。\n'
|
|
337
|
+
+ ' ⚠️ 正文确有排版需要(ZWNJ 之类)的,走人工豁免,不要放宽这道门。\n');
|
|
338
|
+
return 1;
|
|
339
|
+
}
|
|
340
|
+
const warn = r.suspicious.length + r.mixedScript.length + r.compatibility.length;
|
|
341
|
+
process.stderr.write(
|
|
342
|
+
`✔ ${r.files} 个文本文件没有 bidi / 零宽字符`
|
|
343
|
+
+ `${r.skippedBinary ? `(另跳过 ${r.skippedBinary} 个非 UTF-8 的二进制文件)` : ''}`
|
|
344
|
+
+ `${warn ? `,另有 ${warn} 处待人确认(见上)` : ''}。\n`);
|
|
345
|
+
return 0;
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
export { ScanError };
|
|
349
|
+
|
|
350
|
+
// 入口守卫比 realpath —— 见 scripts/release/build-timestamp.mjs 里的说明。
|
|
351
|
+
function invokedDirectly() {
|
|
352
|
+
if (process.argv[1] === undefined) return false;
|
|
353
|
+
try {
|
|
354
|
+
return realpathSync(process.argv[1]) === realpathSync(fileURLToPath(import.meta.url));
|
|
355
|
+
} catch { return true; }
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
if (invokedDirectly()) {
|
|
359
|
+
try { process.exit(main(process.argv.slice(2))); } catch (e) {
|
|
360
|
+
process.stderr.write(`${e.code ? `[${e.code}] ` : ''}${e.message}\n`);
|
|
361
|
+
process.exit(1);
|
|
362
|
+
}
|
|
363
|
+
}
|