dsh-fishpai 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +22 -0
- package/NOTICE.md +46 -0
- package/README.md +169 -0
- package/cordis.patch.yml +10 -0
- package/lib/client.js +2770 -0
- package/package.json +65 -0
- package/plugin/core/diff.mjs +208 -0
- package/plugin/core/markdown.mjs +450 -0
- package/plugin/core/notes.mjs +151 -0
- package/plugin/core/patch.mjs +93 -0
- package/plugin/core/render.mjs +906 -0
- package/plugin/core/runtime.mjs +83 -0
- package/plugin/core/theme-info.mjs +80 -0
- package/plugin/core/theme-spec.mjs +276 -0
- package/plugin/host/assets.mjs +120 -0
- package/plugin/host/custom-theme.mjs +121 -0
- package/plugin/host/routes.mjs +494 -0
- package/plugin/host/store.mjs +719 -0
- package/plugin/host/tools.mjs +573 -0
- package/plugin/index.mjs +127 -0
- package/plugin/vendor/highlight.min.js +1213 -0
- package/plugin/vendor/hljs-map.json +202 -0
- package/plugin/vendor/markdown-it.min.js +2 -0
- package/plugin/vendor/themes.js +414 -0
- package/skills/fishpai/SKILL.md +146 -0
|
@@ -0,0 +1,906 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 渲染主流程:Markdown → 可直接粘贴进微信公众号编辑器的内联样式 HTML。
|
|
3
|
+
*
|
|
4
|
+
* 这是 `mopai/render.js`(迁移前,与墨排线上站点逐条对齐过)的 ESM 移植。
|
|
5
|
+
* **默认参数下,输出与迁移前的实现逐字节相同**,由 `test/golden/**` 守着
|
|
6
|
+
* (golden 由 `test/oracle/render.cjs` 生成,那是移植前的冻结实现)。
|
|
7
|
+
*
|
|
8
|
+
* 相对移植前新增的能力,都**由调用方显式启用**,默认路径一个字节都不变
|
|
9
|
+
* (`test/golden/**` 守的就是"都不生效"这条默认路径):
|
|
10
|
+
* - `annotate: true` 在每个顶层块前插一个不可见的 `<fp-block data-b="ID">` 锚点(只给预览用)
|
|
11
|
+
* - `imageResolver` 把图片 src 换成 data URI(宿主实现读文件;只给复制/导出用)
|
|
12
|
+
* - `wrapText: true` 把文字包进 `<span>`(微信结构校验的兼容层;只给复制/导出用)
|
|
13
|
+
* - `wechatBackground: true` 把 `background` 简写拆成微信肯保留的长写(同上)
|
|
14
|
+
*/
|
|
15
|
+
import { resolveTheme, themes, FONT_MAP, hljsMap, read } from './runtime.mjs'
|
|
16
|
+
import { createMd, makeStyler, collectBlocks } from './markdown.mjs'
|
|
17
|
+
import { classifyTheme } from './theme-info.mjs'
|
|
18
|
+
|
|
19
|
+
// ── 1. 微信脚注:正文外链转上标 + 文末「参考资料」────────────────
|
|
20
|
+
// 站点做法:把 <a> 换成 <span>,沿用该 <a> 已经拿到的主题链接样式(含 border-bottom),
|
|
21
|
+
// 再补一个 cursor: default 的 <sup>;文末追加纯文本「参考资料」(微信不支持正文外链)。
|
|
22
|
+
const A_TAG_RE = /<a\s([^>]*?)href="([^"]*)"([^>]*)>([\s\S]*?)<\/a>/g
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* HTML 实体解码(数字 / 十六进制 / 命名三种写法)。
|
|
26
|
+
*
|
|
27
|
+
* 为什么需要:拼进 HTML 的 URL 在**解析时**才被浏览器解码,而"这是不是一个可执行协议"
|
|
28
|
+
* 必须按解码后的样子判——`javascript:`、`java	script:` 解码后就是 `javascript:`。
|
|
29
|
+
* 数字那趟只跑一次(`&#97;` 单趟解码成 `a`,与浏览器一致,不接力解码)。
|
|
30
|
+
*/
|
|
31
|
+
const NAMED_ENTITIES = {
|
|
32
|
+
amp: '&',
|
|
33
|
+
lt: '<',
|
|
34
|
+
gt: '>',
|
|
35
|
+
quot: '"',
|
|
36
|
+
apos: "'",
|
|
37
|
+
colon: ':',
|
|
38
|
+
sol: '/',
|
|
39
|
+
Tab: '\t',
|
|
40
|
+
NewLine: '\n',
|
|
41
|
+
nbsp: '\u00a0',
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function decodeEntities(text) {
|
|
45
|
+
return String(text)
|
|
46
|
+
.replace(/&#[xX]([0-9a-fA-F]+);?/g, (full, hex) => codePointText(parseInt(hex, 16)))
|
|
47
|
+
.replace(/&#(\d+);?/g, (full, dec) => codePointText(parseInt(dec, 10)))
|
|
48
|
+
.replace(/&([a-zA-Z]+);/g, (full, name) => (name in NAMED_ENTITIES ? NAMED_ENTITIES[name] : full))
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function codePointText(code) {
|
|
52
|
+
return Number.isFinite(code) && code >= 0 && code <= 0x10ffff ? String.fromCodePoint(code) : '\ufffd'
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* 归一化一个 URL:实体解码 → 抹掉空白与控制字符 → 转小写。
|
|
57
|
+
*
|
|
58
|
+
* 顺序不能反:抹空白放在解码之后,`	` 这种"解码出来才是空白"的写法才会被吃掉。
|
|
59
|
+
* 空白与控制字符一律删掉(而不是 trim):浏览器解析 URL 时也会丢掉它们。
|
|
60
|
+
*/
|
|
61
|
+
const IGNORABLE_CHARS_RE = /[\u0000-\u0020\u007f-\u00a0\u1680\u2000-\u200f\u2028\u2029\u202f\u205f\u3000\ufeff]+/g
|
|
62
|
+
|
|
63
|
+
function squashUrl(text) {
|
|
64
|
+
return decodeEntities(text).replace(IGNORABLE_CHARS_RE, '').toLowerCase()
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** 可执行协议:写进 `href`/`src` 就等于在打开产物的那一刻执行。 */
|
|
68
|
+
const EXECUTABLE_SCHEME_RE = /^(?:javascript|vbscript):/
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* 请求里的字号是否合法。
|
|
72
|
+
*
|
|
73
|
+
* 为什么渲染层自己也要判:`fontSize` 是**原样插值进 `style="…"`** 的请求字段
|
|
74
|
+
* (`font-size: ${size};`),一个引号就能闭合属性、把后面的字节变成任意属性;而 `render()`
|
|
75
|
+
* 是公开导出(本仓 CLI、宿主工具都直接调它),不能指望每个调用方都记得先校验。
|
|
76
|
+
* 形态只认 `NNpx`;区间 12–24px:面板给的是 14–18px,这个范围足够宽到不误伤手写值,
|
|
77
|
+
* 又挡住 `9999px` 这类把版面撑烂的写法。不合法一律**回落 16px**(与缺省同一条路)。
|
|
78
|
+
*/
|
|
79
|
+
const FONT_SIZE_RE = /^\d{1,2}(?:\.\d+)?px$/
|
|
80
|
+
const FONT_SIZE_MIN = 12
|
|
81
|
+
const FONT_SIZE_MAX = 24
|
|
82
|
+
|
|
83
|
+
export function isFontSize(value) {
|
|
84
|
+
if (typeof value !== 'string' || !FONT_SIZE_RE.test(value)) return false
|
|
85
|
+
const n = Number.parseFloat(value)
|
|
86
|
+
return n >= FONT_SIZE_MIN && n <= FONT_SIZE_MAX
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* 哪些 `<a>` 会被转成脚注(同一条规则必须被"渲染"与"数给面板看"两处共用)。
|
|
91
|
+
*
|
|
92
|
+
* 为什么要单独数一遍:面板的「脚注」开关以前靠**猜**正文里有没有外链(正则只认带 `//` 的写法),
|
|
93
|
+
* 于是 `github.com/foo/bar` 这种能被 linkify 变成链接、也会生成「参考资料」的写法,
|
|
94
|
+
* 开关却是灰的——用户点了没反应,也关不掉。现在由渲染结果说话。
|
|
95
|
+
*
|
|
96
|
+
* 判定前先归一化(实体解码 + 抹空白 + 转小写),再按**白名单**判:只有 http / https / mailto
|
|
97
|
+
* 算外链。`JavaScript:`、`java\tscript:`、`javascript:` 这类写法都能骗过朴素的前缀比较,
|
|
98
|
+
* 而它们进了「参考资料」就是一条能被点开的可执行链接(导出成 .html 双击即中招);
|
|
99
|
+
* 白名单把 `data:` / `file:` 这些同样不属于"正文外链"的协议一并挡在外面。
|
|
100
|
+
*
|
|
101
|
+
* `#` 是页内锚点:原判据就不把它当外链(它也不该生成「参考资料」),这里保持原语义。
|
|
102
|
+
*
|
|
103
|
+
* **取舍**:白名单是"只认这几种协议",不是"排除危险协议",所以相对链接(`./other.md`)、
|
|
104
|
+
* 根路径(`/abs/path.md`)与 `ftp:` / `tel:` 也不再进「参考资料」——它们会**原样留在产物里**
|
|
105
|
+
* 当普通 `<a>`(`linksToFootnotes` 跳过、`isFootnoteHref` 也把它们排除在计数外)。
|
|
106
|
+
* 这是有意的:粘进公众号的相对链接本来点不开,硬转成文末文本反而把可点的绝对链接和它们混在一起;
|
|
107
|
+
* 要改这条得先想清楚"微信里点不开的链接该长什么样",不要顺手放宽。
|
|
108
|
+
*/
|
|
109
|
+
export function isFootnoteHref(href) {
|
|
110
|
+
const norm = squashUrl(href)
|
|
111
|
+
if (!norm || norm.startsWith('#')) return false
|
|
112
|
+
return /^(?:https?|mailto):\S/.test(norm)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** 正文里会被转成脚注的链接(顺序与编号一致)。 */
|
|
116
|
+
export function footnoteLinks(html) {
|
|
117
|
+
const links = []
|
|
118
|
+
String(html).replace(A_TAG_RE, (full, pre, href, post, inner) => {
|
|
119
|
+
if (isFootnoteHref(href)) links.push({ text: inner.replace(/<[^>]+>/g, ''), href })
|
|
120
|
+
return full
|
|
121
|
+
})
|
|
122
|
+
return links
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
export function linksToFootnotes(html) {
|
|
126
|
+
const links = []
|
|
127
|
+
const supStyle = 'color: inherit; font-size: 80%; vertical-align: super; cursor: default;'
|
|
128
|
+
const out = html.replace(A_TAG_RE, (full, pre, href, post, inner) => {
|
|
129
|
+
if (!isFootnoteHref(href)) return full
|
|
130
|
+
const text = inner.replace(/<[^>]+>/g, '')
|
|
131
|
+
const idx = links.length + 1
|
|
132
|
+
links.push({ idx, text, href })
|
|
133
|
+
// 原 <a> 上的 style 即主题链接样式,站点直接沿用它
|
|
134
|
+
const sm = full.match(/\sstyle="([^"]*)"/)
|
|
135
|
+
const spanStyle = sm ? sm[1] : ''
|
|
136
|
+
return `<span style="${spanStyle}">${inner}<sup style="${supStyle}">[${idx}]</sup></span>`
|
|
137
|
+
})
|
|
138
|
+
if (!links.length) return out
|
|
139
|
+
return (
|
|
140
|
+
out +
|
|
141
|
+
`\n<section style="margin-top: 28px; padding-top: 14px; border-top: 1px solid rgba(0,0,0,0.08); font-size: 13px; line-height: 1.8; color: #999;">\n` +
|
|
142
|
+
`<p style="margin: 0 0 8px; font-weight: 600; color: #666;">参考资料</p>\n` +
|
|
143
|
+
links
|
|
144
|
+
.map(
|
|
145
|
+
(l) =>
|
|
146
|
+
`<p style="margin: 0 0 4px; word-break: break-all;">${l.text && l.text !== l.href ? `[${l.idx}] ${l.text}: ${l.href}` : `[${l.idx}] ${l.href}`}</p>`,
|
|
147
|
+
)
|
|
148
|
+
.join('\n') +
|
|
149
|
+
`\n</section>`
|
|
150
|
+
)
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// ── 2. 清理:去掉公众号不认的属性 ──────────────────────────────
|
|
154
|
+
// 注意:不做全局空白折叠 —— 站点是 DOM 序列化,段落/标题开始的换行会保留成文本节点,
|
|
155
|
+
// 折叠它反而会改变预览观感(列表项还会多出缩进)。
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* 会活动的内容:**解析时无害、打开/插入之后才生效**的那些。
|
|
159
|
+
*
|
|
160
|
+
* 产物经「导出 HTML」落盘,`file://` 双击就是以用户权限跑的一个页面;同一条路径也回退给
|
|
161
|
+
* 浏览器剪贴板(`panel.tsx` 的 `copyRich`)。所以这里剥的不是"微信公众号不认的属性",
|
|
162
|
+
* 而是正文里混进来的可执行内容:
|
|
163
|
+
* - 成对标签整段剥(`<script>` 连脚本体一起);
|
|
164
|
+
* - 没闭合的 `<script>`/`<style>` 剥到文档末尾:它们的内容模型是**原始文本**,
|
|
165
|
+
* 浏览器会把后面的字节全当它们的文本(不是"多包一层元素"),留着等于把正文交给它;
|
|
166
|
+
* - 没闭合的 `<iframe>`/`<object>` **只删标签本身**:它们是普通元素,后面的内容浏览器
|
|
167
|
+
* 照常当标记解析,剥到末尾会把用户后面的正文一起吞掉;
|
|
168
|
+
* - `<link>` / `<meta>` / `<base>` / `<embed>` 是空元素,只删标签(外链样式表、跳转、改基准地址);
|
|
169
|
+
* - `on*` 事件属性与 `srcdoc`:`<img src=x onerror=…>` 这类就是靠它们生效的;
|
|
170
|
+
* - `href`/`src` 上的可执行协议:按**解码后**的形态判(`javascript:` 也算),
|
|
171
|
+
* 值不合法就整个属性去掉——留着半截 `alert(1)` 当 URL 没有意义。
|
|
172
|
+
* 单引号/不带引号的属性值、以及引号没闭合的写法都要认:`html: true` 会把正文里的裸 HTML 原样带进来。
|
|
173
|
+
* 前四条按整段文本处理(它们的前提就是字面出现 `<script>` 这类标签),
|
|
174
|
+
* 属性那两条只在标签内部做,理由见 `cleanTagAttributes`。
|
|
175
|
+
*/
|
|
176
|
+
const DANGEROUS_PAIR_RE = /<(script|style|iframe|object|embed)\b[^>]*>[\s\S]*?<\/\1\s*>/gi
|
|
177
|
+
const RAW_TEXT_OPEN_RE = /<(?:script|style)\b[^>]*>[\s\S]*$/gi
|
|
178
|
+
const RAW_TEXT_CLOSE_RE = /<\/(?:script|style)\s*>/gi
|
|
179
|
+
const CONTAINER_TAG_RE = /<\/?(?:iframe|object)\b[^>]*>/gi
|
|
180
|
+
const VOID_TAG_RE = /<\/?(?:link|meta|base|embed)\b[^>]*>/gi
|
|
181
|
+
// 属性不做"整段正则替换",而是把标签范围**按属性切一遍**(见 `scanAttributes`)。
|
|
182
|
+
// 正则认不出"属性值的边界":`style="color: red" onmouseover="…"` 里第一个 `"` 之后的东西
|
|
183
|
+
// 会被当成新属性(既可能误删正常内容,也可能被引号值后紧接属性名 `src="x"onerror=…` 绕过)。
|
|
184
|
+
const URL_ATTRS = new Set([
|
|
185
|
+
'href',
|
|
186
|
+
'src',
|
|
187
|
+
'xlink:href',
|
|
188
|
+
'srcset',
|
|
189
|
+
'poster',
|
|
190
|
+
'formaction',
|
|
191
|
+
'action',
|
|
192
|
+
'data',
|
|
193
|
+
'background',
|
|
194
|
+
'dynsrc',
|
|
195
|
+
'lowsrc',
|
|
196
|
+
])
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* 这个属性是不是危险属性(事件处理器、srcdoc,或值是执行协议的 URL 属性)。
|
|
200
|
+
*
|
|
201
|
+
* 判定前把属性名**开头的非 ASCII 字母**剥掉:`\onerror=`、`\0onerror=` 这类写法里的 `\`/NUL
|
|
202
|
+
* 不终止属性名(浏览器把它们算进名字里),按原样比 `on` 前缀会漏掉;对合法属性名(一律以
|
|
203
|
+
* 字母开头)这一步是恒等变换,`data-onerror` / `x-onerror` 也不会被误伤。
|
|
204
|
+
*/
|
|
205
|
+
function attributeIsDangerous(name, value) {
|
|
206
|
+
const lower = name.toLowerCase().replace(/^[^a-z]+/, '')
|
|
207
|
+
if (lower.startsWith('on') || lower === 'srcdoc') return true
|
|
208
|
+
if (!URL_ATTRS.has(lower)) return false
|
|
209
|
+
const raw = String(value ?? '').replace(/^["']|["']$/g, '')
|
|
210
|
+
if (lower === 'srcset') {
|
|
211
|
+
return raw.split(',').some((candidate) => EXECUTABLE_SCHEME_RE.test(squashUrl(candidate.trim().split(/\s+/)[0] || '')))
|
|
212
|
+
}
|
|
213
|
+
return EXECUTABLE_SCHEME_RE.test(squashUrl(raw))
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
/**
|
|
217
|
+
* 把 `<` 到范围末尾之间的字节按属性切一遍(返回每个属性的字节范围与值)。
|
|
218
|
+
*
|
|
219
|
+
* 为什么要自己走一遍而不是拿正则扫:属性的边界只能靠"引号里的 `=`/空白不算分隔"来判断,
|
|
220
|
+
* 而正则分不清 `style="color: red" onmouseover="…"` 里那个 `onmouseover` 是属性还是**前一个
|
|
221
|
+
* 属性值里的文字**(把它当属性就会误删正常内容);也认不出引号值后紧接属性名的写法
|
|
222
|
+
* (`src="x"onerror=…`,HTML 分词器允许),那样会漏掉真正的处理器。
|
|
223
|
+
*
|
|
224
|
+
* @returns {Array<{start:number,end:number,name:string,value:string,unterminated:boolean}>}
|
|
225
|
+
* `start`/`end` 是属性名到值末尾的字节范围;`unterminated` 表示它的引号值一直没闭合。
|
|
226
|
+
*/
|
|
227
|
+
function scanAttributes(region) {
|
|
228
|
+
const attrs = []
|
|
229
|
+
let i = 1
|
|
230
|
+
// 跳过标签名(`!`/`/` 开头的那些也当名字处理,反正只用来定位属主的边界)
|
|
231
|
+
while (i < region.length && !/[\s/]/.test(region[i])) i++
|
|
232
|
+
while (i < region.length) {
|
|
233
|
+
while (i < region.length && /[\s/]/.test(region[i])) i++
|
|
234
|
+
if (i >= region.length) break
|
|
235
|
+
const start = i
|
|
236
|
+
while (i < region.length && !/[\s=/]/.test(region[i])) i++
|
|
237
|
+
const name = region.slice(start, i)
|
|
238
|
+
if (!name) {
|
|
239
|
+
i++
|
|
240
|
+
continue
|
|
241
|
+
}
|
|
242
|
+
let j = i
|
|
243
|
+
while (j < region.length && /\s/.test(region[j])) j++
|
|
244
|
+
if (region[j] !== '=') {
|
|
245
|
+
attrs.push({ start, end: i, name, value: '', unterminated: false })
|
|
246
|
+
continue
|
|
247
|
+
}
|
|
248
|
+
j++
|
|
249
|
+
while (j < region.length && /\s/.test(region[j])) j++
|
|
250
|
+
const valueStart = j
|
|
251
|
+
let unterminated = false
|
|
252
|
+
if (region[j] === '"' || region[j] === "'") {
|
|
253
|
+
const quote = region[j]
|
|
254
|
+
j++
|
|
255
|
+
while (j < region.length && region[j] !== quote) j++
|
|
256
|
+
if (j < region.length) j++
|
|
257
|
+
else unterminated = true
|
|
258
|
+
} else {
|
|
259
|
+
while (j < region.length && !/[\s>]/.test(region[j])) j++
|
|
260
|
+
}
|
|
261
|
+
attrs.push({ start, end: j, name, value: region.slice(valueStart, j), unterminated })
|
|
262
|
+
i = j
|
|
263
|
+
}
|
|
264
|
+
return attrs
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* 清洗一个标签范围,回清洗后的文本(`''` = 这个标签整段丢掉)。
|
|
269
|
+
*
|
|
270
|
+
* 值**引号没闭合**时整个标签都丢掉,而不是只剪掉属性:HTML 分词器对这种写法的处理是
|
|
271
|
+
* "这个属性值一路吃到标签结束/EOF",标签名后面的字节全是属性值——剪掉值会留下一个孤零零的
|
|
272
|
+
* 标签名,跟下一个标签粘成一个(`<div` + `<p …>` 会并成一个 div),还会把正常内容卷进属性里。
|
|
273
|
+
* 整段丢掉才是"这个标签从没正常成立过"的等价形态;范围本身收在下一个 `<` 或第一个 `>` 内,
|
|
274
|
+
* 所以丢掉的字节不会越出这个畸形标签。
|
|
275
|
+
*/
|
|
276
|
+
function cleanTagRegion(region) {
|
|
277
|
+
const attrs = scanAttributes(region)
|
|
278
|
+
let out = ''
|
|
279
|
+
let cursor = 0
|
|
280
|
+
let dropped = false
|
|
281
|
+
for (const attr of attrs) {
|
|
282
|
+
if (!attributeIsDangerous(attr.name, attr.value)) continue
|
|
283
|
+
let from = attr.start
|
|
284
|
+
// 连同属性名前的空白一起删(`<img src=x onerror=…>` 洗出来仍是 `<img src=x>`)
|
|
285
|
+
while (from > cursor && /\s/.test(region[from - 1])) from--
|
|
286
|
+
out += region.slice(cursor, from)
|
|
287
|
+
cursor = attr.end
|
|
288
|
+
if (attr.unterminated) dropped = true
|
|
289
|
+
}
|
|
290
|
+
if (dropped) return ''
|
|
291
|
+
return out + region.slice(cursor)
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
/**
|
|
295
|
+
* 这一段是不是真的从标签开头(`<` 后面是标签名或 `/`)。
|
|
296
|
+
*
|
|
297
|
+
* 注释与处理指令(`<!…` / `<?…`)**不算**:里面的东西浏览器一律不求值,
|
|
298
|
+
* 在注释文本里删"属性"只会把注释本身改坏(比如把 `--` 和后面的字节连起来)。
|
|
299
|
+
*/
|
|
300
|
+
const TAG_START_RE = /^<[a-zA-Z/]/
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* 属性清洗只在**标签内部**做。
|
|
304
|
+
*
|
|
305
|
+
* 不能拿属性扫描直接扫整段 HTML:正文里原样写出的 ` onclick="…"` 是**文字**不是属性
|
|
306
|
+
* (`html: true` 放行的裸 HTML 里就会出现),扫进去会把用户的正文字吃掉。
|
|
307
|
+
* 所以先切出标签范围,只对"确实以标签开头"的范围动手——散文里的 `<`(`价格 < 100`)连同
|
|
308
|
+
* 后面的文字原样留下,不参与属性清洗。
|
|
309
|
+
*/
|
|
310
|
+
function cleanTagAttributes(html) {
|
|
311
|
+
let out = ''
|
|
312
|
+
let i = 0
|
|
313
|
+
while (i < html.length) {
|
|
314
|
+
const lt = html.indexOf('<', i)
|
|
315
|
+
if (lt < 0) {
|
|
316
|
+
out += html.slice(i)
|
|
317
|
+
break
|
|
318
|
+
}
|
|
319
|
+
out += html.slice(i, lt)
|
|
320
|
+
const end = tagBodyEnd(html, lt)
|
|
321
|
+
const region = html.slice(lt, end)
|
|
322
|
+
out += TAG_START_RE.test(region) ? cleanTagRegion(region) : region
|
|
323
|
+
i = end
|
|
324
|
+
}
|
|
325
|
+
return out
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
/**
|
|
329
|
+
* 切出一个标签的**标签体**范围(返回结束位置,不含该位置)。
|
|
330
|
+
*
|
|
331
|
+
* 引号感知是必需的:属性值里可以有 `>`(`alt="a>b"`),按"第一个 `>`"切会把标签切断,
|
|
332
|
+
* 断点之后的属性就落到标签外、不再被清洗——而浏览器照样把它们当属性。
|
|
333
|
+
*
|
|
334
|
+
* 窗口边界:正常情况到**第一个不在引号里的 `>`**;引号一直没闭合时收到**下一个 `<`**
|
|
335
|
+
* 为止——那时浏览器也停在"属性值还没结束"的状态,窗口外的字节变不成属性;不在那里收住,
|
|
336
|
+
* 一个畸形引号就会把用户后面的正文整卷进来。`<` 只在引号**开着**时才收窗口:引号已经闭合的
|
|
337
|
+
* `<` 在浏览器那边是畸形标签里的一个字节(会被算进属性名),继续扫到那个 `>` 才能把后面的
|
|
338
|
+
* 属性一起看见、一起清掉。
|
|
339
|
+
*
|
|
340
|
+
* 所以这条保证的是"**窗口内**看得见的危险属性一律清掉";窗口外的字节靠浏览器不把它们解析成
|
|
341
|
+
* 属性兜底(`html: true` 放行裸 HTML 之后的既有语义)。两个已知边界:
|
|
342
|
+
* ① 引号没闭合但**没有**危险属性的标签(`<div class='a>` 这种)原样留下——它不可执行;
|
|
343
|
+
* ② 属性值里含 `>` 时截断点之后的字节不清洗,如 `<img alt=a>b onerror=…>`:那个 `>`
|
|
344
|
+
* 本来就结束了标签,浏览器同样不会把后面的 `onerror` 当成属性。
|
|
345
|
+
*/
|
|
346
|
+
function tagBodyEnd(html, lt) {
|
|
347
|
+
let quote = ''
|
|
348
|
+
let firstGt = -1
|
|
349
|
+
for (let j = lt + 1; j < html.length; j++) {
|
|
350
|
+
const ch = html[j]
|
|
351
|
+
if (ch === '<') {
|
|
352
|
+
if (quote) return firstGt >= 0 ? firstGt : j
|
|
353
|
+
continue
|
|
354
|
+
}
|
|
355
|
+
if (ch === '>') {
|
|
356
|
+
if (!quote) return j
|
|
357
|
+
if (firstGt < 0) firstGt = j
|
|
358
|
+
continue
|
|
359
|
+
}
|
|
360
|
+
if (ch === '"' || ch === "'") {
|
|
361
|
+
if (!quote) quote = ch
|
|
362
|
+
else if (quote === ch) quote = ''
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
return firstGt >= 0 ? firstGt : html.length
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
function stripActiveContent(html) {
|
|
369
|
+
return cleanTagAttributes(
|
|
370
|
+
String(html)
|
|
371
|
+
.replace(DANGEROUS_PAIR_RE, '')
|
|
372
|
+
.replace(RAW_TEXT_OPEN_RE, '')
|
|
373
|
+
.replace(RAW_TEXT_CLOSE_RE, '')
|
|
374
|
+
.replace(CONTAINER_TAG_RE, '')
|
|
375
|
+
.replace(VOID_TAG_RE, ''),
|
|
376
|
+
)
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
export function cleanForWechat(html) {
|
|
380
|
+
return stripActiveContent(html)
|
|
381
|
+
.replace(/\sclass="mermaid-placeholder"/g, '')
|
|
382
|
+
.replace(/\starget="[^"]*"/g, '')
|
|
383
|
+
.replace(/<div >/g, '<div>')
|
|
384
|
+
.replace(/[ \t]+$/gm, '')
|
|
385
|
+
.trim()
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
// ── 3. 「复制到公众号」那层清洗(对应站点 js/publish-utils.js)──
|
|
389
|
+
// 站点点「复制到公众号」时不是照搬预览 HTML,而是走 simplifyHtmlForPlatform:
|
|
390
|
+
// 1) 内联样式按白名单过滤(border-radius / box-shadow / letter-spacing 等会被剥掉)
|
|
391
|
+
// 2) 去掉 class / data-* 属性
|
|
392
|
+
// 3) 代码块扁平化,并把 highlight.js 的 token 颜色内联成 style
|
|
393
|
+
// 4) 图片补 max-width:100%; height:auto
|
|
394
|
+
const STYLE_ALLOWLIST = new Set([
|
|
395
|
+
'background',
|
|
396
|
+
'background-color',
|
|
397
|
+
'border',
|
|
398
|
+
'border-left',
|
|
399
|
+
'border-right',
|
|
400
|
+
'border-top',
|
|
401
|
+
'border-bottom',
|
|
402
|
+
'border-collapse',
|
|
403
|
+
'color',
|
|
404
|
+
'font-family',
|
|
405
|
+
'font-size',
|
|
406
|
+
'font-style',
|
|
407
|
+
'font-weight',
|
|
408
|
+
'height',
|
|
409
|
+
'letter-spacing',
|
|
410
|
+
'line-height',
|
|
411
|
+
'margin',
|
|
412
|
+
'margin-bottom',
|
|
413
|
+
'margin-left',
|
|
414
|
+
'margin-right',
|
|
415
|
+
'margin-top',
|
|
416
|
+
'max-width',
|
|
417
|
+
'padding',
|
|
418
|
+
'padding-bottom',
|
|
419
|
+
'padding-left',
|
|
420
|
+
'padding-right',
|
|
421
|
+
'padding-top',
|
|
422
|
+
'text-align',
|
|
423
|
+
'text-decoration',
|
|
424
|
+
'vertical-align',
|
|
425
|
+
'width',
|
|
426
|
+
])
|
|
427
|
+
|
|
428
|
+
const DEFAULT_PRE_STYLE =
|
|
429
|
+
'margin: 14px 0; padding: 14px 16px; background: #282c34; color: #abb2bf; ' +
|
|
430
|
+
'line-height: 1.7; font-family: SFMono-Regular, Consolas, monospace; font-size: 13px;'
|
|
431
|
+
|
|
432
|
+
export function filterInlineStyles(styleText) {
|
|
433
|
+
return String(styleText || '')
|
|
434
|
+
.split(';')
|
|
435
|
+
.map((r) => r.trim())
|
|
436
|
+
.filter(Boolean)
|
|
437
|
+
.filter((rule) => STYLE_ALLOWLIST.has((rule.split(':')[0] || '').trim().toLowerCase()))
|
|
438
|
+
.join('; ')
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
export function appendInlineStyle(styleText, extra) {
|
|
442
|
+
const cur = String(styleText || '')
|
|
443
|
+
const sep = cur && !cur.trim().endsWith(';') ? '; ' : ''
|
|
444
|
+
return `${cur}${sep}${extra}`
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
function styleFrom(m) {
|
|
448
|
+
if (!m) return null
|
|
449
|
+
let s = ''
|
|
450
|
+
if (m.color) s += `color: ${m.color};`
|
|
451
|
+
if (m.fontWeight && m.fontWeight !== '400' && m.fontWeight !== 'normal') s += ` font-weight: ${m.fontWeight};`
|
|
452
|
+
if (m.fontStyle && m.fontStyle !== 'normal') s += ` font-style: ${m.fontStyle};`
|
|
453
|
+
return s.trim() || null
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
/** 给 highlight.js 的 <span class="hljs-xxx"> 内联颜色
|
|
457
|
+
* (hljs-map.json 是从真实浏览器导出的计算样式,已处理 CSS 优先级) */
|
|
458
|
+
export function hljsStyleFor(cls) {
|
|
459
|
+
const map = hljsMap()
|
|
460
|
+
if (!cls) return null
|
|
461
|
+
// 1) 优先按完整 class 串查(覆盖 "hljs-title function_" 这类复合写法)
|
|
462
|
+
if (map[cls]) return styleFrom(map[cls])
|
|
463
|
+
// 2) 再按 class 出现顺序逐段查,取第一个有规则的
|
|
464
|
+
for (const n of cls.split(/\s+/).filter(Boolean)) {
|
|
465
|
+
if (map[n]) return styleFrom(map[n])
|
|
466
|
+
}
|
|
467
|
+
return null
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
/**
|
|
471
|
+
* 同一份色表的 CSS 形式:预览 iframe(沙箱 srcdoc)里没有 highlight.js 的样式表,
|
|
472
|
+
* 代码块就只剩 class、没有颜色——面板预览看起来是「黑的」,而复制/导出的成品是彩色的。
|
|
473
|
+
* 宿主把它随预览响应下发给客户端注入,**只给预览用**:发布产物走 inlineCodeStyles 内联,
|
|
474
|
+
* 不吃这份 CSS(golden 不受影响)。
|
|
475
|
+
*/
|
|
476
|
+
export function hljsPreviewCss() {
|
|
477
|
+
const map = hljsMap()
|
|
478
|
+
const lines = []
|
|
479
|
+
for (const key of Object.keys(map)) {
|
|
480
|
+
// 复合选择器(".hljs-meta .hljs-string")与无前缀的段("function_")在预览里用不上,
|
|
481
|
+
// 只发 `hljs-xxx` 这一类;hljsStyleFor 的兜底逻辑保证复合写法仍能命中单段规则
|
|
482
|
+
if (!/^hljs-[A-Za-z0-9-]*$/.test(key)) continue
|
|
483
|
+
const s = styleFrom(map[key])
|
|
484
|
+
if (s) lines.push(`.${key}{${s}}`)
|
|
485
|
+
}
|
|
486
|
+
return lines.join('\n')
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
/** mac 标题栏圆点/语言标签:站点也会被 inlineCodeTokenStyles 命中,补上计算样式 */
|
|
490
|
+
export function macSpanStyle(cls) {
|
|
491
|
+
if (/\bmac-dot\b/.test(cls)) {
|
|
492
|
+
let bg = null
|
|
493
|
+
let m = cls.match(/\bmac-dot\s+red\b/)
|
|
494
|
+
if (m) {
|
|
495
|
+
bg = 'rgb(255, 95, 87)'
|
|
496
|
+
}
|
|
497
|
+
m = cls.match(/\bmac-dot\s+yellow\b/)
|
|
498
|
+
if (m) {
|
|
499
|
+
bg = 'rgb(254, 188, 46)'
|
|
500
|
+
}
|
|
501
|
+
m = cls.match(/\bmac-dot\s+green\b/)
|
|
502
|
+
if (m) {
|
|
503
|
+
bg = 'rgb(40, 200, 64)'
|
|
504
|
+
}
|
|
505
|
+
const color = 'rgb(171, 178, 191)'
|
|
506
|
+
let s = `color: ${color};`
|
|
507
|
+
if (bg) s += ` background-color: ${bg};`
|
|
508
|
+
return s
|
|
509
|
+
}
|
|
510
|
+
if (/\bmac-code-lang\b/.test(cls)) return 'color: rgb(171, 178, 191);'
|
|
511
|
+
return null
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
function appendStyle(existing, extra) {
|
|
515
|
+
if (!extra) return existing || ''
|
|
516
|
+
const cur = existing || ''
|
|
517
|
+
if (!cur) return extra
|
|
518
|
+
const sep = cur.trim().endsWith(';') ? ' ' : '; '
|
|
519
|
+
return cur + sep + extra
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
/** 代码块扁平化(站点 simplifyMacCodeBlocks):
|
|
523
|
+
* 站点 DOM 里 mac 结构嵌在 pre>code 内,所以 block.querySelector('pre') 命中内层 pre,
|
|
524
|
+
* 压平只是给内层 pre 重打一套固定样式,外层 pre>code 原样保留。 */
|
|
525
|
+
export function simplifyCodeBlocks(html) {
|
|
526
|
+
return html.replace(/<div class="mac-code-block"([^>]*)>([\s\S]*?)<\/div>\s*<\/code><\/pre>/g, (full, macAttrs, inner) => {
|
|
527
|
+
const innerFixed = inner.replace(/<pre[^>]*>/, `<pre style="${DEFAULT_PRE_STYLE}">`)
|
|
528
|
+
return `<div class="mac-code-block"${macAttrs}>${innerFixed}</div></code></pre>`
|
|
529
|
+
})
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
/**
|
|
533
|
+
* 代码 span 样式内联(站点 inlineCodeTokenStyles)。
|
|
534
|
+
*
|
|
535
|
+
* 真实行为(已实测):站点对 `pre code span` 里的**每一个** span 都写内联样式,
|
|
536
|
+
* 不只是 hljs token —— mac 标题栏的圆点 span 同样在内。所以这里对
|
|
537
|
+
* `<pre>…</pre>` 区间内的所有 span 统一处理。
|
|
538
|
+
*/
|
|
539
|
+
export function inlineCodeStyles(html) {
|
|
540
|
+
// 直接在 `<pre>` 区间内部处理 span,不做"先遮罩后还原":
|
|
541
|
+
// 遮罩用的占位符(`\x01N\x01`)万一在正文里原样出现(模型生成的稿子什么都有可能),
|
|
542
|
+
// 还原那一步会把它当成占位符、把正文替换成某个代码块或空串,静默吃掉正文。
|
|
543
|
+
return html.replace(/<pre[\s\S]*?<\/pre>/g, (m) =>
|
|
544
|
+
m.replace(/<span([^>]*)>/g, (full, attrs) => {
|
|
545
|
+
const cm = attrs.match(/\bclass="([^"]*)"/)
|
|
546
|
+
const cls = cm ? cm[1] : ''
|
|
547
|
+
const sm = attrs.match(/\bstyle="([^"]*)"/)
|
|
548
|
+
const existing = sm ? sm[1] : ''
|
|
549
|
+
const extra = macSpanStyle(cls) || hljsStyleFor(cls)
|
|
550
|
+
if (!extra) return full
|
|
551
|
+
const merged = appendStyle(existing, extra)
|
|
552
|
+
const rest = sm ? attrs.replace(/\sstyle="[^"]*"/, '') : attrs
|
|
553
|
+
return `<span${rest} style="${merged}">`
|
|
554
|
+
})
|
|
555
|
+
)
|
|
556
|
+
}
|
|
557
|
+
|
|
558
|
+
/**
|
|
559
|
+
* 「复制到公众号」那层处理(对应 publish-utils.prepareHtml)。
|
|
560
|
+
*
|
|
561
|
+
* 关键:站点「复制到公众号」按钮走的是 copyToClipboard → copyPublishHtmlToClipboard(false),
|
|
562
|
+
* 即 simple=false —— **不做样式白名单过滤**,也不压平 mac 代码块;
|
|
563
|
+
* 只做一件事:把代码块内 span 的颜色用 getComputedStyle 内联进去。
|
|
564
|
+
* 微信编辑器自己会丢弃不认识的内联属性。
|
|
565
|
+
*/
|
|
566
|
+
export function prepareForPublish(html, opts = {}) {
|
|
567
|
+
if (opts.simple) return simplifyForPublish(html)
|
|
568
|
+
return inlineCodeStyles(html)
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
/** 严格简化(站点 simple=true,用于「一键发布到 14 平台」路径):
|
|
572
|
+
* 样式白名单过滤 + 去 class + 代码块扁平化 + 图片补样式。 */
|
|
573
|
+
export function simplifyForPublish(html) {
|
|
574
|
+
html = inlineCodeStyles(html)
|
|
575
|
+
html = simplifyCodeBlocks(html)
|
|
576
|
+
|
|
577
|
+
// 逐标签过滤 style + 去掉 class / data-*
|
|
578
|
+
html = html.replace(/<([a-zA-Z][\w-]*)((?:\s+[^<>]*?)?)(\/?)>/g, (full, tag, body, selfClose) => {
|
|
579
|
+
if (!body || !body.trim()) return full
|
|
580
|
+
let rest = body.replace(/\sclass="[^"]*"/g, '').replace(/\sdata-[a-z-]+="[^"]*"/g, '')
|
|
581
|
+
const sm = rest.match(/\sstyle="([^"]*)"/)
|
|
582
|
+
if (sm) {
|
|
583
|
+
const kept = filterInlineStyles(sm[1])
|
|
584
|
+
rest = rest.replace(/\sstyle="[^"]*"/, kept ? ` style="${kept}"` : '')
|
|
585
|
+
}
|
|
586
|
+
return `<${tag}${rest}${selfClose || ''}>`
|
|
587
|
+
})
|
|
588
|
+
|
|
589
|
+
// 图片补微信兼容样式
|
|
590
|
+
html = html.replace(/<img([^>]*?)>/g, (full, body) => {
|
|
591
|
+
const sm = body.match(/\sstyle="([^"]*)"/)
|
|
592
|
+
const merged = appendInlineStyle(sm ? sm[1] : '', 'max-width: 100%; height: auto;')
|
|
593
|
+
const cleaned = sm ? body.replace(/\sstyle="[^"]*"/, '') : body
|
|
594
|
+
return `<img${cleaned} style="${merged}">`
|
|
595
|
+
})
|
|
596
|
+
|
|
597
|
+
return html
|
|
598
|
+
}
|
|
599
|
+
|
|
600
|
+
// ── 4. 微信列表项兼容(以行内元素开头的条目前面补一个不换行空格)──
|
|
601
|
+
|
|
602
|
+
/**
|
|
603
|
+
* 给"以行内元素开头"的列表项补一个 ` `。
|
|
604
|
+
*
|
|
605
|
+
* 为什么(两轮对照样张 + 真实公众号实测):`1. **块级 diff**:AI 读到的不是全文…` 粘进公众号会变成
|
|
606
|
+
* "块级 diff"一行、`:AI 读到的…`一行——微信把条目**开头**那一段单独起了行(像术语表)。
|
|
607
|
+
* 实测出来的边界:
|
|
608
|
+
* - 条目以**文字**开头(纯文字、普通 span 包着的文字、或前面先有一个 ` `)→ 正常
|
|
609
|
+
* - 加粗出现在**中段** → 正常;**段落**里的加粗开头 → 正常
|
|
610
|
+
* - "把整条 li 包进一个 span" 试过(`37748d4`)→ **照样拆**,已撤(`32ac970`)
|
|
611
|
+
*
|
|
612
|
+
* 所以只补一个 ` ` 就够了,两个好处:
|
|
613
|
+
* - 结构校验那边安全:它的判据是"直接文字子节点的 `textContent.trim()` 非空",
|
|
614
|
+
* 而 `'\u00a0'.trim() === ''`——li 不会因此变成"直接文字 + 子元素"的候选;
|
|
615
|
+
* - 视觉上约 1/4 字的缩进,看不出来。
|
|
616
|
+
*
|
|
617
|
+
* **只补"以行内元素开头"的条目**:块级标签(`<p>`/`<ul>`/`<blockquote>`…)开头的条目一律不动。
|
|
618
|
+
* 松散列表(条目之间有空行)的条目首个子元素正是 `<p>`,补进去的文字节点会落在 `<p>` **外面**、
|
|
619
|
+
* 变成 `<li>` 的直接文字子节点,结构上多余。松散列表在公众号里本来就正常
|
|
620
|
+
* (条目第一行上方不会多出空白行、加粗开头也不会被拆行),所以这一层不必管它。
|
|
621
|
+
*/
|
|
622
|
+
const BLOCK_TAGS = /^(?:p|ul|ol|blockquote|table|pre|h[1-6]|hr|div|dl|dt|dd|figure|section|article)\b/i
|
|
623
|
+
|
|
624
|
+
export function padListItems(html) {
|
|
625
|
+
if (!html || !html.includes('<li')) return html
|
|
626
|
+
// 只补"紧跟 li 开标签就是**行内**元素"的条目;已经是文字开头、或以块级元素开头的都不用动。
|
|
627
|
+
// 用**真的 U+00A0 字符**而不是 ` ` 实体:真 DOM 和"直接文字子节点"的判定都把它当空白,
|
|
628
|
+
// 而实体在源码级扫描(我们自己的结构测试)里会被当成普通文字。
|
|
629
|
+
return html.replace(/(<li\b[^>]*>)(\s*)(?=<([a-zA-Z][^\s/>]*))/g, (full, open, ws, tag) =>
|
|
630
|
+
BLOCK_TAGS.test(tag) ? full : `${open}\u00a0${ws}`,
|
|
631
|
+
)
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
// ── 5. 微信底色兼容(background 简写 → 长写)───────────────────
|
|
635
|
+
/**
|
|
636
|
+
* 微信编辑器的安全过滤是**按属性名**过的:`background-color` 在名单里,`background` 简写不在。
|
|
637
|
+
* 于是主题里 `background: #f0f8f0` 这种写法粘进公众号后整条声明被丢掉——
|
|
638
|
+
* 实测表现:竹林主题的引用块框、表头底色、行内代码底色全没了(面板预览里明明有)。
|
|
639
|
+
*
|
|
640
|
+
* 所以复制/导出时把简写拆成微信肯保留的长写。这是**纯规范化**:一条声明里只有一个值时,
|
|
641
|
+
* `background: <颜色>` 与 `background-color: <颜色>`、`background: <渐变>` 与
|
|
642
|
+
* `background-image: <渐变>` 的渲染结果完全一样(`background-clip: text` 那类主题也不受影响)。
|
|
643
|
+
*/
|
|
644
|
+
export function splitBackgrounds(styleText) {
|
|
645
|
+
if (!styleText || !/\bbackground\s*:/.test(styleText)) return styleText
|
|
646
|
+
return styleText.replace(/(^|;)(\s*)background\s*:\s*([^;]+)/g, (decl, lead, space, value) => {
|
|
647
|
+
const v = value.trim()
|
|
648
|
+
if (!v) return decl
|
|
649
|
+
const prop = /gradient\(|url\(|image\(/i.test(v) ? 'background-image' : 'background-color'
|
|
650
|
+
return `${lead}${space}${prop}: ${v}`
|
|
651
|
+
})
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
/** 同上,作用在整段 HTML 的每个 `style="…"` 上(只给复制/导出那条路用)。 */
|
|
655
|
+
export function splitBackgroundsInHtml(html) {
|
|
656
|
+
if (!html || !/\bbackground\s*:/.test(html)) return html
|
|
657
|
+
return html.replace(/\sstyle="([^"]*)"/g, (full, styleText) => ` style="${splitBackgrounds(styleText)}"`)
|
|
658
|
+
}
|
|
659
|
+
|
|
660
|
+
// ── 5b. 微信字号兼容(把选定的字号推进到文字所在的块级元素上)─────
|
|
661
|
+
/**
|
|
662
|
+
* 字号默认只挂在最外层 wrapper 上,预览里一切正常。但微信编辑器粘进去时会剥掉
|
|
663
|
+
* 最外层 wrapper 的样式——于是正文全退回微信自己的 16px,「字号」控件看着像没用。
|
|
664
|
+
* 表现与「底色简写被丢」是同一类:**外层**的样式靠不住,**元素自己**的样式粘得过去。
|
|
665
|
+
*
|
|
666
|
+
* 把字号补到正文文字所在的 p / li / blockquote 上即可(引用框、表头底色实测都留得住)。
|
|
667
|
+
* 纯叠加:① 没选字号时不加任何字节;② 已经自带字号的元素一律不动——h1–h3 有自己的
|
|
668
|
+
* 字号、表格自己写了 14px(上游里表格字号本来就不跟着正文字号走,跟着改才是 bug)。
|
|
669
|
+
*/
|
|
670
|
+
const INHERITS_SIZE_TAGS = /^(?:p|li|blockquote)$/
|
|
671
|
+
|
|
672
|
+
export function promoteFontSize(html, size) {
|
|
673
|
+
if (!html || !size) return html
|
|
674
|
+
const promote = (text) =>
|
|
675
|
+
text.replace(/<([a-zA-Z][\w-]*)([^>]*?)style="([^"]*)"/g, (full, tag, mid, style) => {
|
|
676
|
+
if (!INHERITS_SIZE_TAGS.test(tag)) return full
|
|
677
|
+
if (/(^|;)\s*font-size\s*:/.test(style)) return full
|
|
678
|
+
return `<${tag}${mid}style="font-size: ${size}; ${style}"`
|
|
679
|
+
})
|
|
680
|
+
// 文末「参考资料」那一节**自己**带着 font-size(13px 小字):它的 `<section>` 粘进微信
|
|
681
|
+
// 后照样在,字号不会因 wrapper 被剥而丢失,所以里面的 p 不需要推进——推进去反而把这节
|
|
682
|
+
// 撑成正文字号。把所有 `<section>` 整段摘出来、只推进其余部分,再按原位缝回去。
|
|
683
|
+
// 遮的是所有 `<section>`:正文里手写的原始 HTML section 也会被一起遮住(极少见),
|
|
684
|
+
// 不推进对它来说是**安全**的退化(不会错加字节)。
|
|
685
|
+
// 摘出来时不用占位符(同 inlineCodeStyles 的理由):正文里若出现同样的控制字符,
|
|
686
|
+
// 占位符会被当成它,把正文替换成别处的内容或空串。
|
|
687
|
+
let out = ''
|
|
688
|
+
let last = 0
|
|
689
|
+
const re = /<section\b[^>]*>[\s\S]*?<\/section>/g
|
|
690
|
+
for (const m of html.matchAll(re)) {
|
|
691
|
+
if (m.index > last) out += promote(html.slice(last, m.index))
|
|
692
|
+
out += m[0]
|
|
693
|
+
last = m.index + m[0].length
|
|
694
|
+
}
|
|
695
|
+
if (last < html.length) out += promote(html.slice(last))
|
|
696
|
+
return out
|
|
697
|
+
}
|
|
698
|
+
|
|
699
|
+
// ── 6. 预览锚点(annotate)─────────────────────────────────────
|
|
700
|
+
|
|
701
|
+
/** 每个顶层块前插一个不可见锚点元素;只给预览的滚动同步与块高亮用,绝不进复制/导出结果。 */
|
|
702
|
+
function renderWithAnchors(md, markdown, env) {
|
|
703
|
+
const tokens = md.parse(markdown, env)
|
|
704
|
+
if (!tokens.length) return { html: '', blocks: [] }
|
|
705
|
+
const blocks = collectBlocks(markdown, tokens)
|
|
706
|
+
const Marker = tokens[0].constructor
|
|
707
|
+
const out = []
|
|
708
|
+
let cursor = 0
|
|
709
|
+
for (const block of blocks) {
|
|
710
|
+
while (cursor < block.tokenIndex) out.push(tokens[cursor++])
|
|
711
|
+
const marker = new Marker('html_block', '', 0)
|
|
712
|
+
marker.content = `<fp-block data-b="${block.id}"></fp-block>`
|
|
713
|
+
out.push(marker)
|
|
714
|
+
}
|
|
715
|
+
while (cursor < tokens.length) out.push(tokens[cursor++])
|
|
716
|
+
return { html: md.renderer.render(out, md.options, env), blocks }
|
|
717
|
+
}
|
|
718
|
+
|
|
719
|
+
// ── 7. 图片内嵌(复制/导出用)──────────────────────────────────
|
|
720
|
+
/** 用宿主给的解析器把 <img src> 换成 data URI;解析器返回 null 表示保持原样。 */
|
|
721
|
+
function rewriteImages(html, resolver) {
|
|
722
|
+
return html.replace(/<img\b([^>]*?)\bsrc="([^"]*)"([^>]*)>/g, (full, pre, src, post) => {
|
|
723
|
+
let dataUri = null
|
|
724
|
+
try {
|
|
725
|
+
dataUri = resolver(src)
|
|
726
|
+
} catch {
|
|
727
|
+
dataUri = null
|
|
728
|
+
}
|
|
729
|
+
if (!dataUri) return full
|
|
730
|
+
return `<img${pre}src="${dataUri}"${post}>`
|
|
731
|
+
})
|
|
732
|
+
}
|
|
733
|
+
|
|
734
|
+
/**
|
|
735
|
+
* 主渲染。
|
|
736
|
+
*
|
|
737
|
+
* @param {string} markdownText
|
|
738
|
+
* @param {object} [opts]
|
|
739
|
+
* @param {string} [opts.theme] 主题 key 或中文名,默认 default
|
|
740
|
+
* @param {string} [opts.color] 主题色覆盖,如 '#0071e3'
|
|
741
|
+
* @param {string} [opts.font] 字体预设 sans/serif/mono(null 表示不覆盖),默认 sans
|
|
742
|
+
* @param {string} [opts.fontSize] 覆盖 wrapper 字号,如 '17px',默认 16px(非法值一律回落 16px)
|
|
743
|
+
* @param {boolean}[opts.footnotes] 正文外链转脚注,默认 true
|
|
744
|
+
* @param {boolean}[opts.publish] true=按「复制到公众号」的清洗规则输出(默认),false=预览原样
|
|
745
|
+
* @param {boolean}[opts.macCodeBlock] 代码块是否用 mac 标题栏形态,默认 true
|
|
746
|
+
* @param {boolean}[opts.simple] 站点「一键发布 14 平台」那条白名单路径
|
|
747
|
+
* @param {string|object} [opts.theme] 主题 key 或中文名;也可以直接给一个**主题对象**
|
|
748
|
+
* (模型自定义主题,见 `theme-spec.mjs`)——那条路只走数据,不求值任何模型给的代码
|
|
749
|
+
* @param {boolean}[opts.annotate] 预览锚点(隐含 publish=false),默认 false
|
|
750
|
+
* @param {boolean}[opts.wrapText] 把文字包进 `<span>`(微信结构校验的兼容层),默认 false
|
|
751
|
+
* @param {boolean}[opts.wechatBackground] 把 `background` 简写拆成微信肯保留的长写,默认 false
|
|
752
|
+
* @param {(src:string)=>string|null} [opts.imageResolver] 图片内嵌解析器
|
|
753
|
+
* @returns {{html: string, themeKey: string, themeName: string, blocks?: Array<object>}}
|
|
754
|
+
*/
|
|
755
|
+
export function render(markdownText, opts = {}) {
|
|
756
|
+
const THEMES = themes()
|
|
757
|
+
// 主题既可以是 key/中文名(内置),也可以直接给一个对象(模型自定义主题)。
|
|
758
|
+
// 走对象这条路时不查内置表,所以自定义主题不会影响内置主题的任何字节(golden 不受影响)。
|
|
759
|
+
const custom = opts.theme && typeof opts.theme === 'object' ? opts.theme : null
|
|
760
|
+
const key = custom ? String(custom.key || 'custom') : resolveTheme(opts.theme)
|
|
761
|
+
if (!custom && !key) throw new Error(`未知主题: ${opts.theme}(可用 GET /fishpai/api/themes 查看,或在面板里选)`)
|
|
762
|
+
const theme = custom || THEMES[key]
|
|
763
|
+
const styles = makeStyler(theme, opts.color)
|
|
764
|
+
const md = createMd(styles, { macCodeBlock: opts.macCodeBlock, wrapText: opts.wrapText })
|
|
765
|
+
|
|
766
|
+
const env = {}
|
|
767
|
+
let body
|
|
768
|
+
let blocks
|
|
769
|
+
if (opts.annotate) {
|
|
770
|
+
const rendered = renderWithAnchors(md, markdownText, env)
|
|
771
|
+
body = rendered.html
|
|
772
|
+
blocks = rendered.blocks
|
|
773
|
+
} else {
|
|
774
|
+
body = md.render(markdownText, env)
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
// 面板要靠"渲染结果"而不是猜正文来决定两个开关能不能点:
|
|
778
|
+
// 脚注开关 —— 正文里到底有没有会被转成脚注的链接(`github.com/x` 这种没写协议的也算)
|
|
779
|
+
// Mac 代码框开关 —— 正文里到底有没有代码块(缩进式代码块也算,只认 ``` 会漏)
|
|
780
|
+
// 数链接**与开关无关**:若关掉「脚注」就把 linkCount 记成 0,面板会据此把开关置灰,
|
|
781
|
+
// 于是关一次就再也打不开(自锁),提示还写着"正文里没有外链"而正文其实有。
|
|
782
|
+
// 开关只决定要不要真的转成脚注,不决定"正文里有没有链接"。
|
|
783
|
+
const linkCount = footnoteLinks(body).length
|
|
784
|
+
const hasCode = /<pre\b/.test(body)
|
|
785
|
+
if (opts.footnotes !== false) body = linksToFootnotes(body)
|
|
786
|
+
body = cleanForWechat(body)
|
|
787
|
+
if (typeof opts.imageResolver === 'function') body = rewriteImages(body, opts.imageResolver)
|
|
788
|
+
|
|
789
|
+
let wrapperStyle = styles.wrapper || ''
|
|
790
|
+
// 站点行为:字体预设与字号总会覆盖主题自带的 font-family / font-size。
|
|
791
|
+
// `opts.font` **不能注入**:它只当 `FONT_MAP` 的 key 查表,查不到就一个字节都不替换
|
|
792
|
+
// (写进来的永远是表里的常量)。`opts.fontSize` 不一样——它是**原样插值**进 style 的,
|
|
793
|
+
// 所以必须过形态校验;`render()` 是公开导出(本仓 CLI 也在用),不能指望调用方先校验。
|
|
794
|
+
const font = opts.font === null ? null : opts.font || 'sans'
|
|
795
|
+
if (font && FONT_MAP[font]) {
|
|
796
|
+
wrapperStyle = wrapperStyle.replace(/font-family:[^;]+;/, `font-family: ${FONT_MAP[font]};`)
|
|
797
|
+
}
|
|
798
|
+
const requestedSize = isFontSize(opts.fontSize) ? opts.fontSize : null
|
|
799
|
+
const size = requestedSize || '16px'
|
|
800
|
+
wrapperStyle = wrapperStyle.replace(/font-size:\s*\d+px;/, `font-size: ${size};`)
|
|
801
|
+
// 站点会先把双引号换成单引号,避免破坏 style 属性。
|
|
802
|
+
// 只中和 wrapper 是**复刻站点行为**,前提是样式值的四条入口都收得住:
|
|
803
|
+
// 1) `store.mjs` 的 `applyMeta`:`color` / `fontSize` 过形态校验(`saveDoc` 与 `updateMeta` 共用);
|
|
804
|
+
// 2) `routes.mjs` 的 `/render`:`body.meta` 能盖过状态,同样两个字段在校验后才下传;
|
|
805
|
+
// 3) `theme-spec.mjs` 的 `FORBIDDEN_RE`:自定义主题的样式值不许带双引号;
|
|
806
|
+
// 4) 本文件的 `isFontSize`:`render()` 是公开导出(CLI 也直接调),自己再兜一道。
|
|
807
|
+
// 所以这里不必也不该扩大范围——内置 `nyt` 主题合法地带着双引号 font-family,
|
|
808
|
+
// 任何"统一转义"都会改掉 golden 字节(不变量 1)。
|
|
809
|
+
wrapperStyle = wrapperStyle.replace(/"/g, "'")
|
|
810
|
+
|
|
811
|
+
if (opts.publish !== false && !opts.annotate) {
|
|
812
|
+
body = prepareForPublish(body, { simple: !!opts.simple })
|
|
813
|
+
// 微信会剥掉最外层 wrapper 的样式,字号只挂在 wrapper 上等于没设
|
|
814
|
+
// (粘进去正文退回 16px,预览里却好好的)。推进到文字所在的块级元素上。
|
|
815
|
+
if (opts.promoteFontSize && requestedSize) body = promoteFontSize(body, size)
|
|
816
|
+
}
|
|
817
|
+
// wrapText 的第二半:列表项前面补一个不换行空格(微信会把条目开头的行内片段单独起一行)
|
|
818
|
+
if (opts.wrapText) body = padListItems(body)
|
|
819
|
+
// 微信底色兼容:放在最后(代码块 span 的内联色、图片补的样式都已就位),正文与 wrapper 一起规范化
|
|
820
|
+
if (opts.wechatBackground) {
|
|
821
|
+
body = splitBackgroundsInHtml(body)
|
|
822
|
+
wrapperStyle = splitBackgrounds(wrapperStyle)
|
|
823
|
+
}
|
|
824
|
+
|
|
825
|
+
const html = `<div style="${wrapperStyle}">${body}</div>`
|
|
826
|
+
// linkCount / hasCode 只回给面板判断"开关该不该亮",不参与 golden(html 一个字节都没动)
|
|
827
|
+
const out = { html, themeKey: key, themeName: theme.name, linkCount, hasCode }
|
|
828
|
+
if (blocks) out.blocks = blocks
|
|
829
|
+
return out
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
/**
|
|
833
|
+
* 文档状态里存着的主题 key 可能已经不存在了(上游主题被移除、或状态是手改的)。
|
|
834
|
+
* 这种时候退回默认主题——**不要让一篇好好的文档因为一个主题名字打不开**。
|
|
835
|
+
* (`render()` 自己仍然对未知主题抛错:那是"调用方写错了",必须当场说出来。)
|
|
836
|
+
*/
|
|
837
|
+
export function safeThemeKey(name) {
|
|
838
|
+
return resolveTheme(name) || 'default'
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
/**
|
|
842
|
+
* 主题清单(面板、`GET /themes` 与技能共用)。
|
|
843
|
+
* 除名称之外还带上 `classifyTheme()` 算出来的能力标注:面板据此决定
|
|
844
|
+
* 「主题色该不该显示」「要不要提醒这个主题粘进微信有风险」。
|
|
845
|
+
*/
|
|
846
|
+
export function themeCatalog() {
|
|
847
|
+
const THEMES = themes()
|
|
848
|
+
return Object.entries(THEMES).map(([key, t]) => ({
|
|
849
|
+
key,
|
|
850
|
+
name: t.name,
|
|
851
|
+
emoji: t.emoji || '',
|
|
852
|
+
desc: t.desc || '',
|
|
853
|
+
...classifyTheme(t),
|
|
854
|
+
}))
|
|
855
|
+
}
|
|
856
|
+
|
|
857
|
+
// ── 8. CLI(本地排查用;插件本身走 tools/routes)────────────────
|
|
858
|
+
function main(argv) {
|
|
859
|
+
const args = argv.slice(2)
|
|
860
|
+
if (args.includes('--list')) {
|
|
861
|
+
console.log('可用主题(key / 名称):')
|
|
862
|
+
for (const t of themeCatalog()) console.log(` ${t.key.padEnd(11)} ${t.emoji} ${t.name} — ${t.desc}`)
|
|
863
|
+
return 0
|
|
864
|
+
}
|
|
865
|
+
let input = null
|
|
866
|
+
let theme = 'default'
|
|
867
|
+
let out = null
|
|
868
|
+
let color = null
|
|
869
|
+
let fontSize = null
|
|
870
|
+
let publish = true
|
|
871
|
+
let macCodeBlock = true
|
|
872
|
+
let footnotes = true
|
|
873
|
+
for (let i = 0; i < args.length; i++) {
|
|
874
|
+
const a = args[i]
|
|
875
|
+
if (a === '-t' || a === '--theme') theme = args[++i]
|
|
876
|
+
else if (a === '-o' || a === '--out') out = args[++i]
|
|
877
|
+
else if (a === '-c' || a === '--color') color = args[++i]
|
|
878
|
+
else if (a === '-s' || a === '--size') fontSize = args[++i]
|
|
879
|
+
else if (a === '--no-footnotes') footnotes = false
|
|
880
|
+
else if (a === '--no-mac') macCodeBlock = false
|
|
881
|
+
else if (a === '--preview') publish = false
|
|
882
|
+
else if (a === '-' || !a.startsWith('-')) input = a
|
|
883
|
+
}
|
|
884
|
+
if (!input) {
|
|
885
|
+
console.error('用法: node plugin/core/render.mjs <input.md> [-t 主题] [-o out.html] [-c 色值] [-s 字号] [--list] [--preview] [--no-mac] [--no-footnotes]')
|
|
886
|
+
return 2
|
|
887
|
+
}
|
|
888
|
+
// 动态 import 免得 CLI 之外的调用方也被 node:fs 牵连
|
|
889
|
+
return import('node:fs').then((fs) => {
|
|
890
|
+
const text = input === '-' ? fs.readFileSync(0, 'utf8') : fs.readFileSync(input, 'utf8')
|
|
891
|
+
const { html, themeName } = render(text, { theme, color, fontSize, publish, macCodeBlock, footnotes })
|
|
892
|
+
if (out) {
|
|
893
|
+
fs.writeFileSync(out, html, 'utf8')
|
|
894
|
+
console.error(`[ok] 主题「${themeName}」 -> ${out} (${html.length} 字符)`)
|
|
895
|
+
} else {
|
|
896
|
+
process.stdout.write(html)
|
|
897
|
+
}
|
|
898
|
+
return 0
|
|
899
|
+
})
|
|
900
|
+
}
|
|
901
|
+
|
|
902
|
+
if (process.argv[1] && import.meta.url === new URL(`file://${process.argv[1].replace(/\\/g, '/')}`).href) {
|
|
903
|
+
main(process.argv).then((code) => process.exit(code))
|
|
904
|
+
}
|
|
905
|
+
|
|
906
|
+
export { read }
|