dsh-fishpai 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +22 -0
- package/NOTICE.md +46 -0
- package/README.md +169 -0
- package/cordis.patch.yml +10 -0
- package/lib/client.js +2770 -0
- package/package.json +65 -0
- package/plugin/core/diff.mjs +208 -0
- package/plugin/core/markdown.mjs +450 -0
- package/plugin/core/notes.mjs +151 -0
- package/plugin/core/patch.mjs +93 -0
- package/plugin/core/render.mjs +906 -0
- package/plugin/core/runtime.mjs +83 -0
- package/plugin/core/theme-info.mjs +80 -0
- package/plugin/core/theme-spec.mjs +276 -0
- package/plugin/host/assets.mjs +120 -0
- package/plugin/host/custom-theme.mjs +121 -0
- package/plugin/host/routes.mjs +494 -0
- package/plugin/host/store.mjs +719 -0
- package/plugin/host/tools.mjs +573 -0
- package/plugin/index.mjs +127 -0
- package/plugin/vendor/highlight.min.js +1213 -0
- package/plugin/vendor/hljs-map.json +202 -0
- package/plugin/vendor/markdown-it.min.js +2 -0
- package/plugin/vendor/themes.js +414 -0
- package/skills/fishpai/SKILL.md +146 -0
|
@@ -0,0 +1,450 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* markdown-it 管线:与墨排站点逐条对齐的渲染规则 + 顶层块切分。
|
|
3
|
+
*
|
|
4
|
+
* 这一层是"上游行为"的承载体:任何一行改动都必须先问"站点是不是也这样"。
|
|
5
|
+
* 站点做法是「先渲染裸 HTML,再用 DOM 扫描给每个标签 setAttribute('style', 原有 + 主题样式)」,
|
|
6
|
+
* 我们是在 markdown-it 的渲染器规则里直接产出等价结果。
|
|
7
|
+
*
|
|
8
|
+
* 已知的 8 个坑(别"修好"它们):
|
|
9
|
+
* 1. 标题 token 类型统一是 heading_open,真实标签在 token.tag;站点把 h4–h6 归到 h3 样式
|
|
10
|
+
* 2. 紧凑列表内的段落 token 带 hidden=true,必须跳过,否则多出未闭合 <p>
|
|
11
|
+
* 3. 代码块嵌套是"bug 变特性":markdown-it 默认 fence 规则会把 highlight 回调的返回值再包一层
|
|
12
|
+
* <pre><code class="language-x">,而回调本身返回 mac 结构 → 真实 DOM 是
|
|
13
|
+
* pre > code > div.mac-code-block > (header + pre > code)
|
|
14
|
+
* 4. <code> 内的 <div> 会被 HTML 解析器丢弃,所以 mac 结构外层容器 div 是裸的、class 丢了
|
|
15
|
+
* 5. 站点默认 fontFamily='sans'、fontSize=16,且**总会覆盖**主题自带的 font-family / font-size
|
|
16
|
+
* 6. wrapper 样式里的双引号会被换成单引号,避免破坏 style 属性
|
|
17
|
+
* 7. themes.js 用顶层 const 声明,不挂 window,Node 里必须 vm 顶层求值(见 runtime.mjs)
|
|
18
|
+
* 8. hljs token 颜色别手写 CSS 优先级——hljs-map.json 是从真实浏览器导出的计算样式
|
|
19
|
+
*/
|
|
20
|
+
import { read } from './runtime.mjs'
|
|
21
|
+
|
|
22
|
+
/** 主题内联样式的占位符替换({{PRIMARY}} / {{PRIMARY_BG}})。 */
|
|
23
|
+
export function makeStyler(theme, customColor) {
|
|
24
|
+
const primary = customColor || '#4f6ef7'
|
|
25
|
+
const primaryBg = primary + '12'
|
|
26
|
+
const resolve = (s) =>
|
|
27
|
+
(s || '')
|
|
28
|
+
.replace(/\{\{PRIMARY\}\}/g, primary)
|
|
29
|
+
.replace(/\{\{PRIMARY_BG\}\}/g, primaryBg)
|
|
30
|
+
.trim()
|
|
31
|
+
const st = theme.styles
|
|
32
|
+
const map = {}
|
|
33
|
+
for (const k of Object.keys(st)) map[k] = resolve(st[k])
|
|
34
|
+
return map
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* 建一个与站点配置一致的 markdown-it 实例,并覆写渲染规则产出内联样式。
|
|
39
|
+
* @param {Record<string,string>} styles 主题样式表(makeStyler 的产物)
|
|
40
|
+
* @param {{macCodeBlock?: boolean, wrapText?: boolean}} [opts] macCodeBlock 默认 true;wrapText 默认 false
|
|
41
|
+
*/
|
|
42
|
+
export function createMd(styles, opts = {}) {
|
|
43
|
+
const macCodeBlock = opts.macCodeBlock !== false
|
|
44
|
+
const markdownit = read('markdownit')
|
|
45
|
+
const md = markdownit({
|
|
46
|
+
html: true,
|
|
47
|
+
breaks: true,
|
|
48
|
+
linkify: true,
|
|
49
|
+
typographer: true,
|
|
50
|
+
highlight(str, lang) {
|
|
51
|
+
const hljs = read('hljs')
|
|
52
|
+
let body = ''
|
|
53
|
+
if (lang && hljs && hljs.getLanguage(lang)) {
|
|
54
|
+
try {
|
|
55
|
+
body = hljs.highlight(str, { language: lang }).value
|
|
56
|
+
} catch {
|
|
57
|
+
body = md.utils.escapeHtml(str)
|
|
58
|
+
}
|
|
59
|
+
} else {
|
|
60
|
+
body = md.utils.escapeHtml(str)
|
|
61
|
+
}
|
|
62
|
+
return `<pre><code>${body}</code></pre>`
|
|
63
|
+
},
|
|
64
|
+
})
|
|
65
|
+
|
|
66
|
+
// 标题:token 类型统一是 heading_open,真实标签在 token.tag 上
|
|
67
|
+
md.renderer.rules.heading_open = (t, i, o, e, slf) => {
|
|
68
|
+
const h = t[i].tag // h1..h6
|
|
69
|
+
const key = ['h1', 'h2', 'h3'].includes(h) ? h : 'h3' // 站点把 h4-h6 归到 h3
|
|
70
|
+
const a = slf.renderAttrs(t[i])
|
|
71
|
+
return `<${h}${styles[key] ? ` style="${styles[key]}"` : ''}${a}>`
|
|
72
|
+
}
|
|
73
|
+
md.renderer.rules.paragraph_open = (t, i, o, e, slf) => {
|
|
74
|
+
if (t[i].hidden) return '' // 紧凑列表内的段落,站点走的也是默认行为:不输出
|
|
75
|
+
const a = slf.renderAttrs(t[i])
|
|
76
|
+
return `<p${styles.p ? ` style="${styles.p}"` : ''}${a}>`
|
|
77
|
+
}
|
|
78
|
+
md.renderer.rules.paragraph_close = (t, i) => (t[i].hidden ? '' : '</p>\n')
|
|
79
|
+
md.renderer.rules.blockquote_open = () => `<blockquote${styles.blockquote ? ` style="${styles.blockquote}"` : ''}>`
|
|
80
|
+
md.renderer.rules.bullet_list_open = () => `<ul${styles.ul ? ` style="${styles.ul}"` : ''}>`
|
|
81
|
+
md.renderer.rules.ordered_list_open = (t, i, o, e, slf) => {
|
|
82
|
+
const a = slf.renderAttrs(t[i])
|
|
83
|
+
return `<ol${styles.ol ? ` style="${styles.ol}"` : ''}${a}>`
|
|
84
|
+
}
|
|
85
|
+
md.renderer.rules.list_item_open = () => `<li${styles.li ? ` style="${styles.li}"` : ''}>`
|
|
86
|
+
md.renderer.rules.hr = (t, i, o, e, slf) => `<hr${styles.hr ? ` style="${styles.hr}"` : ''}${slf.renderAttrs(t[i])}>`
|
|
87
|
+
|
|
88
|
+
// 链接
|
|
89
|
+
md.renderer.rules.link_open = (t, i, o, e, slf) => {
|
|
90
|
+
const a = slf.renderAttrs(t[i])
|
|
91
|
+
return `<a${styles.a ? ` style="${styles.a}"` : ''}${a}>`
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// 行内 code
|
|
95
|
+
md.renderer.rules.code_inline = (t, i) =>
|
|
96
|
+
`<code${styles.code_inline ? ` style="${styles.code_inline}"` : ''}>${md.utils.escapeHtml(t[i].content)}</code>`
|
|
97
|
+
|
|
98
|
+
// 代码块
|
|
99
|
+
md.renderer.rules.fence = (t, i) => {
|
|
100
|
+
const tok = t[i]
|
|
101
|
+
const info = (tok.info || '').trim()
|
|
102
|
+
if (info.toLowerCase() === 'mermaid') {
|
|
103
|
+
return `<div class="mermaid-placeholder">${md.utils.escapeHtml(tok.content)}</div>`
|
|
104
|
+
}
|
|
105
|
+
const s = styles.code_block || ''
|
|
106
|
+
let body
|
|
107
|
+
const hljs = read('hljs')
|
|
108
|
+
if (info && hljs && hljs.getLanguage(info)) {
|
|
109
|
+
try {
|
|
110
|
+
body = hljs.highlight(tok.content, { language: info }).value
|
|
111
|
+
} catch {
|
|
112
|
+
body = md.utils.escapeHtml(tok.content)
|
|
113
|
+
}
|
|
114
|
+
} else {
|
|
115
|
+
body = md.utils.escapeHtml(tok.content)
|
|
116
|
+
}
|
|
117
|
+
// mac 形态:站点默认开启。markdown-it 的默认 fence 规则会把 highlight 回调的返回值
|
|
118
|
+
// 再包一层 <pre><code class="language-xxx">,而 highlight 回调本身返回 mac 结构,
|
|
119
|
+
// 于是真实 DOM 是 pre > code > div.mac-code-block > (header + pre > code)。
|
|
120
|
+
// 外层 pre/code 随后被主题样式覆盖,内层 pre 被 publish-utils 压平。
|
|
121
|
+
if (macCodeBlock) {
|
|
122
|
+
// 围栏后的语言串是**用户可控文本**,而它被插进两个位置:`class="language-…"` 的属性值
|
|
123
|
+
// 与标题栏的文字。` ```js" style="…` 会凭空多出一个 style 属性(浏览器只取第一个,
|
|
124
|
+
// 于是注入的那条生效),所以这里与同文件的 image 规则同口径一律 escapeHtml;
|
|
125
|
+
// 查语言表仍用原始串(转义后的串进不了 hljs.getLanguage)。
|
|
126
|
+
const infoAttr = md.utils.escapeHtml(info)
|
|
127
|
+
const langLabel = md.utils.escapeHtml(info || 'code')
|
|
128
|
+
const bg = (s.match(/background:\s*([^;]+)/) || [])[1] || '#282c34'
|
|
129
|
+
const fg = (s.match(/(?:^|;\s*)color:\s*([^;]+)/) || [])[1] || '#abb2bf'
|
|
130
|
+
return (
|
|
131
|
+
`<pre${s ? ` style="${s}"` : ''}><code${info ? ` class="language-${infoAttr}"` : ''} style="font-family: inherit; font-size: inherit; color: inherit; background: transparent;">` +
|
|
132
|
+
`<div class="mac-code-block" style="border-radius: 10px; overflow: hidden; margin: 14px 0; box-shadow: 0 4px 12px rgba(0,0,0,0.15);">` +
|
|
133
|
+
`<div class="mac-code-header" style="background: ${bg.trim()}; padding: 10px 16px; display: flex; align-items: center; gap: 6px; border-bottom: 1px solid rgba(255,255,255,0.05);">` +
|
|
134
|
+
`<span class="mac-dot red" style="width: 12px; height: 12px; border-radius: 50%; background: #ff5f57; display: inline-block;"></span>` +
|
|
135
|
+
`<span class="mac-dot yellow" style="width: 12px; height: 12px; border-radius: 50%; background: #febc2e; display: inline-block;"></span>` +
|
|
136
|
+
`<span class="mac-dot green" style="width: 12px; height: 12px; border-radius: 50%; background: #28c840; display: inline-block;"></span>` +
|
|
137
|
+
`<span class="mac-code-lang" style="margin-left: auto; font-size: 12px; color: ${fg.trim()}; opacity: 0.5; font-family: -apple-system, sans-serif;">${langLabel}</span>` +
|
|
138
|
+
`</div>` +
|
|
139
|
+
`<pre style="background: ${bg.trim()}; color: ${fg.trim()}; padding: 16px; margin: 0; font-size: 13px; line-height: 1.7; overflow-x: auto; font-family: "SFMono-Regular", Consolas, monospace;">` +
|
|
140
|
+
`<code style="font-family: inherit; font-size: inherit; color: inherit; background: transparent;">${body}</code></pre>` +
|
|
141
|
+
`</div></code></pre>`
|
|
142
|
+
)
|
|
143
|
+
}
|
|
144
|
+
return `<pre${s ? ` style="${s}"` : ''}><code style="font-family: inherit; font-size: inherit; color: inherit; background: transparent;">${body}</code></pre>\n`
|
|
145
|
+
}
|
|
146
|
+
md.renderer.rules.code_block = (t, i) => {
|
|
147
|
+
const s = styles.code_block ? ` style="${styles.code_block}"` : ''
|
|
148
|
+
return `<pre${s}><code style="font-family: inherit; font-size: inherit; color: inherit; background: transparent;">${md.utils.escapeHtml(t[i].content)}</code></pre>\n`
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
// 图片:站点只给 img 打主题样式
|
|
152
|
+
md.renderer.rules.image = (t, i, o, e, slf) => {
|
|
153
|
+
const tok = t[i]
|
|
154
|
+
tok.attrSet('style', styles.img || '')
|
|
155
|
+
const src = tok.attrGet('src') || ''
|
|
156
|
+
const alt = slf.renderInlineAsText(tok.children, o, e)
|
|
157
|
+
const title = tok.attrGet('title')
|
|
158
|
+
const t2 = title ? ` title="${md.utils.escapeHtml(title)}"` : ''
|
|
159
|
+
return `<img src="${md.utils.escapeHtml(src)}" alt="${md.utils.escapeHtml(alt)}"${t2} style="${styles.img || ''}">`
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// 表格
|
|
163
|
+
md.renderer.rules.table_open = () => `<table${styles.table ? ` style="${styles.table}"` : ''}>\n`
|
|
164
|
+
md.renderer.rules.thead_open = () => '<thead>\n'
|
|
165
|
+
md.renderer.rules.tbody_open = () => '<tbody>\n'
|
|
166
|
+
md.renderer.rules.tr_open = () => '<tr>'
|
|
167
|
+
md.renderer.rules.th_open = (t, i, o, e, slf) => {
|
|
168
|
+
const a = slf.renderAttrs(t[i])
|
|
169
|
+
return `<th${styles.th ? ` style="${styles.th}"` : ''}${a}>`
|
|
170
|
+
}
|
|
171
|
+
md.renderer.rules.td_open = (t, i, o, e, slf) => {
|
|
172
|
+
const a = slf.renderAttrs(t[i])
|
|
173
|
+
return `<td${styles.td ? ` style="${styles.td}"` : ''}${a}>`
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// 加粗 / 斜体
|
|
177
|
+
md.renderer.rules.strong_open = () => `<strong${styles.strong ? ` style="${styles.strong}"` : ''}>`
|
|
178
|
+
md.renderer.rules.em_open = () => `<em${styles.em ? ` style="${styles.em}"` : ''}>`
|
|
179
|
+
|
|
180
|
+
// ── 信息卡片 ::: info/warning/tip/danger ──
|
|
181
|
+
const colorMap = {
|
|
182
|
+
info: { bg: '#eef6ff', border: '#3b82f6', icon: 'ℹ️', color: '#1e40af' },
|
|
183
|
+
warning: { bg: '#fff8e6', border: '#f59e0b', icon: '⚠️', color: '#92400e' },
|
|
184
|
+
tip: { bg: '#ecfdf5', border: '#10b981', icon: '💡', color: '#065f46' },
|
|
185
|
+
danger: { bg: '#fef2f2', border: '#ef4444', icon: '🚫', color: '#991b1b' },
|
|
186
|
+
}
|
|
187
|
+
md.core.ruler.after('block', 'info_card', (state) => {
|
|
188
|
+
const tokens = state.tokens
|
|
189
|
+
let i = 0
|
|
190
|
+
while (i < tokens.length) {
|
|
191
|
+
if (tokens[i].type === 'paragraph_open') {
|
|
192
|
+
const nx = tokens[i + 1]
|
|
193
|
+
if (nx && nx.type === 'inline' && nx.content) {
|
|
194
|
+
const m = nx.content.match(/^:::(info|warning|tip|danger)\s*(.*?)(?:\n|$)([\s\S]*?):::$/)
|
|
195
|
+
if (m) {
|
|
196
|
+
const c = colorMap[m[1]] || colorMap.info
|
|
197
|
+
const title = m[2].trim()
|
|
198
|
+
const body = m[3].trim()
|
|
199
|
+
// 卡片标题是**用户可控文本**,落在元素**内容位**:`:::tip <b style="…">x</b>`
|
|
200
|
+
// 不转义就等于往正文里插任意元素。与同文件的 image 规则同口径 escapeHtml。
|
|
201
|
+
const titleText = md.utils.escapeHtml(title || c.icon + ' ' + m[1].toUpperCase())
|
|
202
|
+
const tk = new state.Token('html_block', '', 0)
|
|
203
|
+
// 仅供块切分使用的元数据:不改渲染输出,只让块模型认出"这是一个卡片"
|
|
204
|
+
tk.map = tokens[i].map
|
|
205
|
+
tk.meta = { fishpai: `card:${m[1]}` }
|
|
206
|
+
tk.content = `<div style="border-left: 4px solid ${c.border}; background: ${c.bg}; padding: 14px 18px; margin: 14px 0; border-radius: 0 8px 8px 0;">
|
|
207
|
+
<div style="font-weight: 600; color: ${c.color}; margin-bottom: 6px; font-size: 15px;">${titleText}</div>
|
|
208
|
+
<div style="color: ${c.color}; opacity: 0.85; font-size: 14px; line-height: 1.7;">${md.renderInline(body)}</div>
|
|
209
|
+
</div>`
|
|
210
|
+
tokens.splice(i, 3, tk)
|
|
211
|
+
continue
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
i++
|
|
216
|
+
}
|
|
217
|
+
})
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* 微信平台兼容(可选,`wrapText`):把文字包进 `<span>`。
|
|
221
|
+
*
|
|
222
|
+
* 为什么需要:微信编辑器的结构校验用 `Range.getClientRects()` 量"文字行数",
|
|
223
|
+
* 而**行内元素(`<span>`/`<a>`/`<strong>`)会把同一行拆成多个矩形**——「内容高度 ÷ 矩形数」
|
|
224
|
+
* 因此被算小,含行内元素的段落会被误判成"行高小于字体大小、多行文字重叠"
|
|
225
|
+
* (实测:`line-height: 1.8` 的两行段落含一个链接 → 6 个矩形 → 判定叠字)。
|
|
226
|
+
*
|
|
227
|
+
* 把文字包进 `<span>` 后,块级元素不再有**直接文字子节点**,那条校验就不再命中
|
|
228
|
+
* (官方校验器实测 `isValid: true`)。视觉上零影响——span 不带任何样式;
|
|
229
|
+
* 而且这正是微信自己插入内容后的结构(`<span leaf="">`)。
|
|
230
|
+
*
|
|
231
|
+
* **但首尾空白要留在 span 外面**(踩过):微信的编辑器会把"以空白开头"的行内元素
|
|
232
|
+
* 当成新的行/块——列表项 `**批注** | 光标…` 粘过去会变成两行("批注"一行、`|` 开头一行),
|
|
233
|
+
* 而面板预览里是正常的一行。空白放在 span 外一样挡得住结构校验(那条规则只看**非空白**的直接文字子节点),
|
|
234
|
+
* 顺带也不再有 `<span></span>` 这种空壳。
|
|
235
|
+
*
|
|
236
|
+
* 默认关闭:默认路径必须与 `test/golden/**` 逐字节一致(那是本项目的地基)。
|
|
237
|
+
* 复制到公众号/导出这条路径才开(见 plugin/host/routes.mjs 与 tools.mjs)。
|
|
238
|
+
*/
|
|
239
|
+
if (opts.wrapText) {
|
|
240
|
+
const escapeHtml = md.utils.escapeHtml
|
|
241
|
+
md.renderer.rules.text = (tokens, idx) => {
|
|
242
|
+
const raw = String(tokens[idx].content)
|
|
243
|
+
const lead = /^\s*/.exec(raw)[0]
|
|
244
|
+
const tail = /\s*$/.exec(raw)[0]
|
|
245
|
+
const core = raw.slice(lead.length, raw.length - tail.length)
|
|
246
|
+
if (!core) return escapeHtml(raw) // 纯空白:原样输出,不包空壳 span
|
|
247
|
+
return `${escapeHtml(lead)}<span>${escapeHtml(core)}</span>${escapeHtml(tail)}`
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
return md
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/** 只要 token 流、不要样式的解析器(块切分与 diff 用)。 */
|
|
255
|
+
export function createParser() {
|
|
256
|
+
return createMd({}, { macCodeBlock: true })
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/** 从 opening token 推出块类型。 */
|
|
260
|
+
function kindOf(token) {
|
|
261
|
+
const fp = token.meta && token.meta.fishpai
|
|
262
|
+
if (typeof fp === 'string' && fp.startsWith('card:')) return { kind: 'card', variant: fp.slice(5) }
|
|
263
|
+
switch (token.type) {
|
|
264
|
+
case 'heading_open':
|
|
265
|
+
return { kind: 'heading', level: Number(token.tag.slice(1)) || 1 }
|
|
266
|
+
case 'paragraph_open':
|
|
267
|
+
return { kind: 'paragraph' }
|
|
268
|
+
case 'blockquote_open':
|
|
269
|
+
return { kind: 'blockquote' }
|
|
270
|
+
case 'bullet_list_open':
|
|
271
|
+
return { kind: 'list', ordered: false }
|
|
272
|
+
case 'ordered_list_open':
|
|
273
|
+
return { kind: 'list', ordered: true }
|
|
274
|
+
case 'table_open':
|
|
275
|
+
return { kind: 'table' }
|
|
276
|
+
case 'fence':
|
|
277
|
+
return { kind: 'code', lang: (token.info || '').trim() || null }
|
|
278
|
+
case 'code_block':
|
|
279
|
+
return { kind: 'code', lang: null }
|
|
280
|
+
case 'hr':
|
|
281
|
+
return { kind: 'hr' }
|
|
282
|
+
case 'html_block':
|
|
283
|
+
return { kind: 'html' }
|
|
284
|
+
default:
|
|
285
|
+
return { kind: token.type }
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
/**
|
|
290
|
+
* 顶层块切分:把 token 流按 level 0 的开合配对切成块,并给出源码行范围。
|
|
291
|
+
*
|
|
292
|
+
* 块身份(id)是**内容寻址**的:同一段文字在文档里重复出现时,第 2 个起带 `#n` 后缀。
|
|
293
|
+
* 所以人改了某块的内容,它的 id 就会变——这正是批注需要"重锚"而不是硬绑 id 的原因。
|
|
294
|
+
*
|
|
295
|
+
* @param {string} markdown 源码
|
|
296
|
+
* @returns {Array<object>} 块记录:index/id/kind/level/headingPath/hash/startLine/endLine/text/tokenIndex
|
|
297
|
+
*/
|
|
298
|
+
export function collectBlocks(markdown, tokens) {
|
|
299
|
+
const lines = markdown.split('\n')
|
|
300
|
+
const blocks = []
|
|
301
|
+
const headingStack = []
|
|
302
|
+
const hashCount = new Map()
|
|
303
|
+
let i = 0
|
|
304
|
+
|
|
305
|
+
while (i < tokens.length) {
|
|
306
|
+
const t = tokens[i]
|
|
307
|
+
if (t.level !== 0) {
|
|
308
|
+
i++
|
|
309
|
+
continue
|
|
310
|
+
}
|
|
311
|
+
const tokenIndex = i
|
|
312
|
+
let end = t.map ? t.map[1] : null
|
|
313
|
+
let j = i
|
|
314
|
+
if (t.nesting === 1) {
|
|
315
|
+
let depth = 0
|
|
316
|
+
while (j < tokens.length) {
|
|
317
|
+
const u = tokens[j]
|
|
318
|
+
if (u.level === 0) {
|
|
319
|
+
if (u.nesting === 1) depth++
|
|
320
|
+
else if (u.nesting === -1) {
|
|
321
|
+
depth--
|
|
322
|
+
if (depth === 0) break
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
if (u.map && (end === null || u.map[1] > end)) end = u.map[1]
|
|
326
|
+
j++
|
|
327
|
+
}
|
|
328
|
+
} else {
|
|
329
|
+
j = i
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
const start = t.map ? t.map[0] : null
|
|
333
|
+
const info = kindOf(t)
|
|
334
|
+
const text = start === null || end === null ? '' : lines.slice(start, end).join('\n')
|
|
335
|
+
|
|
336
|
+
if (info.kind === 'heading') {
|
|
337
|
+
headingStack.length = Math.max(0, info.level - 1)
|
|
338
|
+
headingStack[info.level - 1] = text.replace(/^#+\s*/, '').trim()
|
|
339
|
+
}
|
|
340
|
+
const headingPath = headingStack.filter(Boolean).slice()
|
|
341
|
+
|
|
342
|
+
const hash = shortHash(info.kind + '\u0000' + normalizeText(text))
|
|
343
|
+
const seen = (hashCount.get(hash) || 0) + 1
|
|
344
|
+
hashCount.set(hash, seen)
|
|
345
|
+
|
|
346
|
+
blocks.push({
|
|
347
|
+
index: blocks.length,
|
|
348
|
+
id: seen === 1 ? hash : `${hash}#${seen}`,
|
|
349
|
+
kind: info.kind,
|
|
350
|
+
variant: info.variant || null,
|
|
351
|
+
ordered: info.ordered ?? null,
|
|
352
|
+
lang: info.lang || null,
|
|
353
|
+
level: info.level || null,
|
|
354
|
+
headingPath,
|
|
355
|
+
hash,
|
|
356
|
+
startLine: start === null ? 0 : start + 1, // 1-based,便于人读
|
|
357
|
+
endLine: end === null ? 0 : end,
|
|
358
|
+
text,
|
|
359
|
+
tokenIndex,
|
|
360
|
+
endTokenIndex: j,
|
|
361
|
+
})
|
|
362
|
+
i = j + 1
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
return blocks
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
/** 解析源码并切块(block/diff/notes 的公共入口)。 */
|
|
369
|
+
export function splitBlocks(markdown) {
|
|
370
|
+
const md = createParser()
|
|
371
|
+
const tokens = md.parse(markdown, {})
|
|
372
|
+
return collectBlocks(markdown, tokens)
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
/** 归一化:行尾空白、连续空白压平(只用于 hash,不改变原文)。 */
|
|
376
|
+
export function normalizeText(text) {
|
|
377
|
+
return String(text).replace(/[ \t]+$/gm, '').replace(/\r\n?/g, '\n').replace(/\s+/g, ' ').trim()
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
/** 短哈希(FNV-1a 32bit → base36),够短够稳,且不引入依赖。 */
|
|
381
|
+
export function shortHash(input) {
|
|
382
|
+
let h = 0x811c9dc5
|
|
383
|
+
const s = String(input)
|
|
384
|
+
for (let i = 0; i < s.length; i++) {
|
|
385
|
+
h ^= s.charCodeAt(i)
|
|
386
|
+
h = Math.imul(h, 0x01000193) >>> 0
|
|
387
|
+
}
|
|
388
|
+
return h.toString(36).padStart(7, '0')
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
/** 字符二元组 Dice 相似度:对中文短句比编辑距离更稳,且不需要分词。 */
|
|
392
|
+
export function similarity(a, b) {
|
|
393
|
+
const x = normalizeText(a)
|
|
394
|
+
const y = normalizeText(b)
|
|
395
|
+
if (!x && !y) return 1
|
|
396
|
+
if (!x || !y) return 0
|
|
397
|
+
if (x === y) return 1
|
|
398
|
+
const grams = (s) => {
|
|
399
|
+
const out = new Map()
|
|
400
|
+
if (s.length === 1) {
|
|
401
|
+
out.set(s, 1)
|
|
402
|
+
return out
|
|
403
|
+
}
|
|
404
|
+
for (let i = 0; i < s.length - 1; i++) {
|
|
405
|
+
const g = s.slice(i, i + 2)
|
|
406
|
+
out.set(g, (out.get(g) || 0) + 1)
|
|
407
|
+
}
|
|
408
|
+
return out
|
|
409
|
+
}
|
|
410
|
+
const ga = grams(x)
|
|
411
|
+
const gb = grams(y)
|
|
412
|
+
let overlap = 0
|
|
413
|
+
let totalA = 0
|
|
414
|
+
let totalB = 0
|
|
415
|
+
for (const n of ga.values()) totalA += n
|
|
416
|
+
for (const n of gb.values()) totalB += n
|
|
417
|
+
for (const [g, n] of ga) if (gb.has(g)) overlap += Math.min(n, gb.get(g))
|
|
418
|
+
return (2 * overlap) / (totalA + totalB)
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
/**
|
|
422
|
+
* 引用片段的"包含度":`quote` 的最长连续子串在 `text` 里占了 quote 的多大比例。
|
|
423
|
+
*
|
|
424
|
+
* 为什么不用 Dice:批注的 quote 通常是整段里的一小句,Dice 会被长段落的长度稀释,
|
|
425
|
+
* 于是"人只改了几个字"也会被判成失锚。包含度只问"这句话还在不在这一段里"。
|
|
426
|
+
*/
|
|
427
|
+
export function containment(quote, text) {
|
|
428
|
+
const a = normalizeText(quote)
|
|
429
|
+
const b = normalizeText(text)
|
|
430
|
+
if (!a || !b) return 0
|
|
431
|
+
if (b.includes(a)) return 1
|
|
432
|
+
// 超长文本截断,避免 O(n·m) 爆掉;批注引用本来也不该很长
|
|
433
|
+
const x = a.slice(0, 400)
|
|
434
|
+
const y = b.slice(0, 4000)
|
|
435
|
+
let best = 0
|
|
436
|
+
// dp 只保留上一行
|
|
437
|
+
let prev = new Uint16Array(y.length + 1)
|
|
438
|
+
let cur = new Uint16Array(y.length + 1)
|
|
439
|
+
for (let i = 1; i <= x.length; i++) {
|
|
440
|
+
for (let j = 1; j <= y.length; j++) {
|
|
441
|
+
cur[j] = x[i - 1] === y[j - 1] ? prev[j - 1] + 1 : 0
|
|
442
|
+
if (cur[j] > best) best = cur[j]
|
|
443
|
+
}
|
|
444
|
+
const tmp = prev
|
|
445
|
+
prev = cur
|
|
446
|
+
cur = tmp
|
|
447
|
+
cur.fill(0)
|
|
448
|
+
}
|
|
449
|
+
return best / x.length
|
|
450
|
+
}
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 批注与占位:人留给模型的"这批文字我想怎样"的信号。
|
|
3
|
+
*
|
|
4
|
+
* 两种载体,互不替代:
|
|
5
|
+
* 1. **行内占位** `<!-- 鱼排: 这里补一个过渡句 -->` —— 人在正文里直接打字留下的记号。
|
|
6
|
+
* 它是 Markdown 的一部分,所以天然进 diff;渲染时是 HTML 注释,公众号里不可见。
|
|
7
|
+
* 2. **块级批注**(sidecar)—— 人不想动正文时用;挂在某个块上,人改了那段文字就靠
|
|
8
|
+
* hash / 引用片段重锚,锚不上就标 orphan 而不是悄悄丢掉。
|
|
9
|
+
*
|
|
10
|
+
* 这一层是纯函数:不碰文件系统,宿主负责存取。
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { containment } from './markdown.mjs'
|
|
14
|
+
|
|
15
|
+
/** 行内占位语法:`<!-- 鱼排: 文字 -->`(也接受 fishpai / 中文冒号)。 */
|
|
16
|
+
export const PLACEHOLDER_RE = /<!--\s*(?:鱼排|fishpai)\s*[::]\s*([\s\S]*?)\s*-->/g
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* 把**代码区域**遮成等长空白,行结构不变(行号还能对上原文)。
|
|
20
|
+
*
|
|
21
|
+
* 为什么需要:文章里讲解用法时常常把 `<!-- 鱼排: 这里补个数据 -->` 写进行内代码当例子,
|
|
22
|
+
* 那是"示例"不是"待办"。不遮的话面板会一直显示"待补 1",而人根本找不到要补什么。
|
|
23
|
+
* 遮的是**围栏代码块**与**行内代码**(缩进式代码块不遮:那需要更重的解析,收益也小)。
|
|
24
|
+
*/
|
|
25
|
+
function maskCode(markdown) {
|
|
26
|
+
const lines = String(markdown || '').split('\n')
|
|
27
|
+
let fence = null
|
|
28
|
+
return lines.map((line) => {
|
|
29
|
+
const m = /^\s{0,3}(`{3,}|~{3,})/.exec(line)
|
|
30
|
+
if (fence) {
|
|
31
|
+
if (m && m[1][0] === fence[0] && m[1].length >= fence.length) fence = null
|
|
32
|
+
return ' '.repeat(line.length)
|
|
33
|
+
}
|
|
34
|
+
if (m) {
|
|
35
|
+
fence = m[1]
|
|
36
|
+
return ' '.repeat(line.length)
|
|
37
|
+
}
|
|
38
|
+
return line.replace(/(`+)[^`]*\1/g, (full) => ' '.repeat(full.length))
|
|
39
|
+
})
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** 从 Markdown 里抽出全部行内占位(1-based 行号,便于面板跳转)。代码区域里的不算。 */
|
|
43
|
+
export function extractPlaceholders(markdown) {
|
|
44
|
+
const source = String(markdown || '')
|
|
45
|
+
const mask = maskCode(source).join('\n')
|
|
46
|
+
// 遮罩与原文等长、行结构相同,所以在整篇上跑正则、再按字符偏移拆出行号/列号——
|
|
47
|
+
// 逐行扫会漏掉**跨行**的占位(`<!-- 鱼排: 第一行\n第二行 -->` 这种写法直接整条消失)。
|
|
48
|
+
const lines = source.split('\n')
|
|
49
|
+
const lineStarts = [0]
|
|
50
|
+
for (let i = 0; i < lines.length - 1; i++) lineStarts.push(lineStarts[i] + lines[i].length + 1)
|
|
51
|
+
const out = []
|
|
52
|
+
const re = new RegExp(PLACEHOLDER_RE.source, 'g')
|
|
53
|
+
let lineNo = 0
|
|
54
|
+
let m
|
|
55
|
+
while ((m = re.exec(mask)) !== null) {
|
|
56
|
+
while (lineStarts[lineNo + 1] !== undefined && lineStarts[lineNo + 1] <= m.index) lineNo++
|
|
57
|
+
// `col` = 这一行里的字符偏移:同一行有两处**文字相同**的占位时,
|
|
58
|
+
// 只按 `行号-文字` 做 React key 会撞(重复 key → 警告与错渲染)
|
|
59
|
+
out.push({ line: lineNo + 1, col: m.index - lineStarts[lineNo], text: m[1].trim(), raw: m[0], blockId: null })
|
|
60
|
+
}
|
|
61
|
+
return out
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/** 把占位挂到它所在的块上(面板跳转与模型定位用)。 */
|
|
65
|
+
export function attachPlaceholdersToBlocks(placeholders, blocks) {
|
|
66
|
+
return placeholders.map((p) => {
|
|
67
|
+
const block = blocks.find((b) => b.startLine <= p.line && p.line <= b.endLine)
|
|
68
|
+
return { ...p, blockId: block ? block.id : null, blockIndex: block ? block.index : null }
|
|
69
|
+
})
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** 是否还有未处理的占位(复制到公众号前的提醒用)。 */
|
|
73
|
+
export function hasPlaceholders(markdown) {
|
|
74
|
+
return extractPlaceholders(markdown).length > 0
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
let noteSeq = 0
|
|
78
|
+
|
|
79
|
+
/** 造一条批注。id 只在一次进程内唯一即可(真正持久化靠 sidecar 里的 id)。 */
|
|
80
|
+
export function makeNote({ blockId = null, quote = '', text = '', author = 'human', at = Date.now() } = {}) {
|
|
81
|
+
noteSeq += 1
|
|
82
|
+
return {
|
|
83
|
+
id: `n${at.toString(36)}${noteSeq.toString(36)}`,
|
|
84
|
+
blockId,
|
|
85
|
+
quote: String(quote || '').slice(0, 200),
|
|
86
|
+
text: String(text || '').slice(0, 2000),
|
|
87
|
+
author,
|
|
88
|
+
at,
|
|
89
|
+
resolved: false,
|
|
90
|
+
orphan: false,
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* 重锚批注:人改过正文之后,原来的 blockId 可能已经不存在。
|
|
96
|
+
* 顺序:id 命中 → hash 命中 → 引用片段模糊命中 → orphan。
|
|
97
|
+
*
|
|
98
|
+
* 性能:模糊命中那一步是 notes × blocks 的 `containment`,但每对内部已截断到
|
|
99
|
+
* 400 × 4000 字符(见 markdown.mjs),实测 50 注 × 200 块的全失锚最坏情况约 40ms——
|
|
100
|
+
* 预览那条路有 300ms 防抖,够用。外层不需要再加批注数上限:真加出 500 注的文档时,
|
|
101
|
+
* 该担心的是批注本身而不是这里。
|
|
102
|
+
*
|
|
103
|
+
* @param {Array<object>} notes sidecar 里的批注
|
|
104
|
+
* @param {Array<object>} blocks 当前文档的块
|
|
105
|
+
* @returns {Array<object>} 锚定结果(新对象;不修改入参)
|
|
106
|
+
*/
|
|
107
|
+
export function reanchorNotes(notes, blocks) {
|
|
108
|
+
const byId = new Map(blocks.map((b) => [b.id, b]))
|
|
109
|
+
return notes.map((note) => {
|
|
110
|
+
const exact = note.blockId ? byId.get(note.blockId) : null
|
|
111
|
+
if (exact) return { ...note, blockIndex: exact.index, headingPath: exact.headingPath, orphan: false }
|
|
112
|
+
|
|
113
|
+
const quote = normalize(note.quote)
|
|
114
|
+
if (quote) {
|
|
115
|
+
// 先找包含引用片段的块;多个候选时取最靠前的(人改字后位置通常不变)
|
|
116
|
+
const contains = blocks.find((b) => normalize(b.text).includes(quote))
|
|
117
|
+
if (contains) {
|
|
118
|
+
return { ...note, blockId: contains.id, blockIndex: contains.index, headingPath: contains.headingPath, orphan: false }
|
|
119
|
+
}
|
|
120
|
+
// 再退一步:人把那句话改掉了一部分,就看"这句话还剩多少"在哪个块里
|
|
121
|
+
let best = null
|
|
122
|
+
let bestScore = 0
|
|
123
|
+
for (const b of blocks) {
|
|
124
|
+
const score = containment(quote, b.text)
|
|
125
|
+
if (score > bestScore) {
|
|
126
|
+
bestScore = score
|
|
127
|
+
best = b
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
if (best && bestScore >= 0.6) {
|
|
131
|
+
return { ...note, blockId: best.id, blockIndex: best.index, headingPath: best.headingPath, orphan: false }
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
return { ...note, blockIndex: null, headingPath: [], orphan: true }
|
|
135
|
+
})
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function normalize(s) {
|
|
139
|
+
return String(s || '')
|
|
140
|
+
.replace(/\s+/g, ' ')
|
|
141
|
+
.trim()
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/** 面板上的分组视图:未解决 / 已解决 / 失锚。 */
|
|
145
|
+
export function groupNotes(notes) {
|
|
146
|
+
return {
|
|
147
|
+
open: notes.filter((n) => !n.resolved && !n.orphan),
|
|
148
|
+
orphan: notes.filter((n) => n.orphan),
|
|
149
|
+
resolved: notes.filter((n) => n.resolved),
|
|
150
|
+
}
|
|
151
|
+
}
|