officeparser 7.2.2 → 7.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +161 -17
- package/dist/OfficeGenerator.js +4 -0
- package/dist/OfficeParser.d.ts +2 -0
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +3 -2
- package/dist/defaults.js +3 -3
- package/dist/generators/BaseGenerator.d.ts +11 -0
- package/dist/generators/BaseGenerator.js +29 -0
- package/dist/generators/CsvGenerator.d.ts +9 -1
- package/dist/generators/CsvGenerator.js +24 -14
- package/dist/generators/EpubGenerator.d.ts +18 -0
- package/dist/generators/EpubGenerator.js +242 -0
- package/dist/generators/HtmlGenerator.d.ts +12 -0
- package/dist/generators/HtmlGenerator.js +266 -51
- package/dist/generators/MarkdownGenerator.d.ts +16 -0
- package/dist/generators/MarkdownGenerator.js +173 -24
- package/dist/generators/PdfGenerator.js +32 -0
- package/dist/generators/RtfGenerator.js +12 -15
- package/dist/generators/TextGenerator.js +11 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/officeparser.browser.d.ts +144 -7
- package/dist/officeparser.browser.iife.js +289 -193
- package/dist/officeparser.browser.mjs +289 -193
- package/dist/officeparser.browser.slim.d.ts +2129 -0
- package/dist/officeparser.browser.slim.iife.js +1278 -0
- package/dist/officeparser.browser.slim.mjs +1277 -0
- package/dist/parsers/EpubParser.d.ts +8 -0
- package/dist/parsers/EpubParser.js +217 -0
- package/dist/parsers/HtmlParser.js +284 -20
- package/dist/parsers/MarkdownParser.js +424 -33
- package/dist/parsers/OpenOfficeParser.js +241 -54
- package/dist/parsers/PdfParser.js +4 -1
- package/dist/parsers/WordParser.js +2 -2
- package/dist/sbom.cdx.json +111 -223
- package/dist/types.d.ts +146 -7
- package/dist/types.js +2 -0
- package/dist/utils/errorUtils.js +3 -2
- package/dist/utils/sanitize.d.ts +99 -0
- package/dist/utils/sanitize.js +228 -0
- package/dist/utils/zipUtils.js +76 -26
- package/package.json +19 -10
|
@@ -3,6 +3,53 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
3
3
|
exports.parseMarkdown = void 0;
|
|
4
4
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
5
5
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
6
|
+
// Sentinel node type for a standalone bookmark-anchor block (e.g. `<a id="x"></a>` on its
|
|
7
|
+
// own line). A post-parse pass folds these into the following node's anchorIds so they
|
|
8
|
+
// round-trip as real anchors rather than being escaped to visible text on regeneration.
|
|
9
|
+
const ANCHOR_PLACEHOLDER = '__anchorPlaceholder__';
|
|
10
|
+
/**
|
|
11
|
+
* Splits the inner content of a YAML flow array (`a, "b, c", d`) on top-level commas,
|
|
12
|
+
* ignoring commas inside single- or double-quoted items.
|
|
13
|
+
*/
|
|
14
|
+
const splitFlowArrayItems = (inner) => {
|
|
15
|
+
const items = [];
|
|
16
|
+
let current = '';
|
|
17
|
+
let quote = null;
|
|
18
|
+
for (const ch of inner) {
|
|
19
|
+
if (quote) {
|
|
20
|
+
current += ch;
|
|
21
|
+
if (ch === quote)
|
|
22
|
+
quote = null;
|
|
23
|
+
}
|
|
24
|
+
else if (ch === '"' || ch === '\'') {
|
|
25
|
+
quote = ch;
|
|
26
|
+
current += ch;
|
|
27
|
+
}
|
|
28
|
+
else if (ch === ',') {
|
|
29
|
+
items.push(current.trim());
|
|
30
|
+
current = '';
|
|
31
|
+
}
|
|
32
|
+
else {
|
|
33
|
+
current += ch;
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
if (current.trim() !== '')
|
|
37
|
+
items.push(current.trim());
|
|
38
|
+
return items;
|
|
39
|
+
};
|
|
40
|
+
/**
|
|
41
|
+
* Maps every accepted-on-import admonition type spelling (GitHub's five plus GLFM's
|
|
42
|
+
* `danger`) to the canonical AdmonitionMetadata type. Per MARKDOWN_DIALECT.md's
|
|
43
|
+
* Decisions, `danger` folds into `caution` - there is no separate danger type.
|
|
44
|
+
*/
|
|
45
|
+
const ADMONITION_TYPE_MAP = {
|
|
46
|
+
note: 'note',
|
|
47
|
+
tip: 'tip',
|
|
48
|
+
important: 'important',
|
|
49
|
+
warning: 'warning',
|
|
50
|
+
caution: 'caution',
|
|
51
|
+
danger: 'caution'
|
|
52
|
+
};
|
|
6
53
|
const parseMarkdown = async (buffer, config) => {
|
|
7
54
|
// Honour cancellation requests before the line-by-line Markdown scanning loop begins.
|
|
8
55
|
// Markdown parsing is entirely synchronous and CPU-bound, so failing fast avoids
|
|
@@ -26,9 +73,23 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
26
73
|
const match = line.match(/^([^:]+):\s*(.*)$/);
|
|
27
74
|
if (match) {
|
|
28
75
|
const key = match[1].trim();
|
|
29
|
-
|
|
76
|
+
const rawVal = match[2].trim();
|
|
77
|
+
const val = rawVal.replace(/^"(.*)"$/, '$1');
|
|
30
78
|
let parsedVal = val;
|
|
31
|
-
if (
|
|
79
|
+
if (rawVal.startsWith('[') && rawVal.endsWith(']')) {
|
|
80
|
+
// Flow-array (`tags: [a, b]`) or JSON-array (`tags: ["a","b"]`) value -
|
|
81
|
+
// parse into a real array instead of storing the literal bracket string,
|
|
82
|
+
// so it round-trips symmetrically with MarkdownGenerator's frontmatter output.
|
|
83
|
+
try {
|
|
84
|
+
const jsonParsed = JSON.parse(rawVal);
|
|
85
|
+
parsedVal = Array.isArray(jsonParsed) ? jsonParsed : val;
|
|
86
|
+
}
|
|
87
|
+
catch {
|
|
88
|
+
const inner = rawVal.slice(1, -1).trim();
|
|
89
|
+
parsedVal = inner === '' ? [] : splitFlowArrayItems(inner).map(item => item.replace(/^['"](.*)['"]$/, '$1'));
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
else if (val === 'true')
|
|
32
93
|
parsedVal = true;
|
|
33
94
|
else if (val === 'false')
|
|
34
95
|
parsedVal = false;
|
|
@@ -56,6 +117,22 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
56
117
|
metadata.nativeProperties = nativeProps;
|
|
57
118
|
}
|
|
58
119
|
}
|
|
120
|
+
// Strip MDX/JSX component tags (parse-only - we never author MDX). Components are
|
|
121
|
+
// distinguished from plain HTML by an uppercase-leading tag name, matching React/MDX
|
|
122
|
+
// convention. Self-closing components are removed entirely; paired components keep
|
|
123
|
+
// their inner Markdown content. Iterate to a fixed point so nested components (of
|
|
124
|
+
// different names) are all unwrapped, not just the outermost one.
|
|
125
|
+
// Cap the passes: each iteration unwraps one nesting level, so a pathologically
|
|
126
|
+
// deep `<A><A>...</A></A>` input would otherwise be O(depth * n). Real documents
|
|
127
|
+
// nest only a handful of levels; anything past the cap is left as-is.
|
|
128
|
+
let previousTextStr;
|
|
129
|
+
let mdxPasses = 0;
|
|
130
|
+
const MAX_MDX_PASSES = 100;
|
|
131
|
+
do {
|
|
132
|
+
previousTextStr = textStr;
|
|
133
|
+
textStr = textStr.replace(/<[A-Z][A-Za-z0-9]*(?:\s+[^>]*?)?\/>/g, '');
|
|
134
|
+
textStr = textStr.replace(/<([A-Z][A-Za-z0-9]*)(?:\s+[^>]*?)?>([\s\S]*?)<\/\1>/g, (_m, _name, inner) => inner);
|
|
135
|
+
} while (textStr !== previousTextStr && ++mdxPasses < MAX_MDX_PASSES);
|
|
59
136
|
// Extract code blocks first to protect their contents
|
|
60
137
|
const codeBlocks = [];
|
|
61
138
|
textStr = textStr.replace(/^```(\w*)\n([\s\S]*?)\n```/gm, (match, lang, code) => {
|
|
@@ -63,10 +140,79 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
63
140
|
codeBlocks.push(JSON.stringify({ lang, code }));
|
|
64
141
|
return `\n\n${id}\n\n`;
|
|
65
142
|
});
|
|
143
|
+
// Extract block math ($$\n...\n$$) before block splitting, mirroring the code-block
|
|
144
|
+
// pre-pass above - its body may contain blank lines that would otherwise fragment it.
|
|
145
|
+
// Inline math ($...$) is handled directly in parseInline below.
|
|
146
|
+
const mathBlocks = [];
|
|
147
|
+
textStr = textStr.replace(/^\$\$\n([\s\S]*?)\n\$\$$/gm, (_match, latex) => {
|
|
148
|
+
const id = `__MATH_BLOCK_${mathBlocks.length}__`;
|
|
149
|
+
mathBlocks.push(latex);
|
|
150
|
+
return `\n\n${id}\n\n`;
|
|
151
|
+
});
|
|
152
|
+
// Extract GLFM-style fenced-div admonitions (`:::note ... :::`) before block splitting,
|
|
153
|
+
// since their body may itself contain blank lines that would otherwise fragment them.
|
|
154
|
+
// The `> [!NOTE]` GitHub form doesn't need this - it's detected inline in the blockquote
|
|
155
|
+
// branch below, since a `>`-prefixed block never contains a real blank line.
|
|
156
|
+
const admonitionBlocks = [];
|
|
157
|
+
textStr = textStr.replace(/^:::(\w+)[ \t]*\n([\s\S]*?)\n:::[ \t]*$/gm, (match, type, body) => {
|
|
158
|
+
const admonitionType = ADMONITION_TYPE_MAP[type.toLowerCase()];
|
|
159
|
+
if (!admonitionType)
|
|
160
|
+
return match; // Unrecognised type - leave as literal text.
|
|
161
|
+
const id = `__ADMONITION_${admonitionBlocks.length}__`;
|
|
162
|
+
admonitionBlocks.push(JSON.stringify({ admonitionType, body }));
|
|
163
|
+
return `\n\n${id}\n\n`;
|
|
164
|
+
});
|
|
165
|
+
// Extract footnote definitions (`[^id]: text`) before block splitting, since
|
|
166
|
+
// definitions conventionally live at the end of the document, after every place
|
|
167
|
+
// they're referenced - inline parsing below needs the full map upfront. v1 only
|
|
168
|
+
// supports single-line definitions (MultiMarkdown/Pandoc/GLFM's common baseline).
|
|
169
|
+
const footnoteDefinitions = new Map();
|
|
170
|
+
textStr = textStr.replace(/^\[\^([^\]]+)\]:[ \t]*(.*)$/gm, (_match, id, definition) => {
|
|
171
|
+
footnoteDefinitions.set(id, definition.trim());
|
|
172
|
+
return '';
|
|
173
|
+
});
|
|
174
|
+
// Extract Markdown Extra abbreviation definitions (`*[HTML]: Hypertext Markup Language`)
|
|
175
|
+
// before block splitting, for the same reason as footnotes: they conventionally live
|
|
176
|
+
// at the end of the document.
|
|
177
|
+
const abbreviationDefinitions = new Map();
|
|
178
|
+
textStr = textStr.replace(/^\*\[([^\]]+)\]:[ \t]*(.*)$/gm, (_match, abbr, definition) => {
|
|
179
|
+
abbreviationDefinitions.set(abbr, definition.trim());
|
|
180
|
+
return '';
|
|
181
|
+
});
|
|
182
|
+
// Parses a Pandoc-style attribute list body (the part inside `{...}`), e.g.
|
|
183
|
+
// `width=50% .centered` or `align=right`. Per MARKDOWN_DIALECT.md §15's Decisions,
|
|
184
|
+
// the vocabulary matches ImageMetadata/TableMetadata's own width/align fields;
|
|
185
|
+
// several class-name spellings are accepted on import for compatibility with
|
|
186
|
+
// hand-written content, but the generator only ever emits canonical `align=value`.
|
|
187
|
+
const parseAttributeList = (attrStr) => {
|
|
188
|
+
const result = {};
|
|
189
|
+
for (const token of attrStr.trim().split(/\s+/).filter(Boolean)) {
|
|
190
|
+
const kv = token.match(/^([a-zA-Z-]+)=(.+)$/);
|
|
191
|
+
if (kv) {
|
|
192
|
+
if (kv[1] === 'width')
|
|
193
|
+
result.width = kv[2];
|
|
194
|
+
else if (kv[1] === 'align' && ['left', 'center', 'right'].includes(kv[2]))
|
|
195
|
+
result.align = kv[2];
|
|
196
|
+
}
|
|
197
|
+
else if (token.startsWith('.')) {
|
|
198
|
+
const cls = token.slice(1).toLowerCase();
|
|
199
|
+
if (cls === 'left' || cls === 'align-left')
|
|
200
|
+
result.align = 'left';
|
|
201
|
+
else if (cls === 'center' || cls === 'centered' || cls === 'align-center')
|
|
202
|
+
result.align = 'center';
|
|
203
|
+
else if (cls === 'right' || cls === 'align-right')
|
|
204
|
+
result.align = 'right';
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
return result;
|
|
208
|
+
};
|
|
66
209
|
const parseInline = (text, currentFormatting = {}) => {
|
|
67
210
|
const nodes = [];
|
|
68
|
-
// Regex matches: 1=!, 2=alt, 3=url |
|
|
69
|
-
|
|
211
|
+
// Regex matches: 1=!, 2=alt, 3=url, 4=attrs | 5=bold | 6=italic | 7=strike | 8=code | 9=underline | 10=subscript | 11=superscript | 12=footnote id | 13=citekey | 14=wikilink page | 15=wikilink alias | 16=inline math
|
|
212
|
+
// Inline math requires no whitespace right after the opening $ or right before the
|
|
213
|
+
// closing $, the common heuristic (matching Pandoc/KaTeX) for avoiding false
|
|
214
|
+
// positives on currency like "$5 and $10".
|
|
215
|
+
const regex = /(!?)\[(.*?)\]\((.*?)\)(?:\{([^}]*)\})?|\*\*(.+?)\*\*|\*(.+?)\*|~~(.+?)~~|`(.+?)`|<u>(.+?)<\/u>|<sub>(.+?)<\/sub>|<sup>(.+?)<\/sup>|\[\^([^\]]+)\]|\[@([a-zA-Z0-9_:.-]+)\]|\[\[([^\]|]+)(?:\|([^\]]+))?\]\]|\$(?!\s)([^$\n]+?)(?<!\s)\$/g;
|
|
70
216
|
let lastIndex = 0;
|
|
71
217
|
let match;
|
|
72
218
|
while ((match = regex.exec(text)) !== null) {
|
|
@@ -77,6 +223,8 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
77
223
|
const isImage = match[1] === '!';
|
|
78
224
|
const altText = match[2];
|
|
79
225
|
const url = match[3];
|
|
226
|
+
// Pandoc-style attribute list immediately after an image, e.g. {width=50% .centered}
|
|
227
|
+
const attrs = isImage && match[4] !== undefined ? parseAttributeList(match[4]) : undefined;
|
|
80
228
|
if (isImage) {
|
|
81
229
|
if (url.startsWith('data:')) {
|
|
82
230
|
const dataMatch = url.match(/^data:([^;]+);base64,(.*)$/);
|
|
@@ -91,14 +239,14 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
91
239
|
name,
|
|
92
240
|
extension: mimeType.split('/')[1]
|
|
93
241
|
});
|
|
94
|
-
nodes.push({ type: 'image', metadata: { attachmentName: name, altText } });
|
|
242
|
+
nodes.push({ type: 'image', metadata: { attachmentName: name, altText, ...attrs } });
|
|
95
243
|
}
|
|
96
244
|
else {
|
|
97
|
-
nodes.push({ type: 'image', metadata: { url, altText } });
|
|
245
|
+
nodes.push({ type: 'image', metadata: { url, altText, ...attrs } });
|
|
98
246
|
}
|
|
99
247
|
}
|
|
100
248
|
else {
|
|
101
|
-
nodes.push({ type: 'image', metadata: { url, altText } });
|
|
249
|
+
nodes.push({ type: 'image', metadata: { url, altText, ...attrs } });
|
|
102
250
|
}
|
|
103
251
|
}
|
|
104
252
|
else {
|
|
@@ -111,33 +259,121 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
111
259
|
nodes.push(...linkNodes);
|
|
112
260
|
}
|
|
113
261
|
}
|
|
114
|
-
else if (match[
|
|
115
|
-
nodes.push(...parseInline(match[
|
|
262
|
+
else if (match[5]) { // Bold
|
|
263
|
+
nodes.push(...parseInline(match[5], { ...currentFormatting, bold: true }));
|
|
264
|
+
}
|
|
265
|
+
else if (match[6]) { // Italic
|
|
266
|
+
nodes.push(...parseInline(match[6], { ...currentFormatting, italic: true }));
|
|
267
|
+
}
|
|
268
|
+
else if (match[7]) { // Strikethrough
|
|
269
|
+
nodes.push(...parseInline(match[7], { ...currentFormatting, strikethrough: true }));
|
|
270
|
+
}
|
|
271
|
+
else if (match[8]) { // Inline Code
|
|
272
|
+
nodes.push({ type: 'text', text: match[8], formatting: { ...currentFormatting, font: 'monospace' } });
|
|
273
|
+
}
|
|
274
|
+
else if (match[9]) { // Underline
|
|
275
|
+
nodes.push(...parseInline(match[9], { ...currentFormatting, underline: true }));
|
|
116
276
|
}
|
|
117
|
-
else if (match[
|
|
118
|
-
nodes.push(...parseInline(match[
|
|
277
|
+
else if (match[10]) { // Subscript
|
|
278
|
+
nodes.push(...parseInline(match[10], { ...currentFormatting, subscript: true }));
|
|
119
279
|
}
|
|
120
|
-
else if (match[
|
|
121
|
-
nodes.push(...parseInline(match[
|
|
280
|
+
else if (match[11]) { // Superscript
|
|
281
|
+
nodes.push(...parseInline(match[11], { ...currentFormatting, superscript: true }));
|
|
122
282
|
}
|
|
123
|
-
else if (match[
|
|
124
|
-
|
|
283
|
+
else if (match[12]) { // Footnote reference
|
|
284
|
+
const noteId = match[12];
|
|
285
|
+
const definition = footnoteDefinitions.get(noteId);
|
|
286
|
+
const noteChildren = definition !== undefined ? parseInline(definition) : [];
|
|
287
|
+
const noteNode = {
|
|
288
|
+
type: 'note',
|
|
289
|
+
text: noteChildren.map(c => c.text || '').join(''),
|
|
290
|
+
children: noteChildren,
|
|
291
|
+
metadata: { noteType: 'footnote', noteId }
|
|
292
|
+
};
|
|
293
|
+
// Notes attach to the preceding text node (matches WordParser's convention);
|
|
294
|
+
// fall back to an empty text node if the reference opens the inline run.
|
|
295
|
+
if (nodes.length > 0) {
|
|
296
|
+
const target = nodes[nodes.length - 1];
|
|
297
|
+
if (!target.notes)
|
|
298
|
+
target.notes = [];
|
|
299
|
+
target.notes.push(noteNode);
|
|
300
|
+
}
|
|
301
|
+
else {
|
|
302
|
+
nodes.push({ type: 'text', text: '', notes: [noteNode] });
|
|
303
|
+
}
|
|
125
304
|
}
|
|
126
|
-
else if (match[
|
|
127
|
-
nodes.push(
|
|
305
|
+
else if (match[13]) { // Citation reference
|
|
306
|
+
nodes.push({ type: 'text', text: match[13], metadata: { citationKey: match[13] } });
|
|
128
307
|
}
|
|
129
|
-
else if (match[
|
|
130
|
-
|
|
308
|
+
else if (match[14]) { // Wikilink
|
|
309
|
+
const page = match[14].trim();
|
|
310
|
+
const alias = match[15]?.trim();
|
|
311
|
+
nodes.push({ type: 'text', text: alias || page, metadata: { link: page, linkType: 'internal', wikilink: true } });
|
|
131
312
|
}
|
|
132
|
-
else if (match[
|
|
133
|
-
nodes.push(
|
|
313
|
+
else if (match[16] !== undefined) { // Inline math
|
|
314
|
+
nodes.push({ type: 'code', text: match[16], metadata: { math: 'inline' } });
|
|
134
315
|
}
|
|
135
316
|
lastIndex = regex.lastIndex;
|
|
136
317
|
}
|
|
137
318
|
if (lastIndex < text.length) {
|
|
138
319
|
nodes.push({ type: 'text', text: text.substring(lastIndex), formatting: Object.keys(currentFormatting).length > 0 ? { ...currentFormatting } : undefined });
|
|
139
320
|
}
|
|
140
|
-
return nodes;
|
|
321
|
+
return applyAbbreviations(nodes);
|
|
322
|
+
};
|
|
323
|
+
const escapeRegExpChars = (s) => s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
324
|
+
// Splits abbreviation occurrences out of plain text nodes so they carry
|
|
325
|
+
// TextMetadata.abbreviationTitle, rendered as <abbr title> in HTML/editor output.
|
|
326
|
+
const applyAbbreviations = (nodes) => {
|
|
327
|
+
if (abbreviationDefinitions.size === 0)
|
|
328
|
+
return nodes;
|
|
329
|
+
const pattern = new RegExp(`\\b(${[...abbreviationDefinitions.keys()].map(escapeRegExpChars).join('|')})\\b`, 'g');
|
|
330
|
+
const result = [];
|
|
331
|
+
for (const node of nodes) {
|
|
332
|
+
if (node.type !== 'text' || !node.text || node.metadata) {
|
|
333
|
+
result.push(node);
|
|
334
|
+
continue;
|
|
335
|
+
}
|
|
336
|
+
let lastIndex = 0;
|
|
337
|
+
let match;
|
|
338
|
+
let matched = false;
|
|
339
|
+
pattern.lastIndex = 0;
|
|
340
|
+
while ((match = pattern.exec(node.text)) !== null) {
|
|
341
|
+
matched = true;
|
|
342
|
+
if (match.index > lastIndex) {
|
|
343
|
+
result.push({ type: 'text', text: node.text.substring(lastIndex, match.index), formatting: node.formatting });
|
|
344
|
+
}
|
|
345
|
+
result.push({
|
|
346
|
+
type: 'text',
|
|
347
|
+
text: match[0],
|
|
348
|
+
formatting: node.formatting,
|
|
349
|
+
metadata: { abbreviationTitle: abbreviationDefinitions.get(match[0]) }
|
|
350
|
+
});
|
|
351
|
+
lastIndex = pattern.lastIndex;
|
|
352
|
+
}
|
|
353
|
+
if (!matched) {
|
|
354
|
+
result.push(node);
|
|
355
|
+
continue;
|
|
356
|
+
}
|
|
357
|
+
if (lastIndex < node.text.length) {
|
|
358
|
+
result.push({ type: 'text', text: node.text.substring(lastIndex), formatting: node.formatting });
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
return result;
|
|
362
|
+
};
|
|
363
|
+
// Builds an admonition node from its raw body text, splitting on blank lines into
|
|
364
|
+
// paragraph children. v1 only supports inline content inside admonitions (no nested
|
|
365
|
+
// lists/headings/code) - acceptable per the roadmap's first cut.
|
|
366
|
+
const buildAdmonitionNode = (admonitionType, body) => {
|
|
367
|
+
const paragraphs = body.split(/\n\n+/).map(p => p.trim()).filter(Boolean);
|
|
368
|
+
const children = paragraphs.map(p => ({
|
|
369
|
+
type: 'paragraph',
|
|
370
|
+
children: parseInline(p.replace(/\n/g, ' '))
|
|
371
|
+
}));
|
|
372
|
+
return {
|
|
373
|
+
type: 'admonition',
|
|
374
|
+
metadata: { admonitionType },
|
|
375
|
+
children
|
|
376
|
+
};
|
|
141
377
|
};
|
|
142
378
|
const rawBlocks = textStr.split(/\n\n+/);
|
|
143
379
|
const blocks = [];
|
|
@@ -179,6 +415,21 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
179
415
|
block = block.trim();
|
|
180
416
|
if (!block)
|
|
181
417
|
continue;
|
|
418
|
+
// Standalone anchor-only block: one or more empty `<a name|id="…"></a>` tags on their
|
|
419
|
+
// own line (bookmark targets the MarkdownGenerator emits just before a heading/paragraph).
|
|
420
|
+
// Capture them as a placeholder so the post-loop pass can re-attach them to the following
|
|
421
|
+
// node's anchorIds — otherwise the tag-opening `<` is escaped and they render as visible text.
|
|
422
|
+
if (/^(?:\s*<a\s[^>]*>\s*<\/a>\s*)+$/i.test(block)) {
|
|
423
|
+
const anchorIds = [];
|
|
424
|
+
for (const m of block.matchAll(/<a\s[^>]*\b(?:name|id)="([^"]*)"/gi)) {
|
|
425
|
+
if (m[1])
|
|
426
|
+
anchorIds.push(m[1]);
|
|
427
|
+
}
|
|
428
|
+
if (anchorIds.length > 0) {
|
|
429
|
+
content.push({ type: ANCHOR_PLACEHOLDER, metadata: { anchorIds }, children: [] });
|
|
430
|
+
continue;
|
|
431
|
+
}
|
|
432
|
+
}
|
|
182
433
|
// Check for alignment wrapper start/end
|
|
183
434
|
const alignStartMatch = block.match(/^<div\s+(?:style="text-align:\s*(left|center|right|justify);?"|align="(left|center|right|justify)")>$/i);
|
|
184
435
|
if (alignStartMatch) {
|
|
@@ -196,6 +447,32 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
196
447
|
alignment = (alignMatch[1] || alignMatch[2]).toLowerCase();
|
|
197
448
|
block = alignMatch[3];
|
|
198
449
|
}
|
|
450
|
+
// YouTube embed fallback: MarkdownGenerator's 'embed' case emits a single-line
|
|
451
|
+
// <div data-youtube-video="ID" data-width="…" data-align="…"></div> when fallbackToHtml
|
|
452
|
+
// is on; recognise it here so a saved-then-reopened .md keeps the video.
|
|
453
|
+
const youtubeMatch = block.match(/^<div\s+data-youtube-video="([^"]*)"([^>]*)>\s*<\/div>$/i);
|
|
454
|
+
if (youtubeMatch) {
|
|
455
|
+
const videoId = youtubeMatch[1];
|
|
456
|
+
const attrsStr = youtubeMatch[2];
|
|
457
|
+
const widthMatch = attrsStr.match(/data-width="([^"]*)"/i);
|
|
458
|
+
const youtubeAlignMatch = attrsStr.match(/data-align="([^"]*)"/i);
|
|
459
|
+
const embedAlign = youtubeAlignMatch && ['left', 'center', 'right'].includes(youtubeAlignMatch[1]) ? youtubeAlignMatch[1] : undefined;
|
|
460
|
+
const embedUrl = videoId ? `https://www.youtube.com/watch?v=${videoId}` : undefined;
|
|
461
|
+
content.push({
|
|
462
|
+
type: 'embed',
|
|
463
|
+
// Childless nodes need .text so generic AST consumers (toText, chunking)
|
|
464
|
+
// don't silently drop them.
|
|
465
|
+
text: embedUrl,
|
|
466
|
+
metadata: {
|
|
467
|
+
embedType: 'youtube',
|
|
468
|
+
videoId,
|
|
469
|
+
url: embedUrl,
|
|
470
|
+
width: widthMatch?.[1],
|
|
471
|
+
align: embedAlign
|
|
472
|
+
}
|
|
473
|
+
});
|
|
474
|
+
continue;
|
|
475
|
+
}
|
|
199
476
|
// Code Block
|
|
200
477
|
const codeMatch = block.match(/^__CODE_BLOCK_(\d+)__$/);
|
|
201
478
|
if (codeMatch) {
|
|
@@ -207,6 +484,23 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
207
484
|
});
|
|
208
485
|
continue;
|
|
209
486
|
}
|
|
487
|
+
// GLFM-style fenced-div admonition, extracted to a placeholder above
|
|
488
|
+
const admonitionBlockMatch = block.match(/^__ADMONITION_(\d+)__$/);
|
|
489
|
+
if (admonitionBlockMatch) {
|
|
490
|
+
const data = JSON.parse(admonitionBlocks[parseInt(admonitionBlockMatch[1])]);
|
|
491
|
+
content.push(buildAdmonitionNode(data.admonitionType, data.body));
|
|
492
|
+
continue;
|
|
493
|
+
}
|
|
494
|
+
// Block math ($$...$$), extracted to a placeholder above
|
|
495
|
+
const mathBlockMatch = block.match(/^__MATH_BLOCK_(\d+)__$/);
|
|
496
|
+
if (mathBlockMatch) {
|
|
497
|
+
content.push({
|
|
498
|
+
type: 'code',
|
|
499
|
+
text: mathBlocks[parseInt(mathBlockMatch[1])],
|
|
500
|
+
metadata: { math: 'block' }
|
|
501
|
+
});
|
|
502
|
+
continue;
|
|
503
|
+
}
|
|
210
504
|
// Heading (allowing for leading HTML anchors and trailing {#anchor})
|
|
211
505
|
const headingMatch = block.match(/^((?:<a[^>]*><\/a>)*)\s*(#{1,6})\s+(.*?)(?:\s+\{#([^}]+)\})?\s*$/s);
|
|
212
506
|
if (headingMatch) {
|
|
@@ -215,7 +509,7 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
215
509
|
const explicitAnchor = headingMatch[4];
|
|
216
510
|
const anchorIds = [];
|
|
217
511
|
if (leadingAnchorsRaw) {
|
|
218
|
-
const idMatches = leadingAnchorsRaw.matchAll(/<a\s
|
|
512
|
+
const idMatches = leadingAnchorsRaw.matchAll(/<a\s[^>]*\b(?:name|id)="([^"]+)"/gi);
|
|
219
513
|
for (const m of idMatches)
|
|
220
514
|
anchorIds.push(m[1]);
|
|
221
515
|
}
|
|
@@ -237,10 +531,37 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
237
531
|
// Blockquote
|
|
238
532
|
const quoteMatch = block.match(/^>\s+(.*)$/s);
|
|
239
533
|
if (quoteMatch) {
|
|
534
|
+
// [ \t]? (not \s+) so a bare ">" paragraph-separator line (used between
|
|
535
|
+
// multi-paragraph admonition bodies) also dequotes to an empty line.
|
|
536
|
+
const dequoted = quoteMatch[1].replace(/^>[ \t]?/gm, '');
|
|
537
|
+
// GitHub-style admonition: `> [!NOTE]` on the first quoted line.
|
|
538
|
+
const admonitionHeaderMatch = dequoted.match(/^\[!(NOTE|TIP|IMPORTANT|WARNING|CAUTION)\]\s*\n?([\s\S]*)$/i);
|
|
539
|
+
if (admonitionHeaderMatch) {
|
|
540
|
+
const admonitionType = admonitionHeaderMatch[1].toLowerCase();
|
|
541
|
+
content.push(buildAdmonitionNode(admonitionType, admonitionHeaderMatch[2]));
|
|
542
|
+
continue;
|
|
543
|
+
}
|
|
240
544
|
content.push({
|
|
241
545
|
type: 'paragraph',
|
|
242
546
|
metadata: { style: 'Quote' },
|
|
243
|
-
children: parseInline(
|
|
547
|
+
children: parseInline(dequoted)
|
|
548
|
+
});
|
|
549
|
+
continue;
|
|
550
|
+
}
|
|
551
|
+
// Definition list (Markdown Extra / Pandoc / Kramdown): a term line followed by
|
|
552
|
+
// one or more ": definition" lines, e.g.:
|
|
553
|
+
// Term
|
|
554
|
+
// : Definition of the term.
|
|
555
|
+
const definitionListMatch = block.match(/^([^\n:][^\n]*)\n((?::[ \t]+.+(?:\n:[ \t]+.+)*))$/);
|
|
556
|
+
if (definitionListMatch) {
|
|
557
|
+
const term = definitionListMatch[1];
|
|
558
|
+
const definitions = definitionListMatch[2].split('\n').map(line => line.replace(/^:[ \t]+/, ''));
|
|
559
|
+
content.push({
|
|
560
|
+
type: 'definitionList',
|
|
561
|
+
children: [
|
|
562
|
+
{ type: 'definitionTerm', children: parseInline(term) },
|
|
563
|
+
...definitions.map(def => ({ type: 'definitionDescription', children: parseInline(def) }))
|
|
564
|
+
]
|
|
244
565
|
});
|
|
245
566
|
continue;
|
|
246
567
|
}
|
|
@@ -269,7 +590,16 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
269
590
|
else {
|
|
270
591
|
listCounters[level]++;
|
|
271
592
|
}
|
|
272
|
-
|
|
593
|
+
let itemText = match[3];
|
|
594
|
+
let isTask;
|
|
595
|
+
let checked;
|
|
596
|
+
const taskMatch = itemText.match(/^\[([ xX])\]\s+(.*)$/);
|
|
597
|
+
if (taskMatch) {
|
|
598
|
+
isTask = true;
|
|
599
|
+
checked = taskMatch[1].toLowerCase() === 'x';
|
|
600
|
+
itemText = taskMatch[2];
|
|
601
|
+
}
|
|
602
|
+
const children = parseInline(itemText);
|
|
273
603
|
content.push({
|
|
274
604
|
type: 'list',
|
|
275
605
|
text: children.map(c => c.text || '').join(''),
|
|
@@ -278,7 +608,9 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
278
608
|
indentation: level,
|
|
279
609
|
alignment: alignment || 'left',
|
|
280
610
|
listId,
|
|
281
|
-
itemIndex: listCounters[level]
|
|
611
|
+
itemIndex: listCounters[level],
|
|
612
|
+
isTask,
|
|
613
|
+
checked
|
|
282
614
|
},
|
|
283
615
|
children
|
|
284
616
|
});
|
|
@@ -288,8 +620,19 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
288
620
|
}
|
|
289
621
|
// Table (Simple Pipe or HTML)
|
|
290
622
|
if ((block.includes('|') && block.match(/\n\s*\|?[-:| ]+\|?\s*\n/)) || block.includes('<table')) {
|
|
623
|
+
// Pandoc-style trailing attribute list (`{align=right}`) immediately after the
|
|
624
|
+
// table, or Kramdown's `{: align=right}` on its own following line - both land
|
|
625
|
+
// in this same raw block since there's no blank line separating them.
|
|
626
|
+
let tableAlign;
|
|
627
|
+
const tableAttrLineMatch = block.match(/\n\{:?\s*([^}]*)\}\s*$/);
|
|
628
|
+
if (tableAttrLineMatch) {
|
|
629
|
+
tableAlign = parseAttributeList(tableAttrLineMatch[1]).align;
|
|
630
|
+
block = block.slice(0, tableAttrLineMatch.index);
|
|
631
|
+
}
|
|
291
632
|
if (block.includes('<table')) {
|
|
292
633
|
// Basic HTML table recognition (extracting rows/cells)
|
|
634
|
+
const tableTagMatch = block.match(/<table([^>]*)>/i);
|
|
635
|
+
const tableAlignMatch = tableTagMatch?.[1]?.match(/data-align=["']?(left|center|right)["']?/i);
|
|
293
636
|
const rows = [];
|
|
294
637
|
const trRegex = /<tr[^>]*>([\s\S]*?)<\/tr>/gi;
|
|
295
638
|
let trMatch;
|
|
@@ -314,8 +657,13 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
314
657
|
if (cells.length > 0)
|
|
315
658
|
rows.push({ type: 'row', children: cells });
|
|
316
659
|
}
|
|
660
|
+
const resolvedAlign = tableAlign || (tableAlignMatch ? tableAlignMatch[1].toLowerCase() : undefined);
|
|
317
661
|
if (rows.length > 0) {
|
|
318
|
-
content.push({
|
|
662
|
+
content.push({
|
|
663
|
+
type: 'table',
|
|
664
|
+
metadata: resolvedAlign ? { align: resolvedAlign } : undefined,
|
|
665
|
+
children: rows
|
|
666
|
+
});
|
|
319
667
|
continue;
|
|
320
668
|
}
|
|
321
669
|
}
|
|
@@ -326,13 +674,26 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
326
674
|
if (lines[i].match(/^\|?[-:| ]*---[-:| ]*\|?$/))
|
|
327
675
|
continue; // Separator row (requires at least one triple-hyphen)
|
|
328
676
|
const cellsStr = lines[i].replace(/^\||\|$/g, '').split('|');
|
|
329
|
-
const cells = cellsStr.map(c =>
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
677
|
+
const cells = cellsStr.map(c => {
|
|
678
|
+
// Recognize the MarkdownGenerator's own cell-alignment fallback,
|
|
679
|
+
// `<div style="text-align: X">…</div>`, and lift it into an aligned
|
|
680
|
+
// paragraph so it round-trips as alignment instead of being escaped to
|
|
681
|
+
// visible text on regeneration. Unwrap wherever it sits (e.g. inside **…**).
|
|
682
|
+
let cellText = c.trim();
|
|
683
|
+
let cellAlign;
|
|
684
|
+
cellText = cellText.replace(/<div\s+style="text-align:\s*(left|center|right|justify);?"\s*>([\s\S]*?)<\/div>/gi, (_m, a, inner) => { cellAlign = a.toLowerCase(); return inner; });
|
|
685
|
+
const inline = parseInline(cellText, i === 0 ? { bold: true } : {});
|
|
686
|
+
if (cellAlign && cellAlign !== 'left') {
|
|
687
|
+
return {
|
|
688
|
+
type: 'cell',
|
|
689
|
+
children: [{ type: 'paragraph', metadata: { alignment: cellAlign }, children: inline }]
|
|
690
|
+
};
|
|
691
|
+
}
|
|
692
|
+
return { type: 'cell', children: inline };
|
|
693
|
+
});
|
|
333
694
|
rows.push({ type: 'row', children: cells });
|
|
334
695
|
}
|
|
335
|
-
content.push({ type: 'table', children: rows });
|
|
696
|
+
content.push({ type: 'table', metadata: tableAlign ? { align: tableAlign } : undefined, children: rows });
|
|
336
697
|
continue;
|
|
337
698
|
}
|
|
338
699
|
}
|
|
@@ -348,14 +709,44 @@ const parseMarkdown = async (buffer, config) => {
|
|
|
348
709
|
children: parseInline(block.replace(/\n/g, ' '))
|
|
349
710
|
});
|
|
350
711
|
}
|
|
712
|
+
// Fold standalone anchor placeholders into the following content node's anchorIds so a
|
|
713
|
+
// bookmark target emitted on its own line round-trips as a real anchor. A trailing placeholder
|
|
714
|
+
// with no following node attaches to the previous node instead; if the document is nothing but
|
|
715
|
+
// anchors, they are dropped (there is no node to host them).
|
|
716
|
+
if (content.some(n => n.type === ANCHOR_PLACEHOLDER)) {
|
|
717
|
+
const merged = [];
|
|
718
|
+
let carried = [];
|
|
719
|
+
for (const node of content) {
|
|
720
|
+
if (node.type === ANCHOR_PLACEHOLDER) {
|
|
721
|
+
carried.push(...(node.metadata?.anchorIds || []));
|
|
722
|
+
continue;
|
|
723
|
+
}
|
|
724
|
+
if (carried.length > 0) {
|
|
725
|
+
const meta = node.metadata || (node.metadata = {});
|
|
726
|
+
meta.anchorIds = [...carried, ...(meta.anchorIds || [])];
|
|
727
|
+
carried = [];
|
|
728
|
+
}
|
|
729
|
+
merged.push(node);
|
|
730
|
+
}
|
|
731
|
+
if (carried.length > 0 && merged.length > 0) {
|
|
732
|
+
const last = merged[merged.length - 1].metadata || (merged[merged.length - 1].metadata = {});
|
|
733
|
+
last.anchorIds = [...(last.anchorIds || []), ...carried];
|
|
734
|
+
}
|
|
735
|
+
content.length = 0;
|
|
736
|
+
content.push(...merged);
|
|
737
|
+
}
|
|
351
738
|
const toTextSync = () => content.map(n => {
|
|
352
739
|
const getText = (node) => {
|
|
353
740
|
if (node.type === 'text' || node.type === 'code')
|
|
354
741
|
return node.text || '';
|
|
355
742
|
if (node.type === 'break')
|
|
356
743
|
return '\n';
|
|
744
|
+
// Childless nodes still carry meaningful text - fall back to it instead of
|
|
745
|
+
// silently vanishing from plain-text/RAG-chunk output.
|
|
746
|
+
if (node.type === 'embed')
|
|
747
|
+
return node.metadata?.url || '';
|
|
357
748
|
if (node.children) {
|
|
358
|
-
const isBlock = ['table', 'row', 'list', 'sheet', 'slide'].includes(node.type);
|
|
749
|
+
const isBlock = ['table', 'row', 'list', 'sheet', 'slide', 'admonition', 'definitionList'].includes(node.type);
|
|
359
750
|
return node.children.map(getText).join(isBlock ? config.newlineDelimiter : '');
|
|
360
751
|
}
|
|
361
752
|
return '';
|