extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,702 @@
|
|
|
1
|
+
// @ts-nocheck
|
|
2
|
+
/**
|
|
3
|
+
* @module research/extractor/url-to-content/docx-to-content
|
|
4
|
+
* @description Research library module.
|
|
5
|
+
*/
|
|
6
|
+
import JSZip from "jszip";
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Fetch wrapper for grabbing binary content
|
|
10
|
+
*/
|
|
11
|
+
async function grab(url: string, options: { responseType?: string; timeout?: number } = {}) {
|
|
12
|
+
const timeout = options.timeout ? options.timeout * 1000 : 10000;
|
|
13
|
+
const controller = new AbortController();
|
|
14
|
+
const timeoutId = setTimeout(() => controller.abort(), timeout);
|
|
15
|
+
|
|
16
|
+
try {
|
|
17
|
+
const response = await fetch(url, {
|
|
18
|
+
signal: controller.signal,
|
|
19
|
+
});
|
|
20
|
+
clearTimeout(timeoutId);
|
|
21
|
+
|
|
22
|
+
if (!response.ok) {
|
|
23
|
+
throw new Error(`HTTP ${response.status}`);
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
if (options.responseType === "arraybuffer") {
|
|
27
|
+
return await response.arrayBuffer();
|
|
28
|
+
}
|
|
29
|
+
return await response.text();
|
|
30
|
+
} catch (error) {
|
|
31
|
+
clearTimeout(timeoutId);
|
|
32
|
+
throw error;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Configuration options for DOCX parsing
|
|
38
|
+
* @typedef {Object} DocxOptions
|
|
39
|
+
* @property {boolean} [preserveShapes=true] - Whether to preserve shape elements
|
|
40
|
+
* @property {boolean} [includeStyles=true] - Whether to include document styles
|
|
41
|
+
* @property {string} [imgPath=''] - Base path for image resources
|
|
42
|
+
*/
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Style configuration for elements
|
|
46
|
+
* @typedef {Object} StyleConfig
|
|
47
|
+
* @property {boolean} block - If true, element is rendered as block
|
|
48
|
+
* @property {boolean} [heading] - If true, element is a heading
|
|
49
|
+
* @property {string} element - HTML element name
|
|
50
|
+
* @property {string} [xmlName] - DOCX XML element name
|
|
51
|
+
* @property {string} [class] - CSS class name
|
|
52
|
+
*/
|
|
53
|
+
|
|
54
|
+
const STYLE_MAP = {
|
|
55
|
+
paragraph: { block: true, element: "p" },
|
|
56
|
+
section: { block: true, element: "section" },
|
|
57
|
+
header: { block: true, element: "header" },
|
|
58
|
+
footer: { block: true, element: "footer" },
|
|
59
|
+
table: { block: true, element: "table" },
|
|
60
|
+
textbox: { block: true, element: "div", class: "textbox" },
|
|
61
|
+
h1: { block: true, heading: true, element: "h1", xmlName: "Heading1" },
|
|
62
|
+
h2: { block: true, heading: true, element: "h2", xmlName: "Heading2" },
|
|
63
|
+
text: { element: "span" },
|
|
64
|
+
del: { element: "del" },
|
|
65
|
+
strong: { element: "strong" },
|
|
66
|
+
};
|
|
67
|
+
|
|
68
|
+
const TABLE_STYLES = {
|
|
69
|
+
firstRow: "table-first-row",
|
|
70
|
+
lastRow: "table-last-row",
|
|
71
|
+
oddRow: "table-odd-row",
|
|
72
|
+
evenRow: "table-even-row",
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Converts a DOCX document to HTML
|
|
77
|
+
*
|
|
78
|
+
* @param {string|File|Blob|ArrayBuffer|Buffer|Uint8Array} input - DOCX input to convert
|
|
79
|
+
* @param {DocxOptions} [options] - Conversion options
|
|
80
|
+
* @returns {Promise<string>} The converted HTML
|
|
81
|
+
* @throws {Error} If conversion fails
|
|
82
|
+
* @category Extract
|
|
83
|
+
* @example
|
|
84
|
+
* const html = await convertDOCXToHTML('https://example.com/doc.docx');
|
|
85
|
+
* const html = await convertDOCXToHTML(fileInput.files[0]);
|
|
86
|
+
*/
|
|
87
|
+
export async function convertDOCXToHTML(input, options = {}) {
|
|
88
|
+
// Default options
|
|
89
|
+
const settings = {
|
|
90
|
+
preserveShapes: true,
|
|
91
|
+
includeStyles: true,
|
|
92
|
+
imgPath: "",
|
|
93
|
+
...options,
|
|
94
|
+
};
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Converts input to ArrayBuffer
|
|
98
|
+
* @param {string|File|Blob|ArrayBuffer|Buffer|Uint8Array} input
|
|
99
|
+
* @returns {Promise<ArrayBuffer>}
|
|
100
|
+
*/
|
|
101
|
+
async function getBuffer(input) {
|
|
102
|
+
if (input instanceof ArrayBuffer) {
|
|
103
|
+
return input;
|
|
104
|
+
}
|
|
105
|
+
if (input instanceof Uint8Array) {
|
|
106
|
+
return input.buffer.slice(
|
|
107
|
+
input.byteOffset,
|
|
108
|
+
input.byteOffset + input.byteLength,
|
|
109
|
+
);
|
|
110
|
+
}
|
|
111
|
+
if (typeof Buffer !== "undefined" && Buffer.isBuffer(input)) {
|
|
112
|
+
return input.buffer.slice(
|
|
113
|
+
input.byteOffset,
|
|
114
|
+
input.byteOffset + input.byteLength,
|
|
115
|
+
);
|
|
116
|
+
}
|
|
117
|
+
if (input instanceof Blob || input instanceof File) {
|
|
118
|
+
return await input.arrayBuffer();
|
|
119
|
+
}
|
|
120
|
+
if (typeof input === "string") {
|
|
121
|
+
return await grab(input, { responseType: "arraybuffer" });
|
|
122
|
+
}
|
|
123
|
+
throw new Error("Invalid input type");
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Extracts XML content from zip
|
|
128
|
+
* @param {JSZip} zip
|
|
129
|
+
* @param {string} path
|
|
130
|
+
* @returns {Promise<string>}
|
|
131
|
+
*/
|
|
132
|
+
async function extractXml(zip, path) {
|
|
133
|
+
const file = zip.file(path);
|
|
134
|
+
return file ? await file.async("string") : "";
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Parses document styles
|
|
139
|
+
* @param {string} xml
|
|
140
|
+
* @returns {Object}
|
|
141
|
+
*/
|
|
142
|
+
function parseStyles(xml) {
|
|
143
|
+
if (!xml) return {};
|
|
144
|
+
|
|
145
|
+
const styles = {
|
|
146
|
+
document: {},
|
|
147
|
+
paragraph: {},
|
|
148
|
+
character: {},
|
|
149
|
+
table: {},
|
|
150
|
+
};
|
|
151
|
+
|
|
152
|
+
// Parse default styles
|
|
153
|
+
const defaultMatch = /<w:docDefaults>[\s\S]*?<\/w:docDefaults>/i.exec(xml);
|
|
154
|
+
if (defaultMatch) {
|
|
155
|
+
const defaults = defaultMatch[0];
|
|
156
|
+
// Parse font, size, etc.
|
|
157
|
+
styles.document = {
|
|
158
|
+
fontFamily: /<w:rFonts[^>]*w:ascii="([^"]+)"/.exec(defaults)?.[1],
|
|
159
|
+
fontSize: /<w:sz[^>]*w:val="([^"]+)"/.exec(defaults)?.[1],
|
|
160
|
+
color: /<w:color[^>]*w:val="([^"]+)"/.exec(defaults)?.[1],
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
// Parse named styles
|
|
165
|
+
const styleRegex =
|
|
166
|
+
/<w:style\s+w:type="(\w+)"\s+w:styleId="([^"]+)"[^>]*>([\s\S]*?)<\/w:style>/gi;
|
|
167
|
+
let match;
|
|
168
|
+
while ((match = styleRegex.exec(xml)) !== null) {
|
|
169
|
+
const [_, type, id, content] = match;
|
|
170
|
+
if (styles[type.toLowerCase()]) {
|
|
171
|
+
styles[type.toLowerCase()][id] = parseStyleProperties(content);
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
return styles;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Parses style properties from XML content
|
|
180
|
+
* @param {string} content
|
|
181
|
+
* @returns {Object}
|
|
182
|
+
*/
|
|
183
|
+
function parseStyleProperties(content) {
|
|
184
|
+
return {
|
|
185
|
+
bold: /<w:b\/>/.test(content),
|
|
186
|
+
italic: /<w:i\/>/.test(content),
|
|
187
|
+
underline: /<w:u\/>/.test(content),
|
|
188
|
+
fontSize: /<w:sz[^>]*w:val="([^"]+)"/.exec(content)?.[1],
|
|
189
|
+
color: /<w:color[^>]*w:val="([^"]+)"/.exec(content)?.[1],
|
|
190
|
+
alignment: /<w:jc[^>]*w:val="([^"]+)"/.exec(content)?.[1],
|
|
191
|
+
};
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* Parses document content
|
|
196
|
+
* @param {string} xml
|
|
197
|
+
* @param {Object} context
|
|
198
|
+
* @returns {Array}
|
|
199
|
+
*/
|
|
200
|
+
function parseDocument(xml, context) {
|
|
201
|
+
const blocks = [];
|
|
202
|
+
|
|
203
|
+
// Parse sections
|
|
204
|
+
const sections = xml.split(/<w:sectPr[^>]*>[\s\S]*?<\/w:sectPr>/gi);
|
|
205
|
+
|
|
206
|
+
sections.forEach((section, index) => {
|
|
207
|
+
if (!section.trim()) return;
|
|
208
|
+
|
|
209
|
+
const content = [];
|
|
210
|
+
|
|
211
|
+
// Parse paragraphs
|
|
212
|
+
const pRegex = /<w:p\b[^>]*>[\s\S]*?<\/w:p>/gi;
|
|
213
|
+
let pMatch;
|
|
214
|
+
while ((pMatch = pRegex.exec(section)) !== null) {
|
|
215
|
+
const para = parseParagraph(pMatch[0], context);
|
|
216
|
+
if (para) content.push(para);
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
// Parse tables
|
|
220
|
+
const tblRegex = /<w:tbl\b[^>]*>[\s\S]*?<\/w:tbl>/gi;
|
|
221
|
+
let tblMatch;
|
|
222
|
+
// while ((tblMatch = tblRegex.exec(section)) !== null) {
|
|
223
|
+
// // const table = parseTable(tblMatch[0], context);
|
|
224
|
+
// if (table) content.push(table);
|
|
225
|
+
// }
|
|
226
|
+
|
|
227
|
+
blocks.push({
|
|
228
|
+
type: "section",
|
|
229
|
+
content,
|
|
230
|
+
});
|
|
231
|
+
});
|
|
232
|
+
|
|
233
|
+
return blocks;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
try {
|
|
237
|
+
const buffer = await getBuffer(input);
|
|
238
|
+
const zip = new JSZip();
|
|
239
|
+
const docx = await zip.loadAsync(buffer);
|
|
240
|
+
|
|
241
|
+
// Extract core XML files
|
|
242
|
+
const [docXml, stylesXml, numberingXml, relsXml] = await Promise.all([
|
|
243
|
+
extractXml(docx, "word/document.xml"),
|
|
244
|
+
extractXml(docx, "word/styles.xml"),
|
|
245
|
+
extractXml(docx, "word/numbering.xml"),
|
|
246
|
+
extractXml(docx, "word/_rels/document.xml.rels"),
|
|
247
|
+
]);
|
|
248
|
+
|
|
249
|
+
// Parse document structure
|
|
250
|
+
const styles = settings.includeStyles ? parseStyles(stylesXml) : {};
|
|
251
|
+
const content = parseDocument(docXml, { styles });
|
|
252
|
+
|
|
253
|
+
// Generate final HTML
|
|
254
|
+
return generateHtml(content, styles);
|
|
255
|
+
} catch (error) {
|
|
256
|
+
console.error("Error converting DOCX:", error);
|
|
257
|
+
throw error;
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/**
|
|
262
|
+
* @typedef {Object} ParagraphStyle
|
|
263
|
+
* @property {string} [alignment] - Text alignment (left, right, center, justify)
|
|
264
|
+
* @property {string} [spacing] - Line spacing
|
|
265
|
+
* @property {string} [indentation] - Paragraph indentation
|
|
266
|
+
* @property {boolean} [keepNext] - Keep with next paragraph
|
|
267
|
+
* @property {boolean} [pageBreakBefore] - Force page break before
|
|
268
|
+
*/
|
|
269
|
+
|
|
270
|
+
/**
|
|
271
|
+
* @typedef {Object} RunStyle
|
|
272
|
+
* @property {boolean} [bold] - Bold text
|
|
273
|
+
* @property {boolean} [italic] - Italic text
|
|
274
|
+
* @property {boolean} [underline] - Underlined text
|
|
275
|
+
* @property {string} [color] - Text color
|
|
276
|
+
* @property {string} [highlight] - Highlight color
|
|
277
|
+
* @property {string} [size] - Font size
|
|
278
|
+
* @property {string} [font] - Font family
|
|
279
|
+
*/
|
|
280
|
+
|
|
281
|
+
/**
|
|
282
|
+
* Parses a DOCX paragraph element into a structured object
|
|
283
|
+
* @param {string} xml - Paragraph XML string
|
|
284
|
+
* @param {Object} context - Document context containing styles and relationships
|
|
285
|
+
* @returns {Object|null} Parsed paragraph object or null if invalid
|
|
286
|
+
*/
|
|
287
|
+
function parseParagraph(xml, context) {
|
|
288
|
+
if (!xml || !xml.trim()) return null;
|
|
289
|
+
|
|
290
|
+
/**
|
|
291
|
+
* Extracts paragraph style properties
|
|
292
|
+
* @param {string} pPr - Style properties XML
|
|
293
|
+
* @returns {ParagraphStyle}
|
|
294
|
+
*/
|
|
295
|
+
function getParagraphStyle(pPr) {
|
|
296
|
+
if (!pPr) return {};
|
|
297
|
+
|
|
298
|
+
return {
|
|
299
|
+
alignment: /<w:jc\s+w:val="([^"]+)"/.exec(pPr)?.[1],
|
|
300
|
+
spacing: /<w:spacing\s+w:line="([^"]+)"/.exec(pPr)?.[1],
|
|
301
|
+
indentation: /<w:ind\s+w:left="([^"]+)"/.exec(pPr)?.[1],
|
|
302
|
+
keepNext: /<w:keepNext\s*\/>/.test(pPr),
|
|
303
|
+
pageBreakBefore: /<w:pageBreakBefore\s*\/>/.test(pPr),
|
|
304
|
+
styleId: /<w:pStyle\s+w:val="([^"]+)"/.exec(pPr)?.[1],
|
|
305
|
+
};
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
/**
|
|
309
|
+
* Extracts run (text span) style properties
|
|
310
|
+
* @param {string} rPr - Run properties XML
|
|
311
|
+
* @returns {RunStyle}
|
|
312
|
+
*/
|
|
313
|
+
function getRunStyle(rPr) {
|
|
314
|
+
if (!rPr) return {};
|
|
315
|
+
|
|
316
|
+
return {
|
|
317
|
+
bold: /<w:b\s*\/>/.test(rPr),
|
|
318
|
+
italic: /<w:i\s*\/>/.test(rPr),
|
|
319
|
+
underline: /<w:u\s*\/>/.test(rPr),
|
|
320
|
+
strike: /<w:strike\s*\/>/.test(rPr),
|
|
321
|
+
color: /<w:color\s+w:val="([^"]+)"/.exec(rPr)?.[1],
|
|
322
|
+
highlight: /<w:highlight\s+w:val="([^"]+)"/.exec(rPr)?.[1],
|
|
323
|
+
size: /<w:sz\s+w:val="([^"]+)"/.exec(rPr)?.[1],
|
|
324
|
+
font: /<w:rFonts[^>]*w:ascii="([^"]+)"/.exec(rPr)?.[1],
|
|
325
|
+
};
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
/**
|
|
329
|
+
* Processes text content
|
|
330
|
+
* @param {string} text - Text content
|
|
331
|
+
* @returns {string}
|
|
332
|
+
*/
|
|
333
|
+
function processText(text) {
|
|
334
|
+
return text
|
|
335
|
+
.replace(/&/g, "&")
|
|
336
|
+
.replace(/</g, "<")
|
|
337
|
+
.replace(/>/g, ">")
|
|
338
|
+
.replace(/\s+/g, " ")
|
|
339
|
+
.replace(/[\n\r]/g, " ");
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
try {
|
|
343
|
+
// Extract paragraph properties
|
|
344
|
+
const pPrMatch = /<w:pPr>([\s\S]*?)<\/w:pPr>/.exec(xml);
|
|
345
|
+
const paragraphStyle = getParagraphStyle(pPrMatch?.[1]);
|
|
346
|
+
|
|
347
|
+
// Extract and merge paragraph style from style definitions
|
|
348
|
+
const styleId = paragraphStyle.styleId;
|
|
349
|
+
if (styleId && context.styles?.paragraph?.[styleId]) {
|
|
350
|
+
Object.assign(paragraphStyle, context.styles.paragraph[styleId]);
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
// Parse runs (text spans)
|
|
354
|
+
const runs = [];
|
|
355
|
+
const runRegex = /<w:r\b[^>]*>([\s\S]*?)<\/w:r>/g;
|
|
356
|
+
let runMatch;
|
|
357
|
+
|
|
358
|
+
while ((runMatch = runRegex.exec(xml)) !== null) {
|
|
359
|
+
const runXml = runMatch[1];
|
|
360
|
+
|
|
361
|
+
// Extract run properties
|
|
362
|
+
const rPrMatch = /<w:rPr>([\s\S]*?)<\/w:rPr>/.exec(runXml);
|
|
363
|
+
const runStyle = getRunStyle(rPrMatch?.[1]);
|
|
364
|
+
|
|
365
|
+
// Extract text content
|
|
366
|
+
const textMatch = /<w:t\b[^>]*>([\s\S]*?)<\/w:t>/.exec(runXml);
|
|
367
|
+
if (textMatch) {
|
|
368
|
+
const text = processText(textMatch[1]);
|
|
369
|
+
if (text.trim()) {
|
|
370
|
+
runs.push({
|
|
371
|
+
type: "text",
|
|
372
|
+
text,
|
|
373
|
+
style: runStyle,
|
|
374
|
+
});
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
// Handle special elements
|
|
379
|
+
if (/<w:tab\/>/.test(runXml)) {
|
|
380
|
+
runs.push({ type: "tab" });
|
|
381
|
+
}
|
|
382
|
+
if (/<w:br\/>/.test(runXml)) {
|
|
383
|
+
runs.push({ type: "break" });
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
// Handle hyperlinks
|
|
387
|
+
const hyperlinkMatch = /<w:hyperlink\s+r:id="([^"]+)"/.exec(runXml);
|
|
388
|
+
if (hyperlinkMatch && context.relationships) {
|
|
389
|
+
const relationshipId = hyperlinkMatch[1];
|
|
390
|
+
const target = context.relationships[relationshipId];
|
|
391
|
+
if (target) {
|
|
392
|
+
runs.push({
|
|
393
|
+
type: "hyperlink",
|
|
394
|
+
target,
|
|
395
|
+
style: runStyle,
|
|
396
|
+
});
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
// Skip empty paragraphs unless they contain significant formatting
|
|
402
|
+
if (runs.length === 0 && !paragraphStyle.pageBreakBefore) {
|
|
403
|
+
return null;
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
return {
|
|
407
|
+
type: "paragraph",
|
|
408
|
+
style: paragraphStyle,
|
|
409
|
+
content: runs,
|
|
410
|
+
};
|
|
411
|
+
} catch (error) {
|
|
412
|
+
console.warn("Error parsing paragraph:", error);
|
|
413
|
+
return null;
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
/**
|
|
418
|
+
* Converts parsed DOCX content into HTML
|
|
419
|
+
* @param {Array} content - Array of parsed content blocks
|
|
420
|
+
* @param {Object} styles - Document style definitions
|
|
421
|
+
* @returns {string} Generated HTML
|
|
422
|
+
*/
|
|
423
|
+
function generateHtml(content, styles) {
|
|
424
|
+
/**
|
|
425
|
+
* Converts style object to CSS string
|
|
426
|
+
* @param {Object} style - Style properties
|
|
427
|
+
* @returns {string} CSS string
|
|
428
|
+
*/
|
|
429
|
+
function styleToCSS(style) {
|
|
430
|
+
if (!style) return "";
|
|
431
|
+
|
|
432
|
+
const cssMap = {
|
|
433
|
+
alignment: "text-align",
|
|
434
|
+
color: "color",
|
|
435
|
+
highlight: "background-color",
|
|
436
|
+
size: (value) => `font-size: ${parseInt(value) / 2}pt`,
|
|
437
|
+
spacing: (value) => `line-height: ${parseInt(value) / 240}`,
|
|
438
|
+
indentation: (value) => `margin-left: ${parseInt(value) / 20}pt`,
|
|
439
|
+
font: "font-family",
|
|
440
|
+
};
|
|
441
|
+
|
|
442
|
+
return Object.entries(style)
|
|
443
|
+
.map(([key, value]) => {
|
|
444
|
+
// Skip null/undefined values
|
|
445
|
+
if (value == null) return "";
|
|
446
|
+
|
|
447
|
+
// Handle boolean properties
|
|
448
|
+
if (key === "bold") return value ? "font-weight: bold" : "";
|
|
449
|
+
if (key === "italic") return value ? "font-style: italic" : "";
|
|
450
|
+
if (key === "underline")
|
|
451
|
+
return value ? "text-decoration: underline" : "";
|
|
452
|
+
if (key === "strike")
|
|
453
|
+
return value ? "text-decoration: line-through" : "";
|
|
454
|
+
|
|
455
|
+
// Handle mapped properties
|
|
456
|
+
const cssProperty = cssMap[key];
|
|
457
|
+
if (!cssProperty) return "";
|
|
458
|
+
|
|
459
|
+
if (typeof cssProperty === "function") {
|
|
460
|
+
return cssProperty(value);
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
return `${cssProperty}: ${value}`;
|
|
464
|
+
})
|
|
465
|
+
.filter(Boolean)
|
|
466
|
+
.join("; ");
|
|
467
|
+
}
|
|
468
|
+
|
|
469
|
+
/**
|
|
470
|
+
* Generates HTML for a text run
|
|
471
|
+
* @param {Object} run - Text run object
|
|
472
|
+
* @returns {string} HTML string
|
|
473
|
+
*/
|
|
474
|
+
function generateRunHtml(run) {
|
|
475
|
+
if (!run) return "";
|
|
476
|
+
|
|
477
|
+
switch (run.type) {
|
|
478
|
+
case "text": {
|
|
479
|
+
const style = styleToCSS(run.style);
|
|
480
|
+
return style ? `<span style="${style}">${run.text}</span>` : run.text;
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
case "tab":
|
|
484
|
+
return " ";
|
|
485
|
+
|
|
486
|
+
case "break":
|
|
487
|
+
return "<br>";
|
|
488
|
+
|
|
489
|
+
case "hyperlink": {
|
|
490
|
+
const style = styleToCSS(run.style);
|
|
491
|
+
return `<a href="${run.target}"${style ? ` style="${style}"` : ""}>${run.text || run.target}</a>`;
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
default:
|
|
495
|
+
return "";
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
/**
|
|
500
|
+
* Generates HTML for a paragraph
|
|
501
|
+
* @param {Object} paragraph - Paragraph object
|
|
502
|
+
* @returns {string} HTML string
|
|
503
|
+
*/
|
|
504
|
+
function generateParagraphHtml(paragraph) {
|
|
505
|
+
if (!paragraph?.content) return "";
|
|
506
|
+
|
|
507
|
+
const style = styleToCSS(paragraph.style);
|
|
508
|
+
const content = paragraph.content
|
|
509
|
+
.map((run) => generateRunHtml(run))
|
|
510
|
+
.join("");
|
|
511
|
+
|
|
512
|
+
// Handle special paragraph types based on style
|
|
513
|
+
const styleId = paragraph.style?.styleId;
|
|
514
|
+
if (styleId && styles?.paragraph?.[styleId]) {
|
|
515
|
+
const baseStyle = styles.paragraph[styleId];
|
|
516
|
+
|
|
517
|
+
// Convert headings
|
|
518
|
+
if (baseStyle.heading) {
|
|
519
|
+
const level = parseInt(styleId.match(/Heading(\d+)/)?.[1] || "1");
|
|
520
|
+
return `<h${level}${style ? ` style="${style}"` : ""}>${content}</h${level}>`;
|
|
521
|
+
}
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
// Force page break if specified
|
|
525
|
+
if (paragraph.style?.pageBreakBefore) {
|
|
526
|
+
return `<div style="page-break-before: always"></div><p${style ? ` style="${style}"` : ""}>${content}</p>`;
|
|
527
|
+
}
|
|
528
|
+
|
|
529
|
+
return `<p${style ? ` style="${style}"` : ""}>${content}</p>`;
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
/**
|
|
533
|
+
* Generates HTML for a table
|
|
534
|
+
* @param {Object} table - Table object
|
|
535
|
+
* @returns {string} HTML string
|
|
536
|
+
*/
|
|
537
|
+
function generateTableHtml(table) {
|
|
538
|
+
if (!table?.rows) return "";
|
|
539
|
+
|
|
540
|
+
const style = styleToCSS(table.style);
|
|
541
|
+
const rows = table.rows
|
|
542
|
+
.map((row, rowIndex) => {
|
|
543
|
+
const cells = row.cells
|
|
544
|
+
.map((cell, cellIndex) => {
|
|
545
|
+
const cellStyle = styleToCSS({
|
|
546
|
+
...cell.style,
|
|
547
|
+
width: cell.width ? `${cell.width}pt` : undefined,
|
|
548
|
+
});
|
|
549
|
+
|
|
550
|
+
const content = cell.content
|
|
551
|
+
.map((block) => {
|
|
552
|
+
switch (block.type) {
|
|
553
|
+
case "paragraph":
|
|
554
|
+
return generateParagraphHtml(block);
|
|
555
|
+
default:
|
|
556
|
+
return "";
|
|
557
|
+
}
|
|
558
|
+
})
|
|
559
|
+
.join("");
|
|
560
|
+
|
|
561
|
+
return `<td${cellStyle ? ` style="${cellStyle}"` : ""}>${content}</td>`;
|
|
562
|
+
})
|
|
563
|
+
.join("");
|
|
564
|
+
|
|
565
|
+
// Add row styles based on position
|
|
566
|
+
const rowClasses = [];
|
|
567
|
+
if (rowIndex === 0 && table.style?.firstRow)
|
|
568
|
+
rowClasses.push(TABLE_STYLES.firstRow);
|
|
569
|
+
if (rowIndex === table.rows.length - 1 && table.style?.lastRow)
|
|
570
|
+
rowClasses.push(TABLE_STYLES.lastRow);
|
|
571
|
+
if (rowIndex % 2 === 0) rowClasses.push(TABLE_STYLES.evenRow);
|
|
572
|
+
else rowClasses.push(TABLE_STYLES.oddRow);
|
|
573
|
+
|
|
574
|
+
return `<tr${rowClasses.length ? ` class="${rowClasses.join(" ")}"` : ""}>${cells}</tr>`;
|
|
575
|
+
})
|
|
576
|
+
.join("");
|
|
577
|
+
|
|
578
|
+
return `<table${style ? ` style="${style}"` : ""}>${rows}</table>`;
|
|
579
|
+
}
|
|
580
|
+
|
|
581
|
+
/**
|
|
582
|
+
* Generates HTML for a section
|
|
583
|
+
* @param {Object} section - Section object
|
|
584
|
+
* @returns {string} HTML string
|
|
585
|
+
*/
|
|
586
|
+
function generateSectionHtml(section) {
|
|
587
|
+
if (!section?.content) return "";
|
|
588
|
+
|
|
589
|
+
const blocks = section.content
|
|
590
|
+
.map((block) => {
|
|
591
|
+
switch (block.type) {
|
|
592
|
+
case "paragraph":
|
|
593
|
+
return generateParagraphHtml(block);
|
|
594
|
+
case "table":
|
|
595
|
+
return generateTableHtml(block);
|
|
596
|
+
default:
|
|
597
|
+
return "";
|
|
598
|
+
}
|
|
599
|
+
})
|
|
600
|
+
.filter(Boolean)
|
|
601
|
+
.join("\n");
|
|
602
|
+
|
|
603
|
+
const style = styleToCSS(section.style);
|
|
604
|
+
return `<section${style ? ` style="${style}"` : ""}>${blocks}</section>`;
|
|
605
|
+
}
|
|
606
|
+
|
|
607
|
+
// Generate document-level styles
|
|
608
|
+
let css = "";
|
|
609
|
+
if (styles?.document) {
|
|
610
|
+
const documentStyle = styleToCSS(styles.document);
|
|
611
|
+
if (documentStyle) {
|
|
612
|
+
css = `<style>
|
|
613
|
+
body {
|
|
614
|
+
${documentStyle}
|
|
615
|
+
}
|
|
616
|
+
${Object.entries(TABLE_STYLES)
|
|
617
|
+
.map(
|
|
618
|
+
([key, className]) => `
|
|
619
|
+
.${className} {
|
|
620
|
+
${styles.table?.[key] ? styleToCSS(styles.table[key]) : ""}
|
|
621
|
+
}
|
|
622
|
+
`,
|
|
623
|
+
)
|
|
624
|
+
.join("\n")}
|
|
625
|
+
</style>`;
|
|
626
|
+
}
|
|
627
|
+
}
|
|
628
|
+
|
|
629
|
+
// Generate content HTML
|
|
630
|
+
const bodyContent = content
|
|
631
|
+
.map((block) => {
|
|
632
|
+
switch (block.type) {
|
|
633
|
+
case "section":
|
|
634
|
+
return generateSectionHtml(block);
|
|
635
|
+
case "paragraph":
|
|
636
|
+
return generateParagraphHtml(block);
|
|
637
|
+
case "table":
|
|
638
|
+
return generateTableHtml(block);
|
|
639
|
+
default:
|
|
640
|
+
return "";
|
|
641
|
+
}
|
|
642
|
+
})
|
|
643
|
+
.filter(Boolean)
|
|
644
|
+
.join("\n");
|
|
645
|
+
|
|
646
|
+
return `<!DOCTYPE html>
|
|
647
|
+
<html>
|
|
648
|
+
<head>
|
|
649
|
+
<meta charset="UTF-8">
|
|
650
|
+
${css}
|
|
651
|
+
</head>
|
|
652
|
+
<body>
|
|
653
|
+
${bodyContent}
|
|
654
|
+
</body>
|
|
655
|
+
</html>`;
|
|
656
|
+
}
|
|
657
|
+
|
|
658
|
+
/**
|
|
659
|
+
* Detects if a binary buffer is a DOCX file by checking the file signature
|
|
660
|
+
* DOCX files are ZIP archives with specific internal structure
|
|
661
|
+
*
|
|
662
|
+
* @param {ArrayBuffer|Buffer|Uint8Array} buffer - Binary buffer to check
|
|
663
|
+
* @returns {boolean} True if buffer appears to be a DOCX file
|
|
664
|
+
* @category Extract
|
|
665
|
+
*/
|
|
666
|
+
export function isBufferDOCX(buffer) {
|
|
667
|
+
if (!buffer) return false;
|
|
668
|
+
|
|
669
|
+
try {
|
|
670
|
+
// Convert to Uint8Array for consistent access
|
|
671
|
+
const uint8Array =
|
|
672
|
+
buffer instanceof Uint8Array ? buffer : new Uint8Array(buffer);
|
|
673
|
+
|
|
674
|
+
// Check minimum length (DOCX files are ZIP archives, need at least ZIP header)
|
|
675
|
+
if (uint8Array.length < 30) return false;
|
|
676
|
+
|
|
677
|
+
// Check ZIP file signature (PK header)
|
|
678
|
+
// ZIP files start with "PK" (0x504B)
|
|
679
|
+
if (uint8Array[0] !== 0x50 || uint8Array[1] !== 0x4b) return false;
|
|
680
|
+
|
|
681
|
+
// Check if it's a ZIP file (central directory or local file header)
|
|
682
|
+
const signature = (uint8Array[2] << 8) | uint8Array[3];
|
|
683
|
+
if (signature !== 0x0304 && signature !== 0x0201) return false;
|
|
684
|
+
|
|
685
|
+
// For DOCX, we need to check if it contains the required DOCX structure
|
|
686
|
+
// This is a more thorough check that looks for DOCX-specific files
|
|
687
|
+
const bufferString = new TextDecoder("utf-8", { fatal: false }).decode(
|
|
688
|
+
uint8Array.slice(0, Math.min(1024, uint8Array.length)),
|
|
689
|
+
);
|
|
690
|
+
|
|
691
|
+
// Look for DOCX-specific markers in the ZIP structure
|
|
692
|
+
// DOCX files should contain references to word/document.xml
|
|
693
|
+
return (
|
|
694
|
+
bufferString.includes("word/document.xml") ||
|
|
695
|
+
bufferString.includes("word/styles.xml") ||
|
|
696
|
+
bufferString.includes("[Content_Types].xml")
|
|
697
|
+
);
|
|
698
|
+
} catch (error) {
|
|
699
|
+
// If we can't parse the buffer, assume it's not a DOCX
|
|
700
|
+
return false;
|
|
701
|
+
}
|
|
702
|
+
}
|