extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,702 @@
1
+ // @ts-nocheck
2
+ /**
3
+ * @module research/extractor/url-to-content/docx-to-content
4
+ * @description Research library module.
5
+ */
6
+ import JSZip from "jszip";
7
+
8
+ /**
9
+ * Fetch wrapper for grabbing binary content
10
+ */
11
+ async function grab(url: string, options: { responseType?: string; timeout?: number } = {}) {
12
+ const timeout = options.timeout ? options.timeout * 1000 : 10000;
13
+ const controller = new AbortController();
14
+ const timeoutId = setTimeout(() => controller.abort(), timeout);
15
+
16
+ try {
17
+ const response = await fetch(url, {
18
+ signal: controller.signal,
19
+ });
20
+ clearTimeout(timeoutId);
21
+
22
+ if (!response.ok) {
23
+ throw new Error(`HTTP ${response.status}`);
24
+ }
25
+
26
+ if (options.responseType === "arraybuffer") {
27
+ return await response.arrayBuffer();
28
+ }
29
+ return await response.text();
30
+ } catch (error) {
31
+ clearTimeout(timeoutId);
32
+ throw error;
33
+ }
34
+ }
35
+
36
+ /**
37
+ * Configuration options for DOCX parsing
38
+ * @typedef {Object} DocxOptions
39
+ * @property {boolean} [preserveShapes=true] - Whether to preserve shape elements
40
+ * @property {boolean} [includeStyles=true] - Whether to include document styles
41
+ * @property {string} [imgPath=''] - Base path for image resources
42
+ */
43
+
44
+ /**
45
+ * Style configuration for elements
46
+ * @typedef {Object} StyleConfig
47
+ * @property {boolean} block - If true, element is rendered as block
48
+ * @property {boolean} [heading] - If true, element is a heading
49
+ * @property {string} element - HTML element name
50
+ * @property {string} [xmlName] - DOCX XML element name
51
+ * @property {string} [class] - CSS class name
52
+ */
53
+
54
+ const STYLE_MAP = {
55
+ paragraph: { block: true, element: "p" },
56
+ section: { block: true, element: "section" },
57
+ header: { block: true, element: "header" },
58
+ footer: { block: true, element: "footer" },
59
+ table: { block: true, element: "table" },
60
+ textbox: { block: true, element: "div", class: "textbox" },
61
+ h1: { block: true, heading: true, element: "h1", xmlName: "Heading1" },
62
+ h2: { block: true, heading: true, element: "h2", xmlName: "Heading2" },
63
+ text: { element: "span" },
64
+ del: { element: "del" },
65
+ strong: { element: "strong" },
66
+ };
67
+
68
+ const TABLE_STYLES = {
69
+ firstRow: "table-first-row",
70
+ lastRow: "table-last-row",
71
+ oddRow: "table-odd-row",
72
+ evenRow: "table-even-row",
73
+ };
74
+
75
+ /**
76
+ * Converts a DOCX document to HTML
77
+ *
78
+ * @param {string|File|Blob|ArrayBuffer|Buffer|Uint8Array} input - DOCX input to convert
79
+ * @param {DocxOptions} [options] - Conversion options
80
+ * @returns {Promise<string>} The converted HTML
81
+ * @throws {Error} If conversion fails
82
+ * @category Extract
83
+ * @example
84
+ * const html = await convertDOCXToHTML('https://example.com/doc.docx');
85
+ * const html = await convertDOCXToHTML(fileInput.files[0]);
86
+ */
87
+ export async function convertDOCXToHTML(input, options = {}) {
88
+ // Default options
89
+ const settings = {
90
+ preserveShapes: true,
91
+ includeStyles: true,
92
+ imgPath: "",
93
+ ...options,
94
+ };
95
+
96
+ /**
97
+ * Converts input to ArrayBuffer
98
+ * @param {string|File|Blob|ArrayBuffer|Buffer|Uint8Array} input
99
+ * @returns {Promise<ArrayBuffer>}
100
+ */
101
+ async function getBuffer(input) {
102
+ if (input instanceof ArrayBuffer) {
103
+ return input;
104
+ }
105
+ if (input instanceof Uint8Array) {
106
+ return input.buffer.slice(
107
+ input.byteOffset,
108
+ input.byteOffset + input.byteLength,
109
+ );
110
+ }
111
+ if (typeof Buffer !== "undefined" && Buffer.isBuffer(input)) {
112
+ return input.buffer.slice(
113
+ input.byteOffset,
114
+ input.byteOffset + input.byteLength,
115
+ );
116
+ }
117
+ if (input instanceof Blob || input instanceof File) {
118
+ return await input.arrayBuffer();
119
+ }
120
+ if (typeof input === "string") {
121
+ return await grab(input, { responseType: "arraybuffer" });
122
+ }
123
+ throw new Error("Invalid input type");
124
+ }
125
+
126
+ /**
127
+ * Extracts XML content from zip
128
+ * @param {JSZip} zip
129
+ * @param {string} path
130
+ * @returns {Promise<string>}
131
+ */
132
+ async function extractXml(zip, path) {
133
+ const file = zip.file(path);
134
+ return file ? await file.async("string") : "";
135
+ }
136
+
137
+ /**
138
+ * Parses document styles
139
+ * @param {string} xml
140
+ * @returns {Object}
141
+ */
142
+ function parseStyles(xml) {
143
+ if (!xml) return {};
144
+
145
+ const styles = {
146
+ document: {},
147
+ paragraph: {},
148
+ character: {},
149
+ table: {},
150
+ };
151
+
152
+ // Parse default styles
153
+ const defaultMatch = /<w:docDefaults>[\s\S]*?<\/w:docDefaults>/i.exec(xml);
154
+ if (defaultMatch) {
155
+ const defaults = defaultMatch[0];
156
+ // Parse font, size, etc.
157
+ styles.document = {
158
+ fontFamily: /<w:rFonts[^>]*w:ascii="([^"]+)"/.exec(defaults)?.[1],
159
+ fontSize: /<w:sz[^>]*w:val="([^"]+)"/.exec(defaults)?.[1],
160
+ color: /<w:color[^>]*w:val="([^"]+)"/.exec(defaults)?.[1],
161
+ };
162
+ }
163
+
164
+ // Parse named styles
165
+ const styleRegex =
166
+ /<w:style\s+w:type="(\w+)"\s+w:styleId="([^"]+)"[^>]*>([\s\S]*?)<\/w:style>/gi;
167
+ let match;
168
+ while ((match = styleRegex.exec(xml)) !== null) {
169
+ const [_, type, id, content] = match;
170
+ if (styles[type.toLowerCase()]) {
171
+ styles[type.toLowerCase()][id] = parseStyleProperties(content);
172
+ }
173
+ }
174
+
175
+ return styles;
176
+ }
177
+
178
+ /**
179
+ * Parses style properties from XML content
180
+ * @param {string} content
181
+ * @returns {Object}
182
+ */
183
+ function parseStyleProperties(content) {
184
+ return {
185
+ bold: /<w:b\/>/.test(content),
186
+ italic: /<w:i\/>/.test(content),
187
+ underline: /<w:u\/>/.test(content),
188
+ fontSize: /<w:sz[^>]*w:val="([^"]+)"/.exec(content)?.[1],
189
+ color: /<w:color[^>]*w:val="([^"]+)"/.exec(content)?.[1],
190
+ alignment: /<w:jc[^>]*w:val="([^"]+)"/.exec(content)?.[1],
191
+ };
192
+ }
193
+
194
+ /**
195
+ * Parses document content
196
+ * @param {string} xml
197
+ * @param {Object} context
198
+ * @returns {Array}
199
+ */
200
+ function parseDocument(xml, context) {
201
+ const blocks = [];
202
+
203
+ // Parse sections
204
+ const sections = xml.split(/<w:sectPr[^>]*>[\s\S]*?<\/w:sectPr>/gi);
205
+
206
+ sections.forEach((section, index) => {
207
+ if (!section.trim()) return;
208
+
209
+ const content = [];
210
+
211
+ // Parse paragraphs
212
+ const pRegex = /<w:p\b[^>]*>[\s\S]*?<\/w:p>/gi;
213
+ let pMatch;
214
+ while ((pMatch = pRegex.exec(section)) !== null) {
215
+ const para = parseParagraph(pMatch[0], context);
216
+ if (para) content.push(para);
217
+ }
218
+
219
+ // Parse tables
220
+ const tblRegex = /<w:tbl\b[^>]*>[\s\S]*?<\/w:tbl>/gi;
221
+ let tblMatch;
222
+ // while ((tblMatch = tblRegex.exec(section)) !== null) {
223
+ // // const table = parseTable(tblMatch[0], context);
224
+ // if (table) content.push(table);
225
+ // }
226
+
227
+ blocks.push({
228
+ type: "section",
229
+ content,
230
+ });
231
+ });
232
+
233
+ return blocks;
234
+ }
235
+
236
+ try {
237
+ const buffer = await getBuffer(input);
238
+ const zip = new JSZip();
239
+ const docx = await zip.loadAsync(buffer);
240
+
241
+ // Extract core XML files
242
+ const [docXml, stylesXml, numberingXml, relsXml] = await Promise.all([
243
+ extractXml(docx, "word/document.xml"),
244
+ extractXml(docx, "word/styles.xml"),
245
+ extractXml(docx, "word/numbering.xml"),
246
+ extractXml(docx, "word/_rels/document.xml.rels"),
247
+ ]);
248
+
249
+ // Parse document structure
250
+ const styles = settings.includeStyles ? parseStyles(stylesXml) : {};
251
+ const content = parseDocument(docXml, { styles });
252
+
253
+ // Generate final HTML
254
+ return generateHtml(content, styles);
255
+ } catch (error) {
256
+ console.error("Error converting DOCX:", error);
257
+ throw error;
258
+ }
259
+ }
260
+
261
+ /**
262
+ * @typedef {Object} ParagraphStyle
263
+ * @property {string} [alignment] - Text alignment (left, right, center, justify)
264
+ * @property {string} [spacing] - Line spacing
265
+ * @property {string} [indentation] - Paragraph indentation
266
+ * @property {boolean} [keepNext] - Keep with next paragraph
267
+ * @property {boolean} [pageBreakBefore] - Force page break before
268
+ */
269
+
270
+ /**
271
+ * @typedef {Object} RunStyle
272
+ * @property {boolean} [bold] - Bold text
273
+ * @property {boolean} [italic] - Italic text
274
+ * @property {boolean} [underline] - Underlined text
275
+ * @property {string} [color] - Text color
276
+ * @property {string} [highlight] - Highlight color
277
+ * @property {string} [size] - Font size
278
+ * @property {string} [font] - Font family
279
+ */
280
+
281
+ /**
282
+ * Parses a DOCX paragraph element into a structured object
283
+ * @param {string} xml - Paragraph XML string
284
+ * @param {Object} context - Document context containing styles and relationships
285
+ * @returns {Object|null} Parsed paragraph object or null if invalid
286
+ */
287
+ function parseParagraph(xml, context) {
288
+ if (!xml || !xml.trim()) return null;
289
+
290
+ /**
291
+ * Extracts paragraph style properties
292
+ * @param {string} pPr - Style properties XML
293
+ * @returns {ParagraphStyle}
294
+ */
295
+ function getParagraphStyle(pPr) {
296
+ if (!pPr) return {};
297
+
298
+ return {
299
+ alignment: /<w:jc\s+w:val="([^"]+)"/.exec(pPr)?.[1],
300
+ spacing: /<w:spacing\s+w:line="([^"]+)"/.exec(pPr)?.[1],
301
+ indentation: /<w:ind\s+w:left="([^"]+)"/.exec(pPr)?.[1],
302
+ keepNext: /<w:keepNext\s*\/>/.test(pPr),
303
+ pageBreakBefore: /<w:pageBreakBefore\s*\/>/.test(pPr),
304
+ styleId: /<w:pStyle\s+w:val="([^"]+)"/.exec(pPr)?.[1],
305
+ };
306
+ }
307
+
308
+ /**
309
+ * Extracts run (text span) style properties
310
+ * @param {string} rPr - Run properties XML
311
+ * @returns {RunStyle}
312
+ */
313
+ function getRunStyle(rPr) {
314
+ if (!rPr) return {};
315
+
316
+ return {
317
+ bold: /<w:b\s*\/>/.test(rPr),
318
+ italic: /<w:i\s*\/>/.test(rPr),
319
+ underline: /<w:u\s*\/>/.test(rPr),
320
+ strike: /<w:strike\s*\/>/.test(rPr),
321
+ color: /<w:color\s+w:val="([^"]+)"/.exec(rPr)?.[1],
322
+ highlight: /<w:highlight\s+w:val="([^"]+)"/.exec(rPr)?.[1],
323
+ size: /<w:sz\s+w:val="([^"]+)"/.exec(rPr)?.[1],
324
+ font: /<w:rFonts[^>]*w:ascii="([^"]+)"/.exec(rPr)?.[1],
325
+ };
326
+ }
327
+
328
+ /**
329
+ * Processes text content
330
+ * @param {string} text - Text content
331
+ * @returns {string}
332
+ */
333
+ function processText(text) {
334
+ return text
335
+ .replace(/&/g, "&amp;")
336
+ .replace(/</g, "&lt;")
337
+ .replace(/>/g, "&gt;")
338
+ .replace(/\s+/g, " ")
339
+ .replace(/[\n\r]/g, " ");
340
+ }
341
+
342
+ try {
343
+ // Extract paragraph properties
344
+ const pPrMatch = /<w:pPr>([\s\S]*?)<\/w:pPr>/.exec(xml);
345
+ const paragraphStyle = getParagraphStyle(pPrMatch?.[1]);
346
+
347
+ // Extract and merge paragraph style from style definitions
348
+ const styleId = paragraphStyle.styleId;
349
+ if (styleId && context.styles?.paragraph?.[styleId]) {
350
+ Object.assign(paragraphStyle, context.styles.paragraph[styleId]);
351
+ }
352
+
353
+ // Parse runs (text spans)
354
+ const runs = [];
355
+ const runRegex = /<w:r\b[^>]*>([\s\S]*?)<\/w:r>/g;
356
+ let runMatch;
357
+
358
+ while ((runMatch = runRegex.exec(xml)) !== null) {
359
+ const runXml = runMatch[1];
360
+
361
+ // Extract run properties
362
+ const rPrMatch = /<w:rPr>([\s\S]*?)<\/w:rPr>/.exec(runXml);
363
+ const runStyle = getRunStyle(rPrMatch?.[1]);
364
+
365
+ // Extract text content
366
+ const textMatch = /<w:t\b[^>]*>([\s\S]*?)<\/w:t>/.exec(runXml);
367
+ if (textMatch) {
368
+ const text = processText(textMatch[1]);
369
+ if (text.trim()) {
370
+ runs.push({
371
+ type: "text",
372
+ text,
373
+ style: runStyle,
374
+ });
375
+ }
376
+ }
377
+
378
+ // Handle special elements
379
+ if (/<w:tab\/>/.test(runXml)) {
380
+ runs.push({ type: "tab" });
381
+ }
382
+ if (/<w:br\/>/.test(runXml)) {
383
+ runs.push({ type: "break" });
384
+ }
385
+
386
+ // Handle hyperlinks
387
+ const hyperlinkMatch = /<w:hyperlink\s+r:id="([^"]+)"/.exec(runXml);
388
+ if (hyperlinkMatch && context.relationships) {
389
+ const relationshipId = hyperlinkMatch[1];
390
+ const target = context.relationships[relationshipId];
391
+ if (target) {
392
+ runs.push({
393
+ type: "hyperlink",
394
+ target,
395
+ style: runStyle,
396
+ });
397
+ }
398
+ }
399
+ }
400
+
401
+ // Skip empty paragraphs unless they contain significant formatting
402
+ if (runs.length === 0 && !paragraphStyle.pageBreakBefore) {
403
+ return null;
404
+ }
405
+
406
+ return {
407
+ type: "paragraph",
408
+ style: paragraphStyle,
409
+ content: runs,
410
+ };
411
+ } catch (error) {
412
+ console.warn("Error parsing paragraph:", error);
413
+ return null;
414
+ }
415
+ }
416
+
417
+ /**
418
+ * Converts parsed DOCX content into HTML
419
+ * @param {Array} content - Array of parsed content blocks
420
+ * @param {Object} styles - Document style definitions
421
+ * @returns {string} Generated HTML
422
+ */
423
+ function generateHtml(content, styles) {
424
+ /**
425
+ * Converts style object to CSS string
426
+ * @param {Object} style - Style properties
427
+ * @returns {string} CSS string
428
+ */
429
+ function styleToCSS(style) {
430
+ if (!style) return "";
431
+
432
+ const cssMap = {
433
+ alignment: "text-align",
434
+ color: "color",
435
+ highlight: "background-color",
436
+ size: (value) => `font-size: ${parseInt(value) / 2}pt`,
437
+ spacing: (value) => `line-height: ${parseInt(value) / 240}`,
438
+ indentation: (value) => `margin-left: ${parseInt(value) / 20}pt`,
439
+ font: "font-family",
440
+ };
441
+
442
+ return Object.entries(style)
443
+ .map(([key, value]) => {
444
+ // Skip null/undefined values
445
+ if (value == null) return "";
446
+
447
+ // Handle boolean properties
448
+ if (key === "bold") return value ? "font-weight: bold" : "";
449
+ if (key === "italic") return value ? "font-style: italic" : "";
450
+ if (key === "underline")
451
+ return value ? "text-decoration: underline" : "";
452
+ if (key === "strike")
453
+ return value ? "text-decoration: line-through" : "";
454
+
455
+ // Handle mapped properties
456
+ const cssProperty = cssMap[key];
457
+ if (!cssProperty) return "";
458
+
459
+ if (typeof cssProperty === "function") {
460
+ return cssProperty(value);
461
+ }
462
+
463
+ return `${cssProperty}: ${value}`;
464
+ })
465
+ .filter(Boolean)
466
+ .join("; ");
467
+ }
468
+
469
+ /**
470
+ * Generates HTML for a text run
471
+ * @param {Object} run - Text run object
472
+ * @returns {string} HTML string
473
+ */
474
+ function generateRunHtml(run) {
475
+ if (!run) return "";
476
+
477
+ switch (run.type) {
478
+ case "text": {
479
+ const style = styleToCSS(run.style);
480
+ return style ? `<span style="${style}">${run.text}</span>` : run.text;
481
+ }
482
+
483
+ case "tab":
484
+ return "&nbsp;&nbsp;&nbsp;&nbsp;";
485
+
486
+ case "break":
487
+ return "<br>";
488
+
489
+ case "hyperlink": {
490
+ const style = styleToCSS(run.style);
491
+ return `<a href="${run.target}"${style ? ` style="${style}"` : ""}>${run.text || run.target}</a>`;
492
+ }
493
+
494
+ default:
495
+ return "";
496
+ }
497
+ }
498
+
499
+ /**
500
+ * Generates HTML for a paragraph
501
+ * @param {Object} paragraph - Paragraph object
502
+ * @returns {string} HTML string
503
+ */
504
+ function generateParagraphHtml(paragraph) {
505
+ if (!paragraph?.content) return "";
506
+
507
+ const style = styleToCSS(paragraph.style);
508
+ const content = paragraph.content
509
+ .map((run) => generateRunHtml(run))
510
+ .join("");
511
+
512
+ // Handle special paragraph types based on style
513
+ const styleId = paragraph.style?.styleId;
514
+ if (styleId && styles?.paragraph?.[styleId]) {
515
+ const baseStyle = styles.paragraph[styleId];
516
+
517
+ // Convert headings
518
+ if (baseStyle.heading) {
519
+ const level = parseInt(styleId.match(/Heading(\d+)/)?.[1] || "1");
520
+ return `<h${level}${style ? ` style="${style}"` : ""}>${content}</h${level}>`;
521
+ }
522
+ }
523
+
524
+ // Force page break if specified
525
+ if (paragraph.style?.pageBreakBefore) {
526
+ return `<div style="page-break-before: always"></div><p${style ? ` style="${style}"` : ""}>${content}</p>`;
527
+ }
528
+
529
+ return `<p${style ? ` style="${style}"` : ""}>${content}</p>`;
530
+ }
531
+
532
+ /**
533
+ * Generates HTML for a table
534
+ * @param {Object} table - Table object
535
+ * @returns {string} HTML string
536
+ */
537
+ function generateTableHtml(table) {
538
+ if (!table?.rows) return "";
539
+
540
+ const style = styleToCSS(table.style);
541
+ const rows = table.rows
542
+ .map((row, rowIndex) => {
543
+ const cells = row.cells
544
+ .map((cell, cellIndex) => {
545
+ const cellStyle = styleToCSS({
546
+ ...cell.style,
547
+ width: cell.width ? `${cell.width}pt` : undefined,
548
+ });
549
+
550
+ const content = cell.content
551
+ .map((block) => {
552
+ switch (block.type) {
553
+ case "paragraph":
554
+ return generateParagraphHtml(block);
555
+ default:
556
+ return "";
557
+ }
558
+ })
559
+ .join("");
560
+
561
+ return `<td${cellStyle ? ` style="${cellStyle}"` : ""}>${content}</td>`;
562
+ })
563
+ .join("");
564
+
565
+ // Add row styles based on position
566
+ const rowClasses = [];
567
+ if (rowIndex === 0 && table.style?.firstRow)
568
+ rowClasses.push(TABLE_STYLES.firstRow);
569
+ if (rowIndex === table.rows.length - 1 && table.style?.lastRow)
570
+ rowClasses.push(TABLE_STYLES.lastRow);
571
+ if (rowIndex % 2 === 0) rowClasses.push(TABLE_STYLES.evenRow);
572
+ else rowClasses.push(TABLE_STYLES.oddRow);
573
+
574
+ return `<tr${rowClasses.length ? ` class="${rowClasses.join(" ")}"` : ""}>${cells}</tr>`;
575
+ })
576
+ .join("");
577
+
578
+ return `<table${style ? ` style="${style}"` : ""}>${rows}</table>`;
579
+ }
580
+
581
+ /**
582
+ * Generates HTML for a section
583
+ * @param {Object} section - Section object
584
+ * @returns {string} HTML string
585
+ */
586
+ function generateSectionHtml(section) {
587
+ if (!section?.content) return "";
588
+
589
+ const blocks = section.content
590
+ .map((block) => {
591
+ switch (block.type) {
592
+ case "paragraph":
593
+ return generateParagraphHtml(block);
594
+ case "table":
595
+ return generateTableHtml(block);
596
+ default:
597
+ return "";
598
+ }
599
+ })
600
+ .filter(Boolean)
601
+ .join("\n");
602
+
603
+ const style = styleToCSS(section.style);
604
+ return `<section${style ? ` style="${style}"` : ""}>${blocks}</section>`;
605
+ }
606
+
607
+ // Generate document-level styles
608
+ let css = "";
609
+ if (styles?.document) {
610
+ const documentStyle = styleToCSS(styles.document);
611
+ if (documentStyle) {
612
+ css = `<style>
613
+ body {
614
+ ${documentStyle}
615
+ }
616
+ ${Object.entries(TABLE_STYLES)
617
+ .map(
618
+ ([key, className]) => `
619
+ .${className} {
620
+ ${styles.table?.[key] ? styleToCSS(styles.table[key]) : ""}
621
+ }
622
+ `,
623
+ )
624
+ .join("\n")}
625
+ </style>`;
626
+ }
627
+ }
628
+
629
+ // Generate content HTML
630
+ const bodyContent = content
631
+ .map((block) => {
632
+ switch (block.type) {
633
+ case "section":
634
+ return generateSectionHtml(block);
635
+ case "paragraph":
636
+ return generateParagraphHtml(block);
637
+ case "table":
638
+ return generateTableHtml(block);
639
+ default:
640
+ return "";
641
+ }
642
+ })
643
+ .filter(Boolean)
644
+ .join("\n");
645
+
646
+ return `<!DOCTYPE html>
647
+ <html>
648
+ <head>
649
+ <meta charset="UTF-8">
650
+ ${css}
651
+ </head>
652
+ <body>
653
+ ${bodyContent}
654
+ </body>
655
+ </html>`;
656
+ }
657
+
658
+ /**
659
+ * Detects if a binary buffer is a DOCX file by checking the file signature
660
+ * DOCX files are ZIP archives with specific internal structure
661
+ *
662
+ * @param {ArrayBuffer|Buffer|Uint8Array} buffer - Binary buffer to check
663
+ * @returns {boolean} True if buffer appears to be a DOCX file
664
+ * @category Extract
665
+ */
666
+ export function isBufferDOCX(buffer) {
667
+ if (!buffer) return false;
668
+
669
+ try {
670
+ // Convert to Uint8Array for consistent access
671
+ const uint8Array =
672
+ buffer instanceof Uint8Array ? buffer : new Uint8Array(buffer);
673
+
674
+ // Check minimum length (DOCX files are ZIP archives, need at least ZIP header)
675
+ if (uint8Array.length < 30) return false;
676
+
677
+ // Check ZIP file signature (PK header)
678
+ // ZIP files start with "PK" (0x504B)
679
+ if (uint8Array[0] !== 0x50 || uint8Array[1] !== 0x4b) return false;
680
+
681
+ // Check if it's a ZIP file (central directory or local file header)
682
+ const signature = (uint8Array[2] << 8) | uint8Array[3];
683
+ if (signature !== 0x0304 && signature !== 0x0201) return false;
684
+
685
+ // For DOCX, we need to check if it contains the required DOCX structure
686
+ // This is a more thorough check that looks for DOCX-specific files
687
+ const bufferString = new TextDecoder("utf-8", { fatal: false }).decode(
688
+ uint8Array.slice(0, Math.min(1024, uint8Array.length)),
689
+ );
690
+
691
+ // Look for DOCX-specific markers in the ZIP structure
692
+ // DOCX files should contain references to word/document.xml
693
+ return (
694
+ bufferString.includes("word/document.xml") ||
695
+ bufferString.includes("word/styles.xml") ||
696
+ bufferString.includes("[Content_Types].xml")
697
+ );
698
+ } catch (error) {
699
+ // If we can't parse the buffer, assume it's not a DOCX
700
+ return false;
701
+ }
702
+ }