@bevel-software/platform-core-backend 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/THIRD-PARTY-NOTICES.md +1163 -425
  2. package/dist/core/create-core-server.js +1 -1
  3. package/dist/core/create-core-server.js.map +1 -1
  4. package/dist/core/create-core-services.d.ts +2 -0
  5. package/dist/core/create-core-services.d.ts.map +1 -1
  6. package/dist/core/create-core-services.js +5 -0
  7. package/dist/core/create-core-services.js.map +1 -1
  8. package/dist/core-config.d.ts +7 -0
  9. package/dist/core-config.d.ts.map +1 -1
  10. package/dist/core-config.js +9 -0
  11. package/dist/core-config.js.map +1 -1
  12. package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
  13. package/dist/modules/code-mode/code-mode.tool.js +7 -1
  14. package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
  15. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
  16. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
  17. package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
  18. package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
  19. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
  20. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
  21. package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
  22. package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
  23. package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
  24. package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
  25. package/dist/modules/workspace/file-readers/document-reader.js +59 -0
  26. package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
  27. package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
  28. package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
  29. package/dist/modules/workspace/file-readers/email-reader.js +19 -0
  30. package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
  31. package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
  32. package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
  33. package/dist/modules/workspace/file-readers/email-text.js +151 -0
  34. package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
  35. package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
  36. package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
  37. package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
  38. package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
  39. package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
  40. package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
  41. package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
  42. package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
  43. package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
  44. package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
  45. package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
  46. package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
  47. package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
  48. package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
  49. package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
  50. package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
  51. package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
  52. package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
  53. package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
  54. package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
  55. package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
  56. package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
  57. package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
  58. package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
  59. package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
  60. package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
  61. package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
  62. package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
  63. package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
  64. package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
  65. package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
  66. package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
  67. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
  68. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
  69. package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
  70. package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
  71. package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
  72. package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
  73. package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
  74. package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
  75. package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
  76. package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
  77. package/dist/modules/workspace/file-readers/file-reader.js +55 -0
  78. package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
  79. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
  80. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
  81. package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
  82. package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
  83. package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
  84. package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
  85. package/dist/modules/workspace/file-readers/image-read.js +108 -0
  86. package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
  87. package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
  88. package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
  89. package/dist/modules/workspace/file-readers/image-reader.js +30 -0
  90. package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
  91. package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
  92. package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
  93. package/dist/modules/workspace/file-readers/odf-text.js +116 -0
  94. package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
  95. package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
  96. package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
  97. package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
  98. package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
  99. package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
  100. package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
  101. package/dist/modules/workspace/file-readers/text-reader.js +117 -0
  102. package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
  103. package/dist/modules/workspace/workspace.tools.d.ts +2 -1
  104. package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
  105. package/dist/modules/workspace/workspace.tools.js +158 -15
  106. package/dist/modules/workspace/workspace.tools.js.map +1 -1
  107. package/package.json +9 -4
  108. package/src/core/create-core-server.ts +1 -1
  109. package/src/core/create-core-services.ts +6 -0
  110. package/src/core-config.ts +9 -0
  111. package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
  112. package/src/modules/code-mode/code-mode.tool.ts +7 -1
  113. package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
  114. package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
  115. package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
  116. package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
  117. package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
  118. package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
  119. package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
  120. package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
  121. package/src/modules/workspace/file-readers/document-reader.ts +64 -0
  122. package/src/modules/workspace/file-readers/email-reader.ts +21 -0
  123. package/src/modules/workspace/file-readers/email-text.ts +193 -0
  124. package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
  125. package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
  126. package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
  127. package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
  128. package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
  129. package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
  130. package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
  131. package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
  132. package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
  133. package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
  134. package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
  135. package/src/modules/workspace/file-readers/file-reader.ts +104 -0
  136. package/src/modules/workspace/file-readers/image-read.ts +122 -0
  137. package/src/modules/workspace/file-readers/image-reader.ts +39 -0
  138. package/src/modules/workspace/file-readers/odf-text.ts +123 -0
  139. package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
  140. package/src/modules/workspace/file-readers/text-reader.ts +131 -0
  141. package/src/modules/workspace/workspace.tools.ts +174 -12
@@ -0,0 +1,1658 @@
1
+ import { mkdtemp, readdir, readFile, rm, stat, writeFile } from 'node:fs/promises';
2
+ import { tmpdir } from 'node:os';
3
+ import { join } from 'node:path';
4
+ import AdmZip from 'adm-zip';
5
+ import * as XLSX from 'xlsx';
6
+ import { afterEach, beforeEach, describe, expect, it } from 'vitest';
7
+ import { extractDocx } from '../extract-docx.js';
8
+ import { extractPptx } from '../extract-pptx.js';
9
+ import { extractXlsx } from '../extract-xlsx.js';
10
+ import { extractPdf } from '../extract-pdf.js';
11
+ import { extractOdt } from '../extract-odt.js';
12
+ import { extractOdp } from '../extract-odp.js';
13
+ import { extractOds } from '../extract-ods.js';
14
+ import { DocExtractService, EXTRACTION_SCHEMA } from '../doc-extract.service.js';
15
+ import { DocExtractionCache, gitBlobSha } from '../extraction-cache.js';
16
+ import { fileExtension } from '../doc-extract.types.js';
17
+ import { MAX_ODF_NS_ALIASES, normalizeOdfPrefixes, odfParagraphBlocks } from '../odf-text.js';
18
+ import { decodeXmlEntities, xmlAttrValue, xmlAttrValueByLocalName } from '../ooxml-text.js';
19
+ import { notesTargetFromRels } from '../extract-pptx.js';
20
+
21
+ /**
22
+ * REAL fixtures, no mocks: docx/pptx are handcrafted OOXML zips built with
23
+ * adm-zip in-test, xlsx comes out of SheetJS itself, and the PDF is a minimal
24
+ * one-page document assembled byte-for-byte (valid xref included).
25
+ */
26
+
27
+ // ── fixture builders ───────────────────────────────────────────────────────
28
+
29
+ const W_NS = 'xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"';
30
+ const A_NS = 'xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main"';
31
+
32
+ function docxBytes(bodyXml: string): Buffer {
33
+ const zip = new AdmZip();
34
+ zip.addFile(
35
+ '[Content_Types].xml',
36
+ Buffer.from(
37
+ '<?xml version="1.0"?><Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">' +
38
+ '<Default Extension="xml" ContentType="application/xml"/>' +
39
+ '<Override PartName="/word/document.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"/></Types>',
40
+ ),
41
+ );
42
+ zip.addFile(
43
+ 'word/document.xml',
44
+ Buffer.from(`<?xml version="1.0"?><w:document ${W_NS}><w:body>${bodyXml}</w:body></w:document>`),
45
+ );
46
+ return zip.toBuffer();
47
+ }
48
+
49
+ const para = (...runs: string[]): string =>
50
+ `<w:p>${runs.map((r) => `<w:r><w:t>${r}</w:t></w:r>`).join('')}</w:p>`;
51
+
52
+ function slideXml(...paragraphs: string[][]): string {
53
+ const ps = paragraphs
54
+ .map((runs) => `<a:p>${runs.map((r) => `<a:r><a:t>${r}</a:t></a:r>`).join('')}</a:p>`)
55
+ .join('');
56
+ return `<?xml version="1.0"?><p:sld xmlns:p="http://schemas.openxmlformats.org/presentationml/2006/main" ${A_NS}><p:txBody>${ps}</p:txBody></p:sld>`;
57
+ }
58
+
59
+ function pptxBytes(
60
+ slides: Record<number, string>,
61
+ notes: Record<number, string> = {},
62
+ extra: Record<string, string> = {},
63
+ ): Buffer {
64
+ const zip = new AdmZip();
65
+ zip.addFile('[Content_Types].xml', Buffer.from('<?xml version="1.0"?><Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"/>'));
66
+ for (const [n, xml] of Object.entries(slides)) zip.addFile(`ppt/slides/slide${n}.xml`, Buffer.from(xml));
67
+ for (const [n, xml] of Object.entries(notes)) zip.addFile(`ppt/notesSlides/notesSlide${n}.xml`, Buffer.from(xml));
68
+ for (const [name, xml] of Object.entries(extra)) zip.addFile(name, Buffer.from(xml));
69
+ return zip.toBuffer();
70
+ }
71
+
72
+ function xlsxBytes(sheets: Record<string, unknown[][]>): Buffer {
73
+ const wb = XLSX.utils.book_new();
74
+ for (const [name, rows] of Object.entries(sheets)) {
75
+ XLSX.utils.book_append_sheet(wb, XLSX.utils.aoa_to_sheet(rows), name);
76
+ }
77
+ return XLSX.write(wb, { type: 'buffer', bookType: 'xlsx' }) as Buffer;
78
+ }
79
+
80
+ /** A minimal but VALID one-page PDF whose text layer is `text` ('' = no text layer). */
81
+ export function pdfBytes(text: string): Buffer {
82
+ return pdfWithContentStream(text === '' ? '' : `BT /F1 12 Tf 72 720 Td (${text}) Tj ET`);
83
+ }
84
+
85
+ /** The same one-page shell, but the test dictates the page's raw content stream. */
86
+ function pdfWithContentStream(stream: string): Buffer {
87
+ const objects = [
88
+ '<< /Type /Catalog /Pages 2 0 R >>',
89
+ '<< /Type /Pages /Kids [3 0 R] /Count 1 >>',
90
+ '<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>',
91
+ `<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`,
92
+ '<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>',
93
+ ];
94
+ let pdf = '%PDF-1.4\n';
95
+ const offsets: number[] = [];
96
+ for (let i = 0; i < objects.length; i++) {
97
+ offsets.push(pdf.length);
98
+ pdf += `${i + 1} 0 obj\n${objects[i]}\nendobj\n`;
99
+ }
100
+ const xrefStart = pdf.length;
101
+ pdf += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
102
+ for (const off of offsets) pdf += `${String(off).padStart(10, '0')} 00000 n \n`;
103
+ pdf += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xrefStart}\n%%EOF\n`;
104
+ return Buffer.from(pdf, 'latin1');
105
+ }
106
+
107
+ // ── ODF fixture builders ───────────────────────────────────────────────────
108
+
109
+ const ODF_NS =
110
+ 'xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0" ' +
111
+ 'xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0" ' +
112
+ 'xmlns:table="urn:oasis:names:tc:opendocument:xmlns:table:1.0" ' +
113
+ 'xmlns:draw="urn:oasis:names:tc:opendocument:xmlns:drawing:1.0" ' +
114
+ 'xmlns:presentation="urn:oasis:names:tc:opendocument:xmlns:presentation:1.0"';
115
+
116
+ function odfBytes(mimetype: string, bodyXml: string): Buffer {
117
+ const zip = new AdmZip();
118
+ zip.addFile('mimetype', Buffer.from(mimetype));
119
+ zip.addFile(
120
+ 'content.xml',
121
+ Buffer.from(`<?xml version="1.0"?><office:document-content ${ODF_NS}><office:body>${bodyXml}</office:body></office:document-content>`),
122
+ );
123
+ return zip.toBuffer();
124
+ }
125
+
126
+ /** An ODF package whose ROOT attributes and whole content tree the test dictates. */
127
+ function odfBytesRaw(mimetype: string, rootAttrs: string, contentXml: string): Buffer {
128
+ const zip = new AdmZip();
129
+ zip.addFile('mimetype', Buffer.from(mimetype));
130
+ zip.addFile(
131
+ 'content.xml',
132
+ Buffer.from(`<?xml version="1.0"?><office:document-content ${ODF_NS} ${rootAttrs}>${contentXml}</office:document-content>`),
133
+ );
134
+ return zip.toBuffer();
135
+ }
136
+
137
+ const odtBytes = (textXml: string): Buffer =>
138
+ odfBytes('application/vnd.oasis.opendocument.text', `<office:text>${textXml}</office:text>`);
139
+
140
+ /** One odp slide: frame paragraphs + optional notes paragraphs. */
141
+ const odpPage = (paragraphs: string[], notes: string[] = []): string =>
142
+ '<draw:page draw:name="page">' +
143
+ `<draw:frame><draw:text-box>${paragraphs.map((p) => `<text:p>${p}</text:p>`).join('')}</draw:text-box></draw:frame>` +
144
+ (notes.length > 0
145
+ ? `<presentation:notes><draw:frame><draw:text-box>${notes.map((p) => `<text:p>${p}</text:p>`).join('')}</draw:text-box></draw:frame></presentation:notes>`
146
+ : '') +
147
+ '</draw:page>';
148
+
149
+ const odpBytes = (...pages: string[]): Buffer =>
150
+ odfBytes('application/vnd.oasis.opendocument.presentation', `<office:presentation>${pages.join('')}</office:presentation>`);
151
+
152
+ const odsBytes = (...tables: string[]): Buffer =>
153
+ odfBytes('application/vnd.oasis.opendocument.spreadsheet', `<office:spreadsheet>${tables.join('')}</office:spreadsheet>`);
154
+
155
+ const odsCell = (text: string, repeat?: number): string =>
156
+ `<table:table-cell${repeat ? ` table:number-columns-repeated="${repeat}"` : ''}>${text === '' ? '' : `<text:p>${text}</text:p>`}</table:table-cell>`;
157
+
158
+ // ── extension parsing ──────────────────────────────────────────────────────
159
+ // (Which extension belongs to which reader lives in the registry now — see
160
+ // file-reader.registry.test.ts for the routing, case-insensitivity included.)
161
+
162
+ describe('fileExtension', () => {
163
+ it('lowercases and ignores dot-files', () => {
164
+ expect(fileExtension('A/B/C.DocX')).toBe('.docx');
165
+ expect(fileExtension('dir/.gitignore')).toBe('');
166
+ });
167
+ });
168
+
169
+ // ── docx ───────────────────────────────────────────────────────────────────
170
+
171
+ describe('extractDocx', () => {
172
+ it('joins split runs with NO separator and emits paragraphs as lines', () => {
173
+ const res = extractDocx(docxBytes(para('Hel', 'lo world') + para('Second paragraph')));
174
+ if (!res.ok) throw new Error(res.message);
175
+ expect(res.text.split('\n')).toEqual(['Hello world', 'Second paragraph']);
176
+ expect(res.summary).toContain('2 paragraphs');
177
+ expect(res.summary).toContain('headers/footers skipped');
178
+ });
179
+
180
+ it('decodes named and numeric XML entities', () => {
181
+ const res = extractDocx(docxBytes(para('&amp; &lt;tag&gt; &quot;q&quot; &apos;a&apos; &#65;&#x42;')));
182
+ if (!res.ok) throw new Error(res.message);
183
+ expect(res.text).toBe('& <tag> "q" \'a\' AB');
184
+ });
185
+
186
+ it('renders table rows as tab-separated cell text, without double-counting their paragraphs', () => {
187
+ const table =
188
+ '<w:tbl><w:tblPr/>' +
189
+ `<w:tr><w:tc>${para('A1')}</w:tc><w:tc>${para('B1')}</w:tc></w:tr>` +
190
+ `<w:tr><w:tc>${para('A2')}</w:tc><w:tc>${para('B2')}</w:tc></w:tr>` +
191
+ '</w:tbl>';
192
+ const res = extractDocx(docxBytes(para('Before') + table + para('After')));
193
+ if (!res.ok) throw new Error(res.message);
194
+ expect(res.text.split('\n')).toEqual(['Before', 'A1\tB1', 'A2\tB2', 'After']);
195
+ expect(res.summary).toContain('1 table');
196
+ });
197
+
198
+ it('returns a typed failure for bytes that are not a zip', () => {
199
+ const res = extractDocx(Buffer.from('this is not a docx at all'));
200
+ expect(res.ok).toBe(false);
201
+ if (res.ok) return;
202
+ expect(res.message).toContain('could not be parsed as a .docx');
203
+ });
204
+
205
+ it('returns a typed failure for a zip without word/document.xml', () => {
206
+ const zip = new AdmZip();
207
+ zip.addFile('hello.txt', Buffer.from('hi'));
208
+ const res = extractDocx(zip.toBuffer());
209
+ expect(res.ok).toBe(false);
210
+ if (res.ok) return;
211
+ expect(res.message).toContain('no word/document.xml');
212
+ });
213
+ });
214
+
215
+ // ── pptx ───────────────────────────────────────────────────────────────────
216
+
217
+ describe('extractPptx', () => {
218
+ it('emits [slide N] markers in NUMERIC order with per-paragraph lines and notes', () => {
219
+ const res = extractPptx(
220
+ pptxBytes(
221
+ {
222
+ 2: slideXml(['Second slide']),
223
+ 10: slideXml(['Tenth slide']),
224
+ 1: slideXml(['Road', 'map 2026'], ['Bullet one']),
225
+ },
226
+ { 1: slideXml(['Remember the demo']) },
227
+ ),
228
+ );
229
+ if (!res.ok) throw new Error(res.message);
230
+ expect(res.text.split('\n')).toEqual([
231
+ '[slide 1]',
232
+ 'Roadmap 2026',
233
+ 'Bullet one',
234
+ '[slide 1 notes]',
235
+ 'Remember the demo',
236
+ '[slide 2]',
237
+ 'Second slide',
238
+ '[slide 10]',
239
+ 'Tenth slide',
240
+ ]);
241
+ expect(res.summary).toContain('3 slides + notes');
242
+ });
243
+
244
+ it('omits the notes marker when the notes part has no text, and "+ notes" from the summary', () => {
245
+ const res = extractPptx(pptxBytes({ 1: slideXml(['Only slide']) }, { 1: slideXml([]) }));
246
+ if (!res.ok) throw new Error(res.message);
247
+ expect(res.text).toBe('[slide 1]\nOnly slide');
248
+ expect(res.summary).toContain('1 slide;');
249
+ });
250
+
251
+ it('returns a typed failure for a zip with no slides', () => {
252
+ const zip = new AdmZip();
253
+ zip.addFile('ppt/presentation.xml', Buffer.from('<p/>'));
254
+ const res = extractPptx(zip.toBuffer());
255
+ expect(res.ok).toBe(false);
256
+ if (res.ok) return;
257
+ expect(res.message).toContain('could not be parsed as a .pptx');
258
+ });
259
+ });
260
+
261
+ // ── xlsx ───────────────────────────────────────────────────────────────────
262
+
263
+ describe('extractXlsx', () => {
264
+ it('emits [sheet: Name] markers and rows as tab-separated values', () => {
265
+ const res = extractXlsx(
266
+ xlsxBytes({
267
+ Inventory: [
268
+ ['Name', 'Qty'],
269
+ ['Widget', 3],
270
+ ],
271
+ Empty: [[]],
272
+ }),
273
+ );
274
+ if (!res.ok) throw new Error(res.message);
275
+ const lines = res.text.split('\n');
276
+ expect(lines[0]).toBe('[sheet: Inventory]');
277
+ expect(lines[1]).toBe('Name\tQty');
278
+ expect(lines[2]).toBe('Widget\t3');
279
+ expect(lines).toContain('[sheet: Empty]');
280
+ expect(res.summary).toContain('2 sheets');
281
+ });
282
+
283
+ it('refuses non-zip bytes instead of letting SheetJS misread them as CSV', () => {
284
+ const res = extractXlsx(Buffer.from('a,b,c\n1,2,3\n'));
285
+ expect(res.ok).toBe(false);
286
+ if (res.ok) return;
287
+ expect(res.message).toContain('not a zip archive');
288
+ });
289
+
290
+ it('a sheet name holding a tab or newline cannot split the [sheet: …] marker line', () => {
291
+ // Excel's own UI forbids these, but the name is an XML attribute in a
292
+ // crafted workbook — SheetJS round-trips `&#10;` intact. Same rule as the
293
+ // ods extractor: control separators become spaces so grep line numbers hold.
294
+ const res = extractXlsx(xlsxBytes({ 'bad\nname\ttab': [['x']] }));
295
+ if (!res.ok) throw new Error(res.message);
296
+ const lines = res.text.split('\n');
297
+ expect(lines[0]).toBe('[sheet: bad name tab]');
298
+ expect(lines[1]).toBe('x');
299
+ });
300
+ });
301
+
302
+ // ── pdf ────────────────────────────────────────────────────────────────────
303
+
304
+ describe('extractPdf', () => {
305
+ it('extracts the text layer with [page N] markers', async () => {
306
+ const res = await extractPdf(pdfBytes('Hello PDF world'));
307
+ if (!res.ok) throw new Error(res.message);
308
+ expect(res.text.split('\n')).toEqual(['[page 1]', 'Hello PDF world']);
309
+ expect(res.summary).toContain('1 page');
310
+ });
311
+
312
+ it('says "no text layer" for a page without one (scan-style PDF)', async () => {
313
+ const res = await extractPdf(pdfBytes(''));
314
+ if (!res.ok) throw new Error(res.message);
315
+ expect(res.summary).toContain('no text layer (scanned document?)');
316
+ expect(res.text).toBe('[page 1]');
317
+ });
318
+
319
+ it('returns a typed failure for bytes that are not a PDF', async () => {
320
+ const res = await extractPdf(Buffer.from('definitely not a pdf'));
321
+ expect(res.ok).toBe(false);
322
+ if (res.ok) return;
323
+ expect(res.message).toContain('could not be parsed as a PDF');
324
+ });
325
+
326
+ it('joins same-line items with spaces and breaks lines on a Y jump (items arrive streamed)', async () => {
327
+ // Three separate show-text ops: two on one baseline, one 40pt lower. The
328
+ // extractor consumes them through `streamTextContent` now, so this guards
329
+ // the item walk (space joins, Y-jump line breaks) across chunk boundaries.
330
+ const res = await extractPdf(
331
+ pdfWithContentStream('BT /F1 12 Tf 72 720 Td (Alpha) Tj 60 0 Td (beta) Tj -60 -40 Td (Gamma) Tj ET'),
332
+ );
333
+ if (!res.ok) throw new Error(res.message);
334
+ const lines = res.text.split('\n');
335
+ expect(lines).toHaveLength(3);
336
+ expect(lines[0]).toBe('[page 1]');
337
+ // pdf.js may model the gap as its own whitespace item — spacing WIDTH is
338
+ // its call, but the items must land on one line, space-separated.
339
+ expect(lines[1]).toMatch(/^Alpha +beta$/);
340
+ expect(lines[2]).toBe('Gamma');
341
+ });
342
+ });
343
+
344
+ // ── odt ────────────────────────────────────────────────────────────────────
345
+
346
+ describe('extractOdt', () => {
347
+ it('joins spans with NO separator; headings and paragraphs are lines in document order', () => {
348
+ const res = extractOdt(
349
+ odtBytes(
350
+ '<text:h text:outline-level="1">Ti<text:span text:style-name="T1">tle</text:span></text:h>' +
351
+ '<text:p>Hel<text:span text:style-name="T2">lo world</text:span></text:p>',
352
+ ),
353
+ );
354
+ if (!res.ok) throw new Error(res.message);
355
+ expect(res.text.split('\n')).toEqual(['Title', 'Hello world']);
356
+ expect(res.summary).toBe('2 paragraphs; layout, images and formatting omitted');
357
+ });
358
+
359
+ it('renders <text:tab/>, <text:line-break/> and <text:s text:c="N"/> as real characters', () => {
360
+ const res = extractOdt(odtBytes('<text:p>a<text:tab/>b<text:line-break/>c<text:s text:c="3"/>d<text:s/>e</text:p>'));
361
+ if (!res.ok) throw new Error(res.message);
362
+ expect(res.text).toBe('a\tb\nc d e');
363
+ });
364
+
365
+ it('decodes named and numeric XML entities', () => {
366
+ const res = extractOdt(odtBytes('<text:p>&amp; &lt;tag&gt; &quot;q&quot; &apos;a&apos; &#65;&#x42;</text:p>'));
367
+ if (!res.ok) throw new Error(res.message);
368
+ expect(res.text).toBe('& <tag> "q" \'a\' AB');
369
+ });
370
+
371
+ it('returns a typed failure for bytes that are not a zip', () => {
372
+ const res = extractOdt(Buffer.from('this is not an odt at all'));
373
+ expect(res.ok).toBe(false);
374
+ if (res.ok) return;
375
+ expect(res.message).toContain('could not be parsed as a .odt');
376
+ });
377
+
378
+ it('returns a typed failure for a zip without content.xml', () => {
379
+ const zip = new AdmZip();
380
+ zip.addFile('hello.txt', Buffer.from('hi'));
381
+ const res = extractOdt(zip.toBuffer());
382
+ expect(res.ok).toBe(false);
383
+ if (res.ok) return;
384
+ expect(res.message).toContain('no content.xml');
385
+ });
386
+ });
387
+
388
+ // ── odp ────────────────────────────────────────────────────────────────────
389
+
390
+ describe('extractOdp', () => {
391
+ it('numbers slides by DOCUMENT order of <draw:page>, with notes under [slide N notes]', () => {
392
+ const res = extractOdp(
393
+ odpBytes(
394
+ odpPage(['Road', 'map 2026'], ['Remember the demo']),
395
+ odpPage(['Second slide']),
396
+ odpPage(['Third slide']),
397
+ ),
398
+ );
399
+ if (!res.ok) throw new Error(res.message);
400
+ expect(res.text.split('\n')).toEqual([
401
+ '[slide 1]',
402
+ 'Road',
403
+ 'map 2026',
404
+ '[slide 1 notes]',
405
+ 'Remember the demo',
406
+ '[slide 2]',
407
+ 'Second slide',
408
+ '[slide 3]',
409
+ 'Third slide',
410
+ ]);
411
+ expect(res.summary).toBe('3 slides + notes; layout, images and formatting omitted');
412
+ });
413
+
414
+ it('omits the notes marker (and "+ notes") when no slide has note text', () => {
415
+ const res = extractOdp(odpBytes(odpPage(['Only slide'])));
416
+ if (!res.ok) throw new Error(res.message);
417
+ expect(res.text).toBe('[slide 1]\nOnly slide');
418
+ expect(res.summary).toContain('1 slide;');
419
+ });
420
+
421
+ it('returns a typed failure for a zip whose content.xml has no draw:page', () => {
422
+ const res = extractOdp(odtBytes('<text:p>not a presentation</text:p>'));
423
+ expect(res.ok).toBe(false);
424
+ if (res.ok) return;
425
+ expect(res.message).toContain('could not be parsed as a .odp');
426
+ });
427
+
428
+ it('returns a typed failure for bytes that are not a zip', () => {
429
+ const res = extractOdp(Buffer.from('junk'));
430
+ expect(res.ok).toBe(false);
431
+ if (res.ok) return;
432
+ expect(res.message).toContain('could not be parsed as a .odp');
433
+ });
434
+ });
435
+
436
+ // ── ods ────────────────────────────────────────────────────────────────────
437
+
438
+ describe('extractOds', () => {
439
+ const row = (...cells: string[]): string => `<table:table-row>${cells.join('')}</table:table-row>`;
440
+ const table = (name: string, ...rows: string[]): string =>
441
+ `<table:table table:name="${name}">${rows.join('')}</table:table>`;
442
+
443
+ it('emits [sheet: Name] markers and rows as tab-separated cell text', () => {
444
+ const res = extractOds(
445
+ odsBytes(
446
+ table('Inventory', row(odsCell('Name'), odsCell('Qty')), row(odsCell('Widget'), odsCell('3'))),
447
+ table('Empty'),
448
+ ),
449
+ );
450
+ if (!res.ok) throw new Error(res.message);
451
+ const lines = res.text.split('\n');
452
+ expect(lines[0]).toBe('[sheet: Inventory]');
453
+ expect(lines[1]).toBe('Name\tQty');
454
+ expect(lines[2]).toBe('Widget\t3');
455
+ expect(lines).toContain('[sheet: Empty]');
456
+ expect(res.summary).toContain('2 sheets');
457
+ });
458
+
459
+ it('expands column/row repeats for real data', () => {
460
+ const res = extractOds(
461
+ odsBytes(
462
+ table(
463
+ 'S',
464
+ row(odsCell('x', 3), odsCell('end')),
465
+ `<table:table-row table:number-rows-repeated="2">${odsCell('dup')}</table:table-row>`,
466
+ ),
467
+ ),
468
+ );
469
+ if (!res.ok) throw new Error(res.message);
470
+ expect(res.text.split('\n')).toEqual(['[sheet: S]', 'x\tx\tx\tend', 'dup', 'dup']);
471
+ });
472
+
473
+ it('TRIMS trailing empty cells/rows before applying their repeats — million-wide grid padding costs nothing', () => {
474
+ const res = extractOds(
475
+ odsBytes(
476
+ table(
477
+ 'Padded',
478
+ row(odsCell('a'), odsCell('', 1_000_000)),
479
+ `<table:table-row table:number-rows-repeated="1048576">${odsCell('', 1_000_000)}</table:table-row>`,
480
+ ),
481
+ ),
482
+ );
483
+ if (!res.ok) throw new Error(res.message);
484
+ // No truncation note: the padding was trimmed, not truncated.
485
+ expect(res.text.split('\n')).toEqual(['[sheet: Padded]', 'a']);
486
+ });
487
+
488
+ it('caps NON-empty repeats at 200 columns / 10k rows and says so under the sheet marker', () => {
489
+ const res = extractOds(
490
+ odsBytes(
491
+ table(
492
+ 'Big',
493
+ row(odsCell('w', 300)),
494
+ `<table:table-row table:number-rows-repeated="20000">${odsCell('r')}</table:table-row>`,
495
+ ),
496
+ ),
497
+ );
498
+ if (!res.ok) throw new Error(res.message);
499
+ const lines = res.text.split('\n');
500
+ expect(lines[1]).toBe('[sheet truncated to the first 10000 rows and first 200 columns]');
501
+ expect(lines[2]).toBe(Array(200).fill('w').join('\t'));
502
+ expect(lines).toHaveLength(2 + 10_000);
503
+ });
504
+
505
+ it('the row cap STOPS the scan — 60k explicit rows extract their first 10k, fast', () => {
506
+ // Explicit (non-repeated) rows used to be materialized wholesale before
507
+ // the cap was consulted, so the cap bounded output but neither memory nor
508
+ // scan work. The walk now hands rows over as they parse and stops at the
509
+ // cap. The bound is generous — here to catch a return of the
510
+ // materialize-everything shape, not to police CI's scheduler.
511
+ const bytes = odsBytes(`<table:table table:name="Long">${row(odsCell('r')).repeat(60_000)}</table:table>`);
512
+ const t0 = performance.now();
513
+ const res = extractOds(bytes);
514
+ const ms = performance.now() - t0;
515
+ if (!res.ok) throw new Error(res.message);
516
+ const lines = res.text.split('\n');
517
+ expect(lines[1]).toBe('[sheet truncated to the first 10000 rows]');
518
+ expect(lines).toHaveLength(2 + 10_000);
519
+ expect(ms).toBeLessThan(5_000);
520
+ });
521
+
522
+ it('decodes entities in cell text and the sheet name; covered cells render empty', () => {
523
+ const res = extractOds(
524
+ odsBytes(
525
+ table(
526
+ 'P&amp;L',
527
+ row(odsCell('A&amp;B'), '<table:covered-table-cell/>', odsCell('C')),
528
+ ),
529
+ ),
530
+ );
531
+ if (!res.ok) throw new Error(res.message);
532
+ expect(res.text.split('\n')).toEqual(['[sheet: P&L]', 'A&B\t\tC']);
533
+ });
534
+
535
+ it('returns a typed failure for bytes that are not a zip', () => {
536
+ const res = extractOds(Buffer.from('nope'));
537
+ expect(res.ok).toBe(false);
538
+ if (res.ok) return;
539
+ expect(res.message).toContain('could not be parsed as a .ods');
540
+ });
541
+
542
+ it('returns a typed failure for a content.xml without table:table', () => {
543
+ const res = extractOds(odtBytes('<text:p>a text document</text:p>'));
544
+ expect(res.ok).toBe(false);
545
+ if (res.ok) return;
546
+ expect(res.message).toContain('no table:table');
547
+ });
548
+ });
549
+
550
+ // ── cache + service ────────────────────────────────────────────────────────
551
+
552
+ describe('DocExtractService cache', () => {
553
+ let cacheRoot = '';
554
+ let service: DocExtractService;
555
+
556
+ beforeEach(async () => {
557
+ cacheRoot = await mkdtemp(join(tmpdir(), 'doc-extract-cache-'));
558
+ service = new DocExtractService(cacheRoot);
559
+ });
560
+ afterEach(async () => {
561
+ await rm(cacheRoot, { recursive: true, force: true });
562
+ });
563
+
564
+ it('extracts ONCE per content: the second read is served from the cache file', async () => {
565
+ const bytes = docxBytes(para('Cache me once'));
566
+ const first = await service.extract('a.docx', bytes, extractDocx);
567
+ expect(first.ok).toBe(true);
568
+ const entries = await readdir(cacheRoot);
569
+ expect(entries).toEqual([`${gitBlobSha(bytes)}.docx.${EXTRACTION_SCHEMA}.json`]);
570
+
571
+ // Tamper with the cached entry: if the second extract returns the tampered
572
+ // text, it came from the cache — the parser did NOT run again. (Real-files
573
+ // proof without spying on module internals.)
574
+ await writeFile(join(cacheRoot, entries[0]), JSON.stringify({ summary: 'tampered summary', text: 'FROM CACHE' }), 'utf8');
575
+ const second = await service.extract('a.docx', bytes, extractDocx);
576
+ if (!second.ok) throw new Error(second.message);
577
+ expect(second.text).toBe('FROM CACHE');
578
+ expect(second.marker).toBe('[extracted text of a.docx — tampered summary]');
579
+ });
580
+
581
+ it('is invalidated by CONTENT change (new blob sha → fresh extraction, new entry)', async () => {
582
+ const v1 = docxBytes(para('version one'));
583
+ const v2 = docxBytes(para('version two'));
584
+ await service.extract('a.docx', v1, extractDocx);
585
+ const after1 = await readdir(cacheRoot);
586
+ const res = await service.extract('a.docx', v2, extractDocx);
587
+ if (!res.ok) throw new Error(res.message);
588
+ expect(res.text).toBe('version two');
589
+ const after2 = await readdir(cacheRoot);
590
+ expect(after2).toHaveLength(2);
591
+ expect(after2).toEqual(expect.arrayContaining(after1));
592
+ });
593
+
594
+ it('keys by content, not path: the same bytes under another path reuse the entry, marker names the read path', async () => {
595
+ const bytes = docxBytes(para('Shared content'));
596
+ await service.extract('one.docx', bytes, extractDocx);
597
+ expect(await readdir(cacheRoot)).toHaveLength(1);
598
+ const other = await service.extract('two/other.docx', bytes, extractDocx);
599
+ if (!other.ok) throw new Error(other.message);
600
+ expect(other.marker).toMatch(/^\[extracted text of two\/other\.docx — /);
601
+ expect(await readdir(cacheRoot)).toHaveLength(1);
602
+ });
603
+
604
+ it('getCached misses before extraction and hits after — the grep budget contract', async () => {
605
+ const bytes = docxBytes(para('Budget line'));
606
+ expect(await service.getCached('a.docx', bytes)).toBeUndefined();
607
+ await service.extract('a.docx', bytes, extractDocx);
608
+ const hit = await service.getCached('a.docx', bytes);
609
+ expect(hit?.text).toBe('Budget line');
610
+ });
611
+
612
+ it('does NOT cache failures — a corrupt file is re-tried on the next read', async () => {
613
+ const res = await service.extract('bad.docx', Buffer.from('junk'), extractDocx);
614
+ expect(res.ok).toBe(false);
615
+ expect(await readdir(cacheRoot)).toHaveLength(0);
616
+ });
617
+
618
+ it('computes the git blob sha (matches `git hash-object`)', async () => {
619
+ // echo -n 'hello' | git hash-object --stdin
620
+ expect(gitBlobSha(Buffer.from('hello'))).toBe('b6fc4c620b67d95f953a5c1c1230aaab5db5a1b0');
621
+ });
622
+
623
+ it('the cached entry on disk is the {summary, text} JSON', async () => {
624
+ const bytes = docxBytes(para('On disk'));
625
+ await service.extract('a.docx', bytes, extractDocx);
626
+ const raw = JSON.parse(await readFile(join(cacheRoot, `${gitBlobSha(bytes)}.docx.${EXTRACTION_SCHEMA}.json`), 'utf8')) as { summary: string; text: string };
627
+ expect(raw.text).toBe('On disk');
628
+ expect(typeof raw.summary).toBe('string');
629
+ });
630
+
631
+ it('keys by content PLUS format: identical bytes under another extension re-extract, never cross-hit', async () => {
632
+ // One ODF package that both extractors accept: a text body holding a table.
633
+ const bytes = odfBytes(
634
+ 'application/vnd.oasis.opendocument.text',
635
+ '<office:text><text:p>Prose line</text:p>' +
636
+ '<table:table table:name="Grid"><table:table-row>' +
637
+ `${odsCell('cell')}` +
638
+ '</table:table-row></table:table></office:text>',
639
+ );
640
+ const asOdt = await service.extract('doc.odt', bytes, extractOdt);
641
+ if (!asOdt.ok) throw new Error(asOdt.message);
642
+ expect(asOdt.text).toContain('Prose line');
643
+
644
+ // Same bytes, renamed .ods: the odt extraction must NOT come back.
645
+ const asOds = await service.extract('doc.ods', bytes, extractOds);
646
+ if (!asOds.ok) throw new Error(asOds.message);
647
+ expect(asOds.text.split('\n')[0]).toBe('[sheet: Grid]');
648
+ expect(asOds.marker).toContain('sheet');
649
+
650
+ const entries = await readdir(cacheRoot);
651
+ expect(entries.sort()).toEqual([`${gitBlobSha(bytes)}.ods.${EXTRACTION_SCHEMA}.json`, `${gitBlobSha(bytes)}.odt.${EXTRACTION_SCHEMA}.json`].sort());
652
+
653
+ // getCached honours the same format-qualified key.
654
+ const cachedOds = await service.getCached('doc.ods', bytes);
655
+ expect(cachedOds?.text.split('\n')[0]).toBe('[sheet: Grid]');
656
+ });
657
+ });
658
+
659
+ // ── cache size bounding ────────────────────────────────────────────────────
660
+
661
+ describe('DocExtractionCache pruning', () => {
662
+ let root = '';
663
+ beforeEach(async () => {
664
+ root = await mkdtemp(join(tmpdir(), 'extract-cache-'));
665
+ });
666
+ afterEach(async () => {
667
+ await rm(root, { recursive: true, force: true });
668
+ });
669
+
670
+ const dirTotalBytes = async (): Promise<number> => {
671
+ let total = 0;
672
+ for (const name of await readdir(root)) total += (await stat(join(root, name))).size;
673
+ return total;
674
+ };
675
+
676
+ it('prunes towards the byte bound as sequential puts pass it', async () => {
677
+ const cache = new DocExtractionCache(root, 300);
678
+ for (let i = 0; i < 10; i++) await cache.put(`seq${i}`, { summary: 's', text: 'x'.repeat(80) });
679
+ expect(await dirTotalBytes()).toBeLessThanOrEqual(300);
680
+ });
681
+
682
+ it('keeps writes that race a prune ACCOUNTED — the next put still prunes to the bound', async () => {
683
+ // Concurrent puts can land while a prune's scan runs; resetting the
684
+ // written-bytes counter to zero after the scan discarded them, so later
685
+ // puts trusted a total the scan never saw and skipped pruning entirely.
686
+ const cache = new DocExtractionCache(root, 300);
687
+ await Promise.all(
688
+ Array.from({ length: 12 }, (_, i) => cache.put(`race${i}`, { summary: 's', text: 'y'.repeat(80) })),
689
+ );
690
+ await cache.put('after-the-race', { summary: 's', text: 'z'.repeat(80) });
691
+ expect(await dirTotalBytes()).toBeLessThanOrEqual(300);
692
+ });
693
+ });
694
+
695
+ // ── rels-paired pptx notes ─────────────────────────────────────────────────
696
+
697
+ describe('extractPptx notes pairing via rels', () => {
698
+ const RELS_NS = 'xmlns="http://schemas.openxmlformats.org/package/2006/relationships"';
699
+ const NOTES_TYPE = 'http://schemas.openxmlformats.org/officeDocument/2006/relationships/notesSlide';
700
+
701
+ function pptxWithRels(
702
+ slides: Record<number, string>,
703
+ notesParts: Record<string, string>,
704
+ rels: Record<number, string>,
705
+ ): Buffer {
706
+ const zip = new AdmZip();
707
+ zip.addFile('[Content_Types].xml', Buffer.from('<?xml version="1.0"?><Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"/>'));
708
+ for (const [n, xml] of Object.entries(slides)) zip.addFile(`ppt/slides/slide${n}.xml`, Buffer.from(xml));
709
+ for (const [name, xml] of Object.entries(notesParts)) zip.addFile(`ppt/notesSlides/${name}`, Buffer.from(xml));
710
+ for (const [n, xml] of Object.entries(rels)) zip.addFile(`ppt/slides/_rels/slide${n}.xml.rels`, Buffer.from(xml));
711
+ return zip.toBuffer();
712
+ }
713
+
714
+ it('pairs notes through the slide rels even when the notes part number differs from the slide number', () => {
715
+ const res = extractPptx(
716
+ pptxWithRels(
717
+ { 1: slideXml(['First']), 2: slideXml(['Second']) },
718
+ { 'notesSlide7.xml': slideXml(['note for slide two']) },
719
+ {
720
+ 1: `<?xml version="1.0"?><Relationships ${RELS_NS}></Relationships>`,
721
+ 2:
722
+ `<?xml version="1.0"?><Relationships ${RELS_NS}>` +
723
+ `<Relationship Id="rId9" Type="${NOTES_TYPE}" Target="../notesSlides/notesSlide7.xml"/>` +
724
+ '</Relationships>',
725
+ },
726
+ ),
727
+ );
728
+ if (!res.ok) throw new Error(res.message);
729
+ expect(res.text.split('\n')).toEqual([
730
+ '[slide 1]',
731
+ 'First',
732
+ '[slide 2]',
733
+ 'Second',
734
+ '[slide 2 notes]',
735
+ 'note for slide two',
736
+ ]);
737
+ });
738
+
739
+ it('accepts single-quoted rels attributes and whitespace around =', () => {
740
+ const res = extractPptx(
741
+ pptxWithRels(
742
+ { 1: slideXml(['Solo']) },
743
+ { 'notesSlide3.xml': slideXml(['quoted note']) },
744
+ {
745
+ 1:
746
+ `<?xml version="1.0"?><Relationships ${RELS_NS}>` +
747
+ `<Relationship Id='r1' Type = '${NOTES_TYPE}' Target = '../notesSlides/notesSlide3.xml'/>` +
748
+ '</Relationships>',
749
+ },
750
+ ),
751
+ );
752
+ if (!res.ok) throw new Error(res.message);
753
+ expect(res.text).toBe('[slide 1]\nSolo\n[slide 1 notes]\nquoted note');
754
+ });
755
+
756
+ it('with a rels part that names NO notes slide, the numeric twin is NOT attached', () => {
757
+ const res = extractPptx(
758
+ pptxWithRels(
759
+ { 1: slideXml(['No notes really']) },
760
+ { 'notesSlide1.xml': slideXml(['orphan notes part']) },
761
+ { 1: `<?xml version="1.0"?><Relationships ${RELS_NS}></Relationships>` },
762
+ ),
763
+ );
764
+ if (!res.ok) throw new Error(res.message);
765
+ expect(res.text).toBe('[slide 1]\nNo notes really');
766
+ });
767
+
768
+ it('falls back to numeric pairing when the slide has no rels part', () => {
769
+ const res = extractPptx(
770
+ pptxBytes({ 1: slideXml(['Fallback slide']) }, { 1: slideXml(['numeric note']) }),
771
+ );
772
+ if (!res.ok) throw new Error(res.message);
773
+ expect(res.text).toBe('[slide 1]\nFallback slide\n[slide 1 notes]\nnumeric note');
774
+ });
775
+ });
776
+
777
+ // ── ODF namespace-prefix normalization + attribute quoting ─────────────────
778
+
779
+ describe('ODF prefix normalization and attribute robustness', () => {
780
+ it('extracts an odt whose producer bound NON-conventional prefixes to the ODF namespaces', () => {
781
+ const zip = new AdmZip();
782
+ zip.addFile('mimetype', Buffer.from('application/vnd.oasis.opendocument.text'));
783
+ zip.addFile(
784
+ 'content.xml',
785
+ Buffer.from(
786
+ '<?xml version="1.0"?><o:document-content ' +
787
+ 'xmlns:o="urn:oasis:names:tc:opendocument:xmlns:office:1.0" ' +
788
+ "xmlns:t='urn:oasis:names:tc:opendocument:xmlns:text:1.0'>" +
789
+ '<o:body><o:text>' +
790
+ '<t:h>Title</t:h>' +
791
+ "<t:p>a<t:tab/>b<t:line-break/>c<t:s t:c = '3'/>d</t:p>" +
792
+ '</o:text></o:body></o:document-content>',
793
+ ),
794
+ );
795
+ const res = extractOdt(zip.toBuffer());
796
+ if (!res.ok) throw new Error(res.message);
797
+ expect(res.text.split('\n')).toEqual(['Title', 'a\tb', 'c d']);
798
+ });
799
+
800
+ it('extracts an odp whose draw/presentation prefixes are non-conventional', () => {
801
+ const zip = new AdmZip();
802
+ zip.addFile('mimetype', Buffer.from('application/vnd.oasis.opendocument.presentation'));
803
+ zip.addFile(
804
+ 'content.xml',
805
+ Buffer.from(
806
+ '<?xml version="1.0"?><office:document-content ' +
807
+ 'xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0" ' +
808
+ 'xmlns:txt="urn:oasis:names:tc:opendocument:xmlns:text:1.0" ' +
809
+ 'xmlns:d="urn:oasis:names:tc:opendocument:xmlns:drawing:1.0" ' +
810
+ 'xmlns:pres="urn:oasis:names:tc:opendocument:xmlns:presentation:1.0">' +
811
+ '<office:body><office:presentation>' +
812
+ '<d:page><d:frame><txt:p>Visible</txt:p></d:frame>' +
813
+ '<pres:notes><txt:p>a note</txt:p></pres:notes></d:page>' +
814
+ '</office:presentation></office:body></office:document-content>',
815
+ ),
816
+ );
817
+ const res = extractOdp(zip.toBuffer());
818
+ if (!res.ok) throw new Error(res.message);
819
+ expect(res.text.split('\n')).toEqual(['[slide 1]', 'Visible', '[slide 1 notes]', 'a note']);
820
+ });
821
+
822
+ it('reads single-quoted table:name and repeat attributes in an ods', () => {
823
+ const zip = new AdmZip();
824
+ zip.addFile('mimetype', Buffer.from('application/vnd.oasis.opendocument.spreadsheet'));
825
+ zip.addFile(
826
+ 'content.xml',
827
+ Buffer.from(
828
+ '<?xml version="1.0"?><office:document-content ' +
829
+ 'xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0" ' +
830
+ 'xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0" ' +
831
+ 'xmlns:table="urn:oasis:names:tc:opendocument:xmlns:table:1.0">' +
832
+ '<office:body><office:spreadsheet>' +
833
+ "<table:table table:name = 'Quoted'><table:table-row>" +
834
+ "<table:table-cell table:number-columns-repeated = '3'><text:p>x</text:p></table:table-cell>" +
835
+ '<table:table-cell><text:p>end</text:p></table:table-cell>' +
836
+ '</table:table-row></table:table>' +
837
+ '</office:spreadsheet></office:body></office:document-content>',
838
+ ),
839
+ );
840
+ const res = extractOds(zip.toBuffer());
841
+ if (!res.ok) throw new Error(res.message);
842
+ expect(res.text.split('\n')).toEqual(['[sheet: Quoted]', 'x\tx\tx\tend']);
843
+ });
844
+ });
845
+
846
+ // ── odp self-closing pages / ods in-cell breaks ────────────────────────────
847
+
848
+ describe('ODF edge shapes', () => {
849
+ it('a self-closing <draw:page/> is an EMPTY slide, not a dropped one', () => {
850
+ const res = extractOdp(
851
+ odfBytes(
852
+ 'application/vnd.oasis.opendocument.presentation',
853
+ '<office:presentation>' +
854
+ odpPage(['First real slide']) +
855
+ '<draw:page draw:name="blank"/>' +
856
+ odpPage(['Third slide']) +
857
+ '</office:presentation>',
858
+ ),
859
+ );
860
+ if (!res.ok) throw new Error(res.message);
861
+ expect(res.text.split('\n')).toEqual(['[slide 1]', 'First real slide', '[slide 2]', '[slide 3]', 'Third slide']);
862
+ expect(res.summary).toContain('3 slides');
863
+ });
864
+
865
+ it('a deck of ONLY self-closing pages still parses as its (blank) slides', () => {
866
+ const res = extractOdp(
867
+ odfBytes('application/vnd.oasis.opendocument.presentation', '<office:presentation><draw:page/><draw:page/></office:presentation>'),
868
+ );
869
+ if (!res.ok) throw new Error(res.message);
870
+ expect(res.text).toBe('[slide 1]\n[slide 2]');
871
+ });
872
+
873
+ it('element-produced newlines/tabs INSIDE an ods cell become single spaces — the TSV row survives', () => {
874
+ const res = extractOds(
875
+ odsBytes(
876
+ '<table:table table:name="Wrapped"><table:table-row>' +
877
+ '<table:table-cell><text:p>line one<text:line-break/>line two<text:tab/>tabbed</text:p></table:table-cell>' +
878
+ '<table:table-cell><text:p>next cell</text:p></table:table-cell>' +
879
+ '</table:table-row></table:table>',
880
+ ),
881
+ );
882
+ if (!res.ok) throw new Error(res.message);
883
+ const lines = res.text.split('\n');
884
+ expect(lines).toHaveLength(2);
885
+ expect(lines[1]).toBe('line one line two tabbed\tnext cell');
886
+ });
887
+ });
888
+
889
+ // ── decompression bounds (zip bombs) ───────────────────────────────────────
890
+
891
+ describe('bounded extraction (zip bombs / oversized inputs)', () => {
892
+ const PART_LIMIT = 50 * 1024 * 1024;
893
+ // Compresses to a few KB, inflates past the 50 MB part limit.
894
+ const bigXml = (): Buffer => Buffer.alloc(PART_LIMIT + 1024, 0x20);
895
+
896
+ it('refuses a docx whose word/document.xml declares an over-limit uncompressed size', () => {
897
+ const zip = new AdmZip();
898
+ zip.addFile('word/document.xml', bigXml());
899
+ const res = extractDocx(zip.toBuffer());
900
+ expect(res.ok).toBe(false);
901
+ if (res.ok) return;
902
+ expect(res.message).toContain('extraction limit');
903
+ expect(res.message).toContain('word/document.xml');
904
+ });
905
+
906
+ it('refuses an odt whose content.xml declares an over-limit uncompressed size', () => {
907
+ const zip = new AdmZip();
908
+ zip.addFile('mimetype', Buffer.from('application/vnd.oasis.opendocument.text'));
909
+ zip.addFile('content.xml', bigXml());
910
+ const res = extractOdt(zip.toBuffer());
911
+ expect(res.ok).toBe(false);
912
+ if (res.ok) return;
913
+ expect(res.message).toContain('extraction limit');
914
+ });
915
+
916
+ it('refuses a pptx with an over-limit slide part', () => {
917
+ const zip = new AdmZip();
918
+ zip.addFile('ppt/slides/slide1.xml', bigXml());
919
+ const res = extractPptx(zip.toBuffer());
920
+ expect(res.ok).toBe(false);
921
+ if (res.ok) return;
922
+ expect(res.message).toContain('extraction limit');
923
+ });
924
+
925
+ it('refuses an xlsx with an over-limit part BEFORE SheetJS inflates it', () => {
926
+ const zip = new AdmZip();
927
+ zip.addFile('xl/worksheets/sheet1.xml', bigXml());
928
+ const res = extractXlsx(zip.toBuffer());
929
+ expect(res.ok).toBe(false);
930
+ if (res.ok) return;
931
+ expect(res.message).toContain('extraction limit');
932
+ });
933
+
934
+ it('refuses a PDF larger than the 50 MB byte cap without parsing it', async () => {
935
+ const big = Buffer.alloc(PART_LIMIT + 1, 0x25);
936
+ const res = await extractPdf(big);
937
+ expect(res.ok).toBe(false);
938
+ if (res.ok) return;
939
+ expect(res.message).toContain('extraction limit');
940
+ });
941
+ });
942
+
943
+ // ── single-pass prefix normalization + alias bound ─────────────────────────
944
+
945
+ describe('markup that only LOOKS like metadata', () => {
946
+ it('a <Relationship> written inside a comment does not point the notes lookup', async () => {
947
+ // A decoy in a comment used to answer as live metadata, so the notes for a
948
+ // slide came from whichever part its author named.
949
+ const bytes = pptxBytes(
950
+ { 1: slideXml(['Real slide']) },
951
+ { 2: slideXml(['the decoy notes']), 7: slideXml(['the real notes']) },
952
+ {
953
+ 'ppt/slides/_rels/slide1.xml.rels':
954
+ '<?xml version="1.0"?><Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">' +
955
+ '<!-- <Relationship Id="rX" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/notesSlide" Target="../notesSlides/notesSlide2.xml"/> -->' +
956
+ '<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/notesSlide" Target="../notesSlides/notesSlide7.xml"/>' +
957
+ '</Relationships>',
958
+ },
959
+ );
960
+ const res = await extractPptx(bytes);
961
+ if (!res.ok) throw new Error(res.message);
962
+ expect(res.text).toContain('the real notes');
963
+ expect(res.text).not.toContain('the decoy notes');
964
+ });
965
+
966
+ it('a commented <w:body> does not answer as the document body', () => {
967
+ const res = extractDocx(docxBytes('<!-- <w:body>' + para('ghost') + '</w:body> -->' + para('real')));
968
+ if (!res.ok) throw new Error(res.message);
969
+ expect(res.text).toBe('real');
970
+ });
971
+ });
972
+ describe('ODF namespace prefixes are the document’s own choice', () => {
973
+ // These used to be answered by rewriting every non-conventional prefix across
974
+ // content.xml before scanning it — a pass that had to be bounded against
975
+ // crafted alias lists and could still corrupt paragraph text that happened to
976
+ // look like an alias. Elements are matched on their LOCAL name now, so the
977
+ // prefix simply does not matter and the pass is gone.
978
+ const odtWith = (rootAttrs: string, body: string): Buffer =>
979
+ odfBytesRaw('application/vnd.oasis.opendocument.text', rootAttrs, body);
980
+
981
+ it('extracts a document that binds the ODF namespaces to UNUSUAL prefixes', () => {
982
+ const res = extractOdt(
983
+ odtWith(
984
+ 'xmlns:o="urn:oasis:names:tc:opendocument:xmlns:office:1.0" xmlns:zz="urn:oasis:names:tc:opendocument:xmlns:text:1.0"',
985
+ '<o:body><o:text><zz:p>Renamed prefixes</zz:p><zz:h>And a heading</zz:h></o:text></o:body>',
986
+ ),
987
+ );
988
+ if (!res.ok) throw new Error(res.message);
989
+ expect(res.text).toBe('Renamed prefixes\nAnd a heading');
990
+ });
991
+
992
+ it('extracts a document that DEFAULTS the text namespace and uses no prefix at all', () => {
993
+ const res = extractOdt(
994
+ odtWith(
995
+ 'xmlns="urn:oasis:names:tc:opendocument:xmlns:text:1.0"',
996
+ '<body><text><p>No prefix anywhere</p></text></body>',
997
+ ),
998
+ );
999
+ if (!res.ok) throw new Error(res.message);
1000
+ expect(res.text).toBe('No prefix anywhere');
1001
+ });
1002
+
1003
+ it('leaves paragraph text that merely LOOKS like a prefix alone', () => {
1004
+ const res = extractOdt(
1005
+ odtWith(
1006
+ 'xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0"',
1007
+ '<office:body><office:text><text:p>see t: and zz: in the prose</text:p></office:text></office:body>',
1008
+ ),
1009
+ );
1010
+ if (!res.ok) throw new Error(res.message);
1011
+ expect(res.text).toBe('see t: and zz: in the prose');
1012
+ });
1013
+ });
1014
+
1015
+ // ── attribute tokenizer (quoted-value skipping) ────────────────────────────
1016
+
1017
+ describe('xmlAttrValue tokenizer', () => {
1018
+ it('never matches an attribute-looking sequence INSIDE another attribute quoted value', () => {
1019
+ expect(xmlAttrValue(`<a foo="Target='evil'" Target="real"/>`, 'Target')).toBe('real');
1020
+ expect(xmlAttrValue(`<a foo="Target='evil'"/>`, 'Target')).toBeUndefined();
1021
+ expect(xmlAttrValue(`<a foo='Target="evil"' Target='real'/>`, 'Target')).toBe('real');
1022
+ });
1023
+
1024
+ it('accepts both quote styles and whitespace around =', () => {
1025
+ expect(xmlAttrValue(`<t:s t:c = '3'/>`, 't:c')).toBe('3');
1026
+ expect(xmlAttrValue('<t:s t:c="4"/>', 't:c')).toBe('4');
1027
+ });
1028
+
1029
+ it('matches the full (prefixed) name exactly, and scans attrs-only fragments', () => {
1030
+ expect(xmlAttrValue('<x a:n="p" n="bare"/>', 'n')).toBe('bare');
1031
+ expect(xmlAttrValue(' table:number-columns-repeated="7"', 'table:number-columns-repeated')).toBe('7');
1032
+ });
1033
+
1034
+ it('xmlAttrValueByLocalName matches through any namespace prefix', () => {
1035
+ expect(xmlAttrValueByLocalName('<r:Relationship r:Target="notes.xml"/>', 'Target')).toBe('notes.xml');
1036
+ expect(xmlAttrValueByLocalName('<Relationship Target="notes.xml"/>', 'Target')).toBe('notes.xml');
1037
+ expect(xmlAttrValueByLocalName(`<r:Rel foo="Target='no'" r:Target='yes'/>`, 'Target')).toBe('yes');
1038
+ });
1039
+
1040
+ it('xmlAttrValueByLocalName never mistakes an xmlns declaration for the attribute', () => {
1041
+ // `xmlns:Target` binds a namespace prefix named "Target" — it is a
1042
+ // declaration, not a Target attribute, and its value is a URI.
1043
+ expect(
1044
+ xmlAttrValueByLocalName('<Relationship xmlns:Target="http://ns.example/x" Target="real.xml"/>', 'Target'),
1045
+ ).toBe('real.xml');
1046
+ expect(xmlAttrValueByLocalName('<Relationship xmlns:Type="http://ns.example/x"/>', 'Type')).toBeUndefined();
1047
+ expect(xmlAttrValueByLocalName('<Relationship xmlns="http://ns.example/x"/>', 'xmlns')).toBeUndefined();
1048
+ });
1049
+
1050
+ it('stops safely on a malformed tail (unterminated quote, unquoted value)', () => {
1051
+ expect(xmlAttrValue('<a b="ok" c="unterminated', 'b')).toBe('ok');
1052
+ expect(xmlAttrValue('<a b=unquoted c="x"/>', 'c')).toBeUndefined();
1053
+ });
1054
+ });
1055
+
1056
+ // ── crafted ODF: unmatched openers must not rescan the document ────────────
1057
+
1058
+ describe('ODF element scanning stays linear on crafted content.xml', () => {
1059
+ /**
1060
+ * The shape that used to pin the extractor: a wall of OPENERS whose close
1061
+ * tag never comes. Each one made the old
1062
+ * `<text:p…(?:\/>|>([\s\S]*?)<\/text:p>)` regex re-scan everything to the
1063
+ * right, so cost grew as openers x bytes — 464 KB measured at 2.3 s and
1064
+ * QUADRUPLING per doubling, i.e. hours at the 50 MB `MAX_DOC_PART_BYTES`
1065
+ * cap, from an upload that zips down to a few kilobytes.
1066
+ *
1067
+ * The bound is deliberately generous (seconds): it is here to catch a
1068
+ * return of the quadratic, not to police CI's scheduler. The scanner does
1069
+ * these in single-digit milliseconds; the old regex needed ~30 s.
1070
+ */
1071
+ const GENEROUS_MS = 5_000;
1072
+
1073
+ it('an odt of 40k unmatched <text:p> openers extracts its real paragraph, fast', () => {
1074
+ const bytes = odtBytes('<text:p>Kept</text:p>' + '<text:p text:style-name="s">x'.repeat(40_000));
1075
+ const t0 = performance.now();
1076
+ const res = extractOdt(bytes);
1077
+ const ms = performance.now() - t0;
1078
+ if (!res.ok) throw new Error(res.message);
1079
+ // The real paragraph comes first and is intact. CHANGED: the openers nest
1080
+ // rather than being dropped, so the `x` between them is recovered as one
1081
+ // more paragraph — bounded by MAX_ELEMENT_DEPTH, and it IS text the file
1082
+ // contains. What matters here is that this stays fast and bounded.
1083
+ expect(res.text.startsWith('Kept\n')).toBe(true);
1084
+ expect(ms).toBeLessThan(GENEROUS_MS);
1085
+ });
1086
+
1087
+ it('an odp of 40k unmatched <draw:page> openers keeps its real slide, fast', () => {
1088
+ const bytes = odfBytes(
1089
+ 'application/vnd.oasis.opendocument.presentation',
1090
+ '<office:presentation>' +
1091
+ odpPage(['Real slide']) +
1092
+ '<draw:page draw:name="a">'.repeat(40_000) +
1093
+ '</office:presentation>',
1094
+ );
1095
+ const t0 = performance.now();
1096
+ const res = extractOdp(bytes);
1097
+ const ms = performance.now() - t0;
1098
+ if (!res.ok) throw new Error(res.message);
1099
+ // CHANGED: the dangling openers nest into one recovered (empty) page
1100
+ // instead of being dropped. The real slide is intact and first.
1101
+ expect(res.text.startsWith('[slide 1]\nReal slide')).toBe(true);
1102
+ expect(ms).toBeLessThan(GENEROUS_MS);
1103
+ });
1104
+
1105
+ it('an ods of 20k unmatched row/cell openers keeps its real row, fast', () => {
1106
+ const bytes = odsBytes(
1107
+ '<table:table table:name="Q">' +
1108
+ `<table:table-row>${odsCell('alive')}</table:table-row>` +
1109
+ '<table:table-row><table:table-cell>'.repeat(20_000) +
1110
+ '</table:table>',
1111
+ );
1112
+ const t0 = performance.now();
1113
+ const res = extractOds(bytes);
1114
+ const ms = performance.now() - t0;
1115
+ if (!res.ok) throw new Error(res.message);
1116
+ expect(res.text.split('\n')).toEqual(['[sheet: Q]', 'alive']);
1117
+ expect(ms).toBeLessThan(GENEROUS_MS);
1118
+ });
1119
+
1120
+ it('an UNTERMINATED attribute quote ends the scan instead of restarting it per opener', () => {
1121
+ // The other half of the old blow-up: the lazy attribute region could never
1122
+ // pass an unclosed quote, so every opener paid a full scan to find that out.
1123
+ const bytes = odtBytes(
1124
+ '<text:p>Kept</text:p>' + '<text:p text:style-name="never closed'.repeat(40_000),
1125
+ );
1126
+ const t0 = performance.now();
1127
+ const res = extractOdt(bytes);
1128
+ const ms = performance.now() - t0;
1129
+ if (!res.ok) throw new Error(res.message);
1130
+ expect(res.text).toBe('Kept');
1131
+ expect(ms).toBeLessThan(GENEROUS_MS);
1132
+ });
1133
+
1134
+ it('an odt wall of openers ending at a FAR close tag stays linear too', () => {
1135
+ // The residual hole in the round-4/5 ODF scanner, closed by sharing
1136
+ // `xmlElementBlocks` with the OOXML walks: a failure-only memo recorded
1137
+ // nothing when every attribute scan succeeded at a far-away `>`. Measured
1138
+ // on this exact shape before the fix: 740 KB took 44.6 s.
1139
+ const bytes = odtBytes(
1140
+ '<text:p>Kept</text:p>' + '<text:p text:style-name="never closed'.repeat(20_000),
1141
+ );
1142
+ const t0 = performance.now();
1143
+ const res = extractOdt(bytes);
1144
+ const ms = performance.now() - t0;
1145
+ if (!res.ok) throw new Error(res.message);
1146
+ expect(res.text).toBe('Kept');
1147
+ expect(ms).toBeLessThan(GENEROUS_MS);
1148
+ });
1149
+
1150
+ it('reads a self-closing paragraph, a quoted /> and a quoted > the way the regex did', () => {
1151
+ const text = (bodyXml: string): string => {
1152
+ const res = extractOdt(odtBytes(bodyXml));
1153
+ if (!res.ok) throw new Error(res.message);
1154
+ return res.text;
1155
+ };
1156
+ // A self-closing paragraph is an EMPTY line, with attributes or without.
1157
+ expect(text('<text:p>a</text:p><text:p/><text:p text:style-name="s"/><text:p>b</text:p>')).toBe(
1158
+ 'a\n\n\nb',
1159
+ );
1160
+ // A `/>` or a `>` inside a quoted attribute value is part of the value.
1161
+ expect(text('<text:p n="a/>b">body</text:p>')).toBe('body');
1162
+ expect(text("<text:p n='a>b'>body</text:p>")).toBe('body');
1163
+ // A heading and a paragraph interleave in document order; `<text:page-number>`
1164
+ // is not a paragraph.
1165
+ expect(text('<text:h>H</text:h><text:page-number>9</text:page-number><text:p>P</text:p>')).toBe(
1166
+ 'H\nP',
1167
+ );
1168
+ // CHANGED: an opener with no close tag anywhere used to contribute nothing.
1169
+ // The parser closes it at end of input, so its text is recovered.
1170
+ expect(text('<text:p>dangling')).toBe('dangling');
1171
+ });
1172
+ });
1173
+
1174
+ // ── numeric character references ───────────────────────────────────────────
1175
+
1176
+ describe('decodeXmlEntities — numeric character references', () => {
1177
+ it('decodes well-formed decimal and hex references', () => {
1178
+ expect(decodeXmlEntities('&#65;&#x41;&#x1F600;')).toBe('AA\u{1F600}');
1179
+ expect(decodeXmlEntities('&amp;&lt;&gt;&quot;&apos;')).toBe('&<>"\'');
1180
+ });
1181
+
1182
+ it('leaves a MALFORMED reference literal instead of inventing a character', () => {
1183
+ // `&#12A;` is not a reference at all. The old pattern accepted hex digits
1184
+ // after a bare `#`, then parsed them as DECIMAL — parseInt('12A', 10)
1185
+ // stops at the 'A' and yields 12, so the text silently became U+000C.
1186
+ expect(decodeXmlEntities('&#12A;')).toBe('&#12A;');
1187
+ expect(decodeXmlEntities('price &#12A; each')).toBe('price &#12A; each');
1188
+ expect(decodeXmlEntities('&#;')).toBe('&#;');
1189
+ expect(decodeXmlEntities('&#xZZ;')).toBe('&#xZZ;');
1190
+ });
1191
+
1192
+ it('follows XML, not HTML, on the hex marker: `&#X41;` is not a reference', () => {
1193
+ // XML 1.0 §4.1 spells the marker lowercase `x` only; HTML5 also accepts
1194
+ // `X`. This decoder serves XML parts (and an email body strip that leans
1195
+ // on it), so the strict reading wins and a capital X stays literal rather
1196
+ // than being read as an unrelated decimal.
1197
+ expect(decodeXmlEntities('&#X41;')).toBe('&#X41;');
1198
+ });
1199
+
1200
+ it('leaves an out-of-range code point literal rather than throwing', () => {
1201
+ expect(decodeXmlEntities('&#1114112;')).toBe('&#1114112;'); // 0x110000
1202
+ expect(decodeXmlEntities('&#x110000;')).toBe('&#x110000;');
1203
+ });
1204
+ });
1205
+
1206
+ // ── duplicate numeric slide parts ──────────────────────────────────────────
1207
+
1208
+ describe('extractPptx — two part names parsing to the SAME slide number', () => {
1209
+ it('keeps ONE slide per number, choosing the first in ascending part-name order', () => {
1210
+ // A producer can ship both `slide1.xml` and `slide01.xml`. Zip entry order
1211
+ // is not a contract, so the winner is picked by part NAME — the same rule
1212
+ // the browser twin (`pptxOutline.ts`) applies, so the viewer and an
1213
+ // agent's `read_file` describe the same deck.
1214
+ const zip = new AdmZip();
1215
+ zip.addFile('ppt/slides/slide1.xml', Buffer.from(slideXml(['canonical one'])));
1216
+ zip.addFile('ppt/slides/slide01.xml', Buffer.from(slideXml(['zero-padded twin'])));
1217
+ zip.addFile('ppt/slides/slide2.xml', Buffer.from(slideXml(['two'])));
1218
+ const res = extractPptx(zip.toBuffer());
1219
+ if (!res.ok) throw new Error(res.message);
1220
+ expect(res.text.split('\n')).toEqual(['[slide 1]', 'zero-padded twin', '[slide 2]', 'two']);
1221
+ expect(res.summary).toContain('2 slides');
1222
+ });
1223
+
1224
+ const NOTES_REL = 'http://schemas.openxmlformats.org/officeDocument/2006/relationships/notesSlide';
1225
+ const relsXml = (target: string): string =>
1226
+ '<?xml version="1.0"?><Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">' +
1227
+ `<Relationship Id="rId1" Type="${NOTES_REL}" Target="${target}"/></Relationships>`;
1228
+
1229
+ it('follows the SELECTED part name to the rels — a zero-padded winner keeps its own notes', () => {
1230
+ // The regression the dedup introduced: the winner is chosen by NAME, but
1231
+ // the notes lookup rebuilt the rels path from the NUMBER. With
1232
+ // `slide01.xml` winning, `ppt/slides/_rels/slide1.xml.rels` is the LOSING
1233
+ // part's relationships — so slide 1 either lost its notes or was handed
1234
+ // the other file's.
1235
+ const zip = new AdmZip();
1236
+ zip.addFile('ppt/slides/slide1.xml', Buffer.from(slideXml(['canonical one'])));
1237
+ zip.addFile('ppt/slides/slide01.xml', Buffer.from(slideXml(['zero-padded twin'])));
1238
+ zip.addFile('ppt/slides/_rels/slide01.xml.rels', Buffer.from(relsXml('../notesSlides/notesSlide01.xml')));
1239
+ zip.addFile('ppt/notesSlides/notesSlide01.xml', Buffer.from(slideXml(['the padded twin speaks'])));
1240
+ zip.addFile('ppt/slides/_rels/slide1.xml.rels', Buffer.from(relsXml('../notesSlides/notesSlide1.xml')));
1241
+ zip.addFile('ppt/notesSlides/notesSlide1.xml', Buffer.from(slideXml(['notes of the part that LOST'])));
1242
+ const res = extractPptx(zip.toBuffer());
1243
+ if (!res.ok) throw new Error(res.message);
1244
+ expect(res.text.split('\n')).toEqual([
1245
+ '[slide 1]',
1246
+ 'zero-padded twin',
1247
+ '[slide 1 notes]',
1248
+ 'the padded twin speaks',
1249
+ ]);
1250
+ });
1251
+
1252
+ it('and the no-rels fallback mirrors the name too — slide01.xml pairs with notesSlide01.xml', () => {
1253
+ const zip = new AdmZip();
1254
+ zip.addFile('ppt/slides/slide01.xml', Buffer.from(slideXml(['only slide'])));
1255
+ zip.addFile('ppt/notesSlides/notesSlide01.xml', Buffer.from(slideXml(['padded notes'])));
1256
+ const res = extractPptx(zip.toBuffer());
1257
+ if (!res.ok) throw new Error(res.message);
1258
+ expect(res.text.split('\n')).toEqual(['[slide 1]', 'only slide', '[slide 1 notes]', 'padded notes']);
1259
+ });
1260
+ });
1261
+
1262
+ // ── rels parsing: namespace-prefixed Relationship elements ─────────────────
1263
+
1264
+ describe('notesTargetFromRels — prefixed rels', () => {
1265
+ const NOTES_TYPE = 'http://schemas.openxmlformats.org/officeDocument/2006/relationships/notesSlide';
1266
+
1267
+ it('finds the notes Target on a namespace-prefixed <r:Relationship>', () => {
1268
+ const rels =
1269
+ '<?xml version="1.0"?><r:Relationships xmlns:r="http://schemas.openxmlformats.org/package/2006/relationships">' +
1270
+ `<r:Relationship r:Id="rId2" r:Type="${NOTES_TYPE}" r:Target="../notesSlides/notesSlide9.xml"/>` +
1271
+ '</r:Relationships>';
1272
+ expect(notesTargetFromRels(rels)).toBe('../notesSlides/notesSlide9.xml');
1273
+ });
1274
+
1275
+ it('accepts a NON-ASCII namespace prefix (full XML NCName) — an ASCII-only \\w match dropped these', () => {
1276
+ const rels =
1277
+ '<?xml version="1.0"?><sé:Relationships xmlns:sé="http://schemas.openxmlformats.org/package/2006/relationships">' +
1278
+ `<sé:Relationship sé:Id="rId2" sé:Type="${NOTES_TYPE}" sé:Target="../notesSlides/notesSlide5.xml"/>` +
1279
+ '</sé:Relationships>';
1280
+ expect(notesTargetFromRels(rels)).toBe('../notesSlides/notesSlide5.xml');
1281
+ });
1282
+
1283
+ it('ignores xmlns:Type / xmlns:Target declarations on the Relationship element itself', () => {
1284
+ const rels =
1285
+ '<?xml version="1.0"?><Relationships>' +
1286
+ `<Relationship xmlns:Target="http://ns.example/decl" Id="r1" Type="${NOTES_TYPE}" Target="../notesSlides/notesSlide2.xml"/>` +
1287
+ '</Relationships>';
1288
+ expect(notesTargetFromRels(rels)).toBe('../notesSlides/notesSlide2.xml');
1289
+ });
1290
+
1291
+ it('end-to-end: a pptx whose rels use a prefixed Relationship still pairs its notes', () => {
1292
+ const zip = new AdmZip();
1293
+ zip.addFile(
1294
+ '[Content_Types].xml',
1295
+ Buffer.from('<?xml version="1.0"?><Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"/>'),
1296
+ );
1297
+ zip.addFile('ppt/slides/slide1.xml', Buffer.from(slideXml(['Deck'])));
1298
+ zip.addFile('ppt/notesSlides/notesSlide4.xml', Buffer.from(slideXml(['prefixed note'])));
1299
+ zip.addFile(
1300
+ 'ppt/slides/_rels/slide1.xml.rels',
1301
+ Buffer.from(
1302
+ '<?xml version="1.0"?><r:Relationships xmlns:r="http://schemas.openxmlformats.org/package/2006/relationships">' +
1303
+ `<r:Relationship r:Id="rId2" Type="${NOTES_TYPE}" Target="../notesSlides/notesSlide4.xml"/>` +
1304
+ '</r:Relationships>',
1305
+ ),
1306
+ );
1307
+ const res = extractPptx(zip.toBuffer());
1308
+ if (!res.ok) throw new Error(res.message);
1309
+ expect(res.text).toBe('[slide 1]\nDeck\n[slide 1 notes]\nprefixed note');
1310
+ });
1311
+ });
1312
+
1313
+ // ── quote-aware block delimiters (`/>` inside attribute values) ────────────
1314
+
1315
+ describe('ODF block regexes are quote-aware about /> inside attribute values', () => {
1316
+ it('a draw:page whose attribute value contains /> keeps its body', () => {
1317
+ const res = extractOdp(
1318
+ odfBytes(
1319
+ 'application/vnd.oasis.opendocument.presentation',
1320
+ '<office:presentation>' +
1321
+ '<draw:page draw:name="a/>b" other="x/>y"><draw:frame><text:p>Survived</text:p></draw:frame></draw:page>' +
1322
+ '</office:presentation>',
1323
+ ),
1324
+ );
1325
+ if (!res.ok) throw new Error(res.message);
1326
+ expect(res.text).toBe('[slide 1]\nSurvived');
1327
+ });
1328
+
1329
+ it('a text:p whose attribute value contains /> keeps its body (odt)', () => {
1330
+ const res = extractOdt(odtBytes('<text:p text:style-name="s/>t">Kept</text:p>'));
1331
+ if (!res.ok) throw new Error(res.message);
1332
+ expect(res.text).toBe('Kept');
1333
+ });
1334
+
1335
+ it('ods rows and cells with /> inside attribute values keep their content', () => {
1336
+ const res = extractOds(
1337
+ odsBytes(
1338
+ '<table:table table:name="Q"><table:table-row table:style-name="r/>x">' +
1339
+ '<table:table-cell table:style-name="c/>y"><text:p>alive</text:p></table:table-cell>' +
1340
+ '<table:table-cell><text:p>next</text:p></table:table-cell>' +
1341
+ '</table:table-row></table:table>',
1342
+ ),
1343
+ );
1344
+ if (!res.ok) throw new Error(res.message);
1345
+ expect(res.text.split('\n')).toEqual(['[sheet: Q]', 'alive\tnext']);
1346
+ });
1347
+
1348
+ it('self-closing elements WITH attributes still hit the /> branch after the rewrite', () => {
1349
+ const res = extractOds(
1350
+ odsBytes(
1351
+ '<table:table table:name="S"><table:table-row>' +
1352
+ '<table:table-cell table:number-columns-repeated="2"/>' +
1353
+ '<table:table-cell><text:p>end</text:p></table:table-cell>' +
1354
+ '</table:table-row></table:table>',
1355
+ ),
1356
+ );
1357
+ if (!res.ok) throw new Error(res.message);
1358
+ expect(res.text.split('\n')).toEqual(['[sheet: S]', '\t\tend']);
1359
+ });
1360
+ });
1361
+
1362
+ // ── docx/pptx: the OOXML scanner ───────────────────────────────────────────
1363
+
1364
+ describe('ooxml-text — the docx/pptx walks are linear and quote-aware', () => {
1365
+ /**
1366
+ * The same shape that pinned the ODF extractors, in the OOXML twins that
1367
+ * were deliberately left standing last round. `<w:p(?:\s[^>]*)?>([\s\S]*?)
1368
+ * </w:p>` re-scanned everything to the right from EVERY opener whose close
1369
+ * tag never came, so cost grew as openers x bytes. Measured before the
1370
+ * rewrite: 80k openers (391 KB) took 19.4 s, QUADRUPLING per doubling —
1371
+ * days of pinned CPU at the 50 MB `MAX_DOC_PART_BYTES` cap, from a
1372
+ * .docx/.pptx that zips down to a few kilobytes. `word/document.xml` and
1373
+ * `ppt/slides/slideN.xml` are user-uploaded bytes, so anyone who can add a
1374
+ * file to a knowledge base could reach it.
1375
+ *
1376
+ * Generous by design (seconds): here to catch a RETURN of the quadratic,
1377
+ * not to police CI's scheduler. The scanner does these in single-digit ms.
1378
+ */
1379
+ const GENEROUS_MS = 5_000;
1380
+
1381
+ it('a docx of 80k unmatched <w:p> openers extracts its real paragraph, fast', () => {
1382
+ const bytes = docxBytes('<w:p><w:r><w:t>Kept</w:t></w:r></w:p>' + '<w:p>'.repeat(80_000));
1383
+ const t0 = performance.now();
1384
+ const res = extractDocx(bytes);
1385
+ const ms = performance.now() - t0;
1386
+ if (!res.ok) throw new Error(res.message);
1387
+ expect(res.text).toBe('Kept');
1388
+ expect(ms).toBeLessThan(GENEROUS_MS);
1389
+ });
1390
+
1391
+ it('a pptx of 80k unmatched <a:p> openers still reads its real slide text, fast', () => {
1392
+ const bytes = pptxBytes({
1393
+ 1: `<p:sld ${A_NS}><a:p><a:r><a:t>Alive</a:t></a:r></a:p>${'<a:p algn="ctr">'.repeat(80_000)}</p:sld>`,
1394
+ });
1395
+ const t0 = performance.now();
1396
+ const res = extractPptx(bytes);
1397
+ const ms = performance.now() - t0;
1398
+ if (!res.ok) throw new Error(res.message);
1399
+ expect(res.text.split('\n')).toEqual(['[slide 1]', 'Alive']);
1400
+ expect(ms).toBeLessThan(GENEROUS_MS);
1401
+ });
1402
+
1403
+ it('40k unmatched <w:t> RUN openers inside one paragraph stay linear too', () => {
1404
+ // paragraphRunText carried the identical pattern, one level down.
1405
+ const bytes = docxBytes(
1406
+ `<w:p><w:r><w:t>Kept</w:t></w:r>${'<w:t xml:space="preserve">'.repeat(40_000)}</w:p>`,
1407
+ );
1408
+ const t0 = performance.now();
1409
+ const res = extractDocx(bytes);
1410
+ const ms = performance.now() - t0;
1411
+ if (!res.ok) throw new Error(res.message);
1412
+ expect(res.text).toBe('Kept');
1413
+ expect(ms).toBeLessThan(GENEROUS_MS);
1414
+ });
1415
+
1416
+ it('an UNTERMINATED attribute quote ends the scan instead of restarting it per opener', () => {
1417
+ // The second failure mode: quote-awareness means an unclosed quote runs to
1418
+ // the end of the part, so without the dead-opener memo every opener would
1419
+ // pay a full scan to discover that. Measured at 40k openers before the
1420
+ // memo existed in this walk: 85 s.
1421
+ const bytes = docxBytes(
1422
+ '<w:p><w:r><w:t>Kept</w:t></w:r></w:p>' + '<w:p w:rsidR="never closed'.repeat(40_000),
1423
+ );
1424
+ const t0 = performance.now();
1425
+ const res = extractDocx(bytes);
1426
+ const ms = performance.now() - t0;
1427
+ if (!res.ok) throw new Error(res.message);
1428
+ expect(res.text).toBe('Kept');
1429
+ expect(ms).toBeLessThan(GENEROUS_MS);
1430
+ });
1431
+
1432
+ it('a wall of openers whose attribute scan ends at a FAR close tag stays linear', () => {
1433
+ // The shape a failure-only memo misses. Here every opener's attribute scan
1434
+ // reaches the `>` of the `</w:body>` past the wall, so every scan SUCCEEDS
1435
+ // — nothing is recorded as dead — and each one is then rejected by the
1436
+ // `lastIndexOf` guard, at the cost of a full traversal apiece. Measured on
1437
+ // the ODF twin, which shipped with exactly this hole: 740 KB took 44.6 s
1438
+ // and quadrupled per doubling. The memo now records successes too.
1439
+ const bytes = docxBytes(
1440
+ '<w:p><w:r><w:t>Kept</w:t></w:r></w:p>' + '<w:p w:rsidR="never closed'.repeat(20_000),
1441
+ );
1442
+ const t0 = performance.now();
1443
+ const res = extractDocx(bytes);
1444
+ const ms = performance.now() - t0;
1445
+ if (!res.ok) throw new Error(res.message);
1446
+ expect(res.text).toBe('Kept');
1447
+ expect(ms).toBeLessThan(GENEROUS_MS);
1448
+ });
1449
+
1450
+ it('CHANGED: a self-closing <w:p/> is an EMPTY paragraph, not a dropped one', () => {
1451
+ // Before the rewrite the pattern admitted no `/>` branch, so `<w:p/>`
1452
+ // matched nothing and vanished from the output entirely — an empty line
1453
+ // the author wrote simply disappeared. It now reads as the empty
1454
+ // paragraph it is, which is what `<w:p></w:p>` has always produced and
1455
+ // what the ODF scanner does for `<text:p/>`.
1456
+ const res = extractDocx(
1457
+ docxBytes('<w:p><w:r><w:t>A</w:t></w:r></w:p><w:p/><w:p><w:r><w:t>B</w:t></w:r></w:p>'),
1458
+ );
1459
+ if (!res.ok) throw new Error(res.message);
1460
+ expect(res.text.split('\n')).toEqual(['A', '', 'B']);
1461
+ expect(res.summary).toContain('3 paragraphs');
1462
+ });
1463
+
1464
+ it('CHANGED: a self-closing <w:p/> WITH attributes is an empty paragraph too', () => {
1465
+ const res = extractDocx(
1466
+ docxBytes('<w:p w:rsidR="00A1"/><w:p><w:r><w:t>B</w:t></w:r></w:p>'),
1467
+ );
1468
+ if (!res.ok) throw new Error(res.message);
1469
+ expect(res.text.split('\n')).toEqual(['', 'B']);
1470
+ });
1471
+
1472
+ it('CHANGED: a self-closing <w:tc/> is an EMPTY table cell, not a dropped column', () => {
1473
+ const res = extractDocx(
1474
+ docxBytes(
1475
+ '<w:tbl><w:tr><w:tc/><w:tc><w:p><w:r><w:t>b</w:t></w:r></w:p></w:tc></w:tr></w:tbl>',
1476
+ ),
1477
+ );
1478
+ if (!res.ok) throw new Error(res.message);
1479
+ expect(res.text.split('\n')).toEqual(['\tb']);
1480
+ });
1481
+
1482
+ it('CHANGED: a `>` inside a quoted attribute no longer leaks into the text', () => {
1483
+ // `[^>]*` stopped at the FIRST `>`, even inside a quoted value, so the tag
1484
+ // was truncated: the block boundary landed mid-attribute and the attribute
1485
+ // tail was emitted as document text. This paragraph used to extract as
1486
+ // ` b">Hi` — the run property's value and a stray quote in the body.
1487
+ const res = extractDocx(
1488
+ docxBytes('<w:p><w:r><w:t w:val="a &gt; b" w:x="a > b">Hi</w:t></w:r></w:p>'),
1489
+ );
1490
+ if (!res.ok) throw new Error(res.message);
1491
+ expect(res.text).toBe('Hi');
1492
+ });
1493
+
1494
+ it('CHANGED: a quoted `>` on the PARAGRAPH tag keeps the block boundary right', () => {
1495
+ const res = extractDocx(
1496
+ docxBytes('<w:p w:rsidR="a > b"><w:r><w:t>Hi</w:t></w:r></w:p>'),
1497
+ );
1498
+ if (!res.ok) throw new Error(res.message);
1499
+ expect(res.text).toBe('Hi');
1500
+ });
1501
+
1502
+ it('a pptx paragraph with a quoted `>` and a self-closing empty one read correctly', () => {
1503
+ // pptx DROPS blank paragraphs (a deck's outline is its non-empty lines),
1504
+ // so `<a:p/>` stays invisible there — but the quoted `>` used to prepend
1505
+ // `q">` to the slide's text.
1506
+ const res = extractPptx(
1507
+ pptxBytes({
1508
+ 1: `<p:sld ${A_NS}><a:p/><a:p><a:r><a:t x="p>q">Hi</a:t></a:r></a:p></p:sld>`,
1509
+ }),
1510
+ );
1511
+ if (!res.ok) throw new Error(res.message);
1512
+ expect(res.text.split('\n')).toEqual(['[slide 1]', 'Hi']);
1513
+ });
1514
+
1515
+ it('still refuses to read <w:pPr> as a <w:p>, and RECOVERS a dangling opener', () => {
1516
+ // A longer name that merely STARTS with the wanted one is still not a
1517
+ // match. CHANGED with the parser: an element left open at end of input is
1518
+ // closed implicitly, so `<w:p>dangling` is a paragraph holding its text
1519
+ // rather than an opener dropped on the floor. Recovering the characters a
1520
+ // truncated document does contain beats discarding them.
1521
+ const res = extractDocx(
1522
+ docxBytes('<w:pPr><w:r><w:t>props</w:t></w:r></w:pPr><w:p><w:r><w:t>real</w:t></w:r></w:p><w:p>dangling'),
1523
+ );
1524
+ if (!res.ok) throw new Error(res.message);
1525
+ expect(res.text).toBe('real');
1526
+ expect(res.summary).toContain('2 paragraphs');
1527
+ });
1528
+
1529
+ it('CHANGED: a malformed `<w:t/x>` is read the way a parser reads it', () => {
1530
+ // The hand-rolled scanner ruled that `/` ends a name only as the `/` of a
1531
+ // `/>`, so `<w:t/x>` named nothing and its content was withheld.
1532
+ // htmlparser2 recovers instead — `<w:t x>` — and the text inside comes
1533
+ // out. Neither is "wrong" for markup this broken, and recovery is the
1534
+ // behaviour of the parser the rest of the world reads these files with.
1535
+ const res = extractDocx(docxBytes('<w:p><w:r><w:t/x>leaked</w:t><w:t>kept</w:t></w:r></w:p>'));
1536
+ if (!res.ok) throw new Error(res.message);
1537
+ expect(res.text).toBe('leakedkept');
1538
+ });
1539
+
1540
+ it('an attribute quote that never closes still costs two traversals, not one per opener', () => {
1541
+ // Cubic's round-5 note on the (now deleted) ODF scanner: a `<` met INSIDE
1542
+ // an unterminated quote is recorded by nobody, so — the worry went — every
1543
+ // opener behind it rescans the document. It does not. The FIRST scan that
1544
+ // begins OUTSIDE the quote walks that same tail in the no-quote state and
1545
+ // records every one of them, so the wall costs two traversals in total.
1546
+ // Measured on the shared scanner at 3.2 M openers (15.6 MB): 1.78 s,
1547
+ // growing ~2.6x per doubling — linear plus the memo's own GC cost, not the
1548
+ // 4x a quadratic gives. With the memo bounded (see MAX_TAG_MEMO): 61 ms.
1549
+ const bytes = docxBytes(
1550
+ '<w:p><w:r><w:t>Kept</w:t></w:r></w:p><w:p w:rsidR="' + '<w:p '.repeat(40_000),
1551
+ );
1552
+ const t0 = performance.now();
1553
+ const res = extractDocx(bytes);
1554
+ const ms = performance.now() - t0;
1555
+ if (!res.ok) throw new Error(res.message);
1556
+ expect(res.text).toBe('Kept');
1557
+ expect(ms).toBeLessThan(GENEROUS_MS);
1558
+ });
1559
+
1560
+ it('a wall of unterminated openers is bounded, and the real paragraphs survive', () => {
1561
+ // 300 k openers that never close. The hand-rolled scanner needed a tag-end
1562
+ // memo to stay linear here, then a cap on the memo because it reached
1563
+ // 1170 MB on a 50 MB part, and the cap then had to abandon the tail. The
1564
+ // parser needs none of it: the wall is one unterminated tag, and the
1565
+ // paragraphs on either side of it both come out.
1566
+ const bytes = docxBytes(para('first') + '<w:p '.repeat(300_000) + para('last'));
1567
+ const t0 = performance.now();
1568
+ const res = extractDocx(bytes);
1569
+ const ms = performance.now() - t0;
1570
+ if (!res.ok) throw new Error(res.message);
1571
+ expect(res.text).toBe('first\nlast');
1572
+ expect(ms).toBeLessThan(GENEROUS_MS);
1573
+ });
1574
+
1575
+ it('a commented-out paragraph is not extracted, however big the comment', () => {
1576
+ // A comment is WELL-FORMED xml and may say anything, `<w:p>` included.
1577
+ // Reading its insides as markup extracted a commented paragraph as if it
1578
+ // were document text — and, in the scanner this replaced, charged every
1579
+ // `<` in it to a memo whose cap then dropped everything after the comment.
1580
+ const small = extractDocx(docxBytes(`<!-- ${para('ghost')} -->${para('real')}`));
1581
+ if (!small.ok) throw new Error(small.message);
1582
+ expect(small.text).toBe('real');
1583
+
1584
+ const bytes = docxBytes(`<!-- ${'<w:p '.repeat(300_000)} -->${para('survives')}`);
1585
+ const t0 = performance.now();
1586
+ const res = extractDocx(bytes);
1587
+ const ms = performance.now() - t0;
1588
+ if (!res.ok) throw new Error(res.message);
1589
+ expect(res.text).toBe('survives');
1590
+ expect(ms).toBeLessThan(GENEROUS_MS);
1591
+ });
1592
+
1593
+ it('a `</w:p>` inside a comment or CDATA does not end the paragraph holding it', () => {
1594
+ const commented = extractDocx(
1595
+ docxBytes('<w:p><w:r><w:t>a</w:t></w:r><!-- </w:p> --><w:r><w:t>b</w:t></w:r></w:p>'),
1596
+ );
1597
+ if (!commented.ok) throw new Error(commented.message);
1598
+ expect(commented.text).toBe('ab');
1599
+ expect(commented.summary).toContain('1 paragraph');
1600
+
1601
+ const cdata = extractDocx(
1602
+ docxBytes('<w:p><w:r><w:t>a</w:t></w:r><![CDATA[</w:p>]]><w:r><w:t>b</w:t></w:r></w:p>'),
1603
+ );
1604
+ if (!cdata.ok) throw new Error(cdata.message);
1605
+ expect(cdata.text).toBe('ab');
1606
+ });
1607
+
1608
+ it('CDATA and processing instructions are text, not markup', () => {
1609
+ const res = extractDocx(
1610
+ docxBytes(`<![CDATA[${para('ghost')}]]><?xml-stylesheet href="x"?>${para('real')}`),
1611
+ );
1612
+ if (!res.ok) throw new Error(res.message);
1613
+ expect(res.text).toBe('real');
1614
+ expect(res.summary).toContain('1 paragraph');
1615
+ });
1616
+
1617
+ it('stays fast on an ORDINARY document — 60k plain paragraphs', () => {
1618
+ // Worth keeping through the rewrite: an earlier attempt at the comment rule
1619
+ // asked "does a section start before this close tag?" with a scan per
1620
+ // element, which on a document containing no comments at all — nearly every
1621
+ // real document — made ordinary extraction quadratic.
1622
+ const bytes = docxBytes(para('x').repeat(60_000));
1623
+ const t0 = performance.now();
1624
+ const res = extractDocx(bytes);
1625
+ const ms = performance.now() - t0;
1626
+ if (!res.ok) throw new Error(res.message);
1627
+ expect(res.summary).toContain('60000 paragraphs');
1628
+ expect(ms).toBeLessThan(GENEROUS_MS);
1629
+ });
1630
+
1631
+ it('reads a document full of comments without an index to blow up on', () => {
1632
+ // `<!---->` is seven bytes, so a 50 MB part spells seven million valid
1633
+ // comments. Indexing their spans to answer "is this close tag commented
1634
+ // out?" allocated per comment and had to be capped, which gave up on parts
1635
+ // that were merely comment-heavy. The parser tracks no such thing.
1636
+ const bytes = docxBytes(para('first') + '<!---->'.repeat(200_000) + para('last'));
1637
+ const t0 = performance.now();
1638
+ const res = extractDocx(bytes);
1639
+ const ms = performance.now() - t0;
1640
+ if (!res.ok) throw new Error(res.message);
1641
+ expect(res.text).toBe('first\nlast');
1642
+ expect(ms).toBeLessThan(GENEROUS_MS);
1643
+ });
1644
+
1645
+ it('stays bounded when 50k paragraph openers nest without ever closing', () => {
1646
+ // Every close tag here sits inside a comment, so none of them closes
1647
+ // anything and the openers nest 50 k deep. The nesting is what costs: the
1648
+ // parser holds a stack entry per open element and its own cost climbs past
1649
+ // linear on an enormous one, so MAX_ELEMENT_DEPTH stops the parse — with
1650
+ // the text reached so far kept rather than discarded.
1651
+ const bytes = docxBytes('<w:p>'.repeat(50_000) + '<!-- </w:p> -->'.repeat(2_000));
1652
+ const t0 = performance.now();
1653
+ const res = extractDocx(bytes);
1654
+ const ms = performance.now() - t0;
1655
+ if (!res.ok) throw new Error(res.message);
1656
+ expect(ms).toBeLessThan(GENEROUS_MS);
1657
+ });
1658
+ });