@bevel-software/platform-core-backend 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/THIRD-PARTY-NOTICES.md +1163 -425
  2. package/dist/core/create-core-server.js +1 -1
  3. package/dist/core/create-core-server.js.map +1 -1
  4. package/dist/core/create-core-services.d.ts +2 -0
  5. package/dist/core/create-core-services.d.ts.map +1 -1
  6. package/dist/core/create-core-services.js +5 -0
  7. package/dist/core/create-core-services.js.map +1 -1
  8. package/dist/core-config.d.ts +7 -0
  9. package/dist/core-config.d.ts.map +1 -1
  10. package/dist/core-config.js +9 -0
  11. package/dist/core-config.js.map +1 -1
  12. package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
  13. package/dist/modules/code-mode/code-mode.tool.js +7 -1
  14. package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
  15. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
  16. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
  17. package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
  18. package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
  19. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
  20. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
  21. package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
  22. package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
  23. package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
  24. package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
  25. package/dist/modules/workspace/file-readers/document-reader.js +59 -0
  26. package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
  27. package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
  28. package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
  29. package/dist/modules/workspace/file-readers/email-reader.js +19 -0
  30. package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
  31. package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
  32. package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
  33. package/dist/modules/workspace/file-readers/email-text.js +151 -0
  34. package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
  35. package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
  36. package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
  37. package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
  38. package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
  39. package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
  40. package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
  41. package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
  42. package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
  43. package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
  44. package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
  45. package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
  46. package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
  47. package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
  48. package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
  49. package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
  50. package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
  51. package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
  52. package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
  53. package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
  54. package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
  55. package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
  56. package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
  57. package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
  58. package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
  59. package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
  60. package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
  61. package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
  62. package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
  63. package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
  64. package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
  65. package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
  66. package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
  67. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
  68. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
  69. package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
  70. package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
  71. package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
  72. package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
  73. package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
  74. package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
  75. package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
  76. package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
  77. package/dist/modules/workspace/file-readers/file-reader.js +55 -0
  78. package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
  79. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
  80. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
  81. package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
  82. package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
  83. package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
  84. package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
  85. package/dist/modules/workspace/file-readers/image-read.js +108 -0
  86. package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
  87. package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
  88. package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
  89. package/dist/modules/workspace/file-readers/image-reader.js +30 -0
  90. package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
  91. package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
  92. package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
  93. package/dist/modules/workspace/file-readers/odf-text.js +116 -0
  94. package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
  95. package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
  96. package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
  97. package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
  98. package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
  99. package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
  100. package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
  101. package/dist/modules/workspace/file-readers/text-reader.js +117 -0
  102. package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
  103. package/dist/modules/workspace/workspace.tools.d.ts +2 -1
  104. package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
  105. package/dist/modules/workspace/workspace.tools.js +158 -15
  106. package/dist/modules/workspace/workspace.tools.js.map +1 -1
  107. package/package.json +9 -4
  108. package/src/core/create-core-server.ts +1 -1
  109. package/src/core/create-core-services.ts +6 -0
  110. package/src/core-config.ts +9 -0
  111. package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
  112. package/src/modules/code-mode/code-mode.tool.ts +7 -1
  113. package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
  114. package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
  115. package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
  116. package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
  117. package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
  118. package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
  119. package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
  120. package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
  121. package/src/modules/workspace/file-readers/document-reader.ts +64 -0
  122. package/src/modules/workspace/file-readers/email-reader.ts +21 -0
  123. package/src/modules/workspace/file-readers/email-text.ts +193 -0
  124. package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
  125. package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
  126. package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
  127. package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
  128. package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
  129. package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
  130. package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
  131. package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
  132. package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
  133. package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
  134. package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
  135. package/src/modules/workspace/file-readers/file-reader.ts +104 -0
  136. package/src/modules/workspace/file-readers/image-read.ts +122 -0
  137. package/src/modules/workspace/file-readers/image-reader.ts +39 -0
  138. package/src/modules/workspace/file-readers/odf-text.ts +123 -0
  139. package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
  140. package/src/modules/workspace/file-readers/text-reader.ts +131 -0
  141. package/src/modules/workspace/workspace.tools.ts +174 -12
@@ -0,0 +1,477 @@
1
+ /**
2
+ * XML text helpers shared by the docx, pptx and ODF extractors.
3
+ *
4
+ * The scanning here is `htmlparser2` in XML mode. It used to be hand-rolled,
5
+ * on the reasoning that the extractors only need "the character content of
6
+ * `<w:t>`/`<a:t>` runs, grouped by paragraph" and a parser dependency was
7
+ * heavyweight for a single linear scan. That reasoning had one flaw: the input
8
+ * is UPLOADED, so the scan has to be right about all of XML's lexical rules
9
+ * and not merely the ones a well-formed document exercises. It was not — a `>`
10
+ * inside a quoted attribute value, a `/` that ends a name only as part of
11
+ * `/>`, a `</w:p>` written inside a comment or a CDATA section, a namespace
12
+ * prefix outside ASCII — and the machinery each fix needed (a tag-end memo, a
13
+ * section index, caps on both) grew defects of its own, twice worse than what
14
+ * it was fixing. A parser knows those rules already.
15
+ */
16
+ import { Parser } from 'htmlparser2';
17
+ import type AdmZip from 'adm-zip';
18
+
19
+ /**
20
+ * Decompression bounds for the document extractors (OOXML, ODF and — as a
21
+ * plain byte cap — PDF). A zip's central directory declares each entry's
22
+ * UNCOMPRESSED size, so a zip bomb (a few KB that inflate to gigabytes) is
23
+ * detectable BEFORE any inflation happens; 50 MB of XML is far beyond any
24
+ * real office document part (a huge deck's slide parts run to single-digit
25
+ * MB) while staying well inside what a server can afford to decode.
26
+ * `MAX_DOC_TOTAL_BYTES` additionally bounds the SUM of the parts a multi-part
27
+ * extraction reads (pptx slides/notes, xlsx sheet parts) at 200 MB.
28
+ */
29
+ export const MAX_DOC_PART_BYTES = 50 * 1024 * 1024; // 50 MB uncompressed, per part
30
+ export const MAX_DOC_TOTAL_BYTES = 200 * 1024 * 1024; // 200 MB uncompressed, per document
31
+
32
+ /**
33
+ * The typed-failure fragment for a zip entry whose DECLARED uncompressed size
34
+ * exceeds {@link MAX_DOC_PART_BYTES}, or null when the entry is within bounds.
35
+ * Checked against the central-directory header BEFORE `getData()` inflates
36
+ * anything, so an oversized (or bomb) entry costs nothing.
37
+ */
38
+ export function zipEntryOversize(entry: AdmZip.IZipEntry): string | null {
39
+ const size = entry.header.size;
40
+ return size > MAX_DOC_PART_BYTES
41
+ ? `${entry.entryName} is ${size} bytes uncompressed — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit`
42
+ : null;
43
+ }
44
+
45
+ // (The quote-aware `TAG_ATTRS` regex fragment used to live here. Every reader
46
+ // that built a tag pattern from it — the email strip, the ODF paragraph, page,
47
+ // row and cell walks — now uses a single-pass scanner instead: lazily expanding
48
+ // that fragment re-scanned the rest of the document from every opener that
49
+ // failed to match, which turned a crafted upload into minutes of pinned CPU.
50
+ // See `htmlToEmailText` and `xmlElementBlocks`.)
51
+
52
+ /**
53
+ * Regex FRAGMENT matching one XML NCName — the legal shape of a namespace
54
+ * prefix. `\w` would be wrong here: XML names admit most of Unicode (letters,
55
+ * combining marks, …), and a producer is free to bind a namespace to a
56
+ * non-ASCII prefix — an ASCII-only prefix match would silently drop such
57
+ * elements. Astral characters ride along as surrogate pairs so the fragment
58
+ * works without the `u` flag.
59
+ */
60
+ const NC_START =
61
+ 'A-Za-z_' +
62
+ '\u00C0-\u00D6\u00D8-\u00F6\u00F8-\u02FF\u0370-\u037D\u037F-\u1FFF\u200C-\u200D' +
63
+ '\u2070-\u218F\u2C00-\u2FEF\u3001-\uD7FF\uF900-\uFDCF\uFDF0-\uFFFD';
64
+ const NC_EXTRA = '0-9.\u00B7\u0300-\u036F\u203F-\u2040-'; // dash LAST: literal in the class, never a range
65
+ export const XML_NCNAME =
66
+ `(?:[${NC_START}]|[\uD800-\uDB7F][\uDC00-\uDFFF])` +
67
+ `(?:[${NC_START}${NC_EXTRA}]|[\uD800-\uDB7F][\uDC00-\uDFFF])*`;
68
+
69
+ /**
70
+ * One tag's attributes as `name → raw value` tokens, in document order. A
71
+ * real left-to-right tokenizer, not a regex probe: quoted values (either
72
+ * quote style, whitespace around `=` tolerated) are skipped over WHOLE, so a
73
+ * `target='…'`-looking sequence INSIDE another attribute's value can never
74
+ * be mistaken for an attribute of its own. Values are RAW (entities not
75
+ * decoded); a malformed tail (unterminated quote) simply ends the scan.
76
+ */
77
+ export function xmlAttrTokens(tagXml: string): Array<{ name: string; value: string }> {
78
+ const out: Array<{ name: string; value: string }> = [];
79
+ let i = 0;
80
+ // Skip '<' (with an optional '/' or '?') and the tag name itself.
81
+ if (tagXml[i] === '<') {
82
+ i++;
83
+ if (tagXml[i] === '/' || tagXml[i] === '?') i++;
84
+ }
85
+ while (i < tagXml.length && !/[\s/>]/.test(tagXml[i])) i++;
86
+ while (i < tagXml.length) {
87
+ while (i < tagXml.length && /[\s/]/.test(tagXml[i])) i++;
88
+ if (i >= tagXml.length || tagXml[i] === '>') return out;
89
+ const nameStart = i;
90
+ while (i < tagXml.length && !/[\s=/>]/.test(tagXml[i])) i++;
91
+ const name = tagXml.slice(nameStart, i);
92
+ while (i < tagXml.length && /\s/.test(tagXml[i])) i++;
93
+ if (tagXml[i] !== '=') continue; // no value (not legal XML) — skip the token
94
+ i++;
95
+ while (i < tagXml.length && /\s/.test(tagXml[i])) i++;
96
+ const quote = tagXml[i];
97
+ if (quote !== '"' && quote !== "'") return out; // unquoted/malformed — stop
98
+ const valueStart = ++i;
99
+ const end = tagXml.indexOf(quote, i);
100
+ if (end === -1) return out; // unterminated quote — stop
101
+ if (name !== '') out.push({ name, value: tagXml.slice(valueStart, end) });
102
+ i = end + 1;
103
+ }
104
+ return out;
105
+ }
106
+
107
+ /**
108
+ * The value of attribute `name` inside one tag's text, or undefined. Exact
109
+ * (prefix-included) name match over the {@link xmlAttrTokens} scan — see there
110
+ * for the quoting guarantees. The value is returned RAW (entities not
111
+ * decoded); callers decode where display matters.
112
+ */
113
+ export function xmlAttrValue(tagXml: string, name: string): string | undefined {
114
+ for (const attr of xmlAttrTokens(tagXml)) {
115
+ if (attr.name === name) return attr.value;
116
+ }
117
+ return undefined;
118
+ }
119
+
120
+ /**
121
+ * Like {@link xmlAttrValue}, but matching the attribute's LOCAL name — the
122
+ * part after any namespace prefix. For parsers that scan by local element
123
+ * name (OPC `.rels` parts, whose producer is free to prefix the relationship
124
+ * namespace) and must accept `r:Target` wherever `Target` is meant.
125
+ */
126
+ export function xmlAttrValueByLocalName(tagXml: string, localName: string): string | undefined {
127
+ for (const attr of xmlAttrTokens(tagXml)) {
128
+ // `xmlns="…"` / `xmlns:Foo="…"` are namespace DECLARATIONS, not attributes
129
+ // — under local-name matching, `xmlns:Target` would otherwise read as a
130
+ // `Target` attribute and hand back a namespace URI.
131
+ if (attr.name === 'xmlns' || attr.name.startsWith('xmlns:')) continue;
132
+ if (attr.name.slice(attr.name.lastIndexOf(':') + 1) === localName) return attr.value;
133
+ }
134
+ return undefined;
135
+ }
136
+
137
+ /**
138
+ * Is `code` a character XML actually admits (the `Char` production)?
139
+ *
140
+ * Being inside Unicode's range is not enough: XML forbids NUL and the other
141
+ * C0 controls, the surrogate halves, and U+FFFE/U+FFFF. `&#0;` and `&#xD800;`
142
+ * used to decode anyway, putting a character into extracted text that no XML
143
+ * document can contain — and, for a lone surrogate, one that cannot even be
144
+ * encoded. Such a reference stays literal instead.
145
+ */
146
+ function isXmlChar(code: number): boolean {
147
+ if (!Number.isFinite(code)) return false;
148
+ return (
149
+ code === 0x9 ||
150
+ code === 0xa ||
151
+ code === 0xd ||
152
+ (code >= 0x20 && code <= 0xd7ff) ||
153
+ (code >= 0xe000 && code <= 0xfffd) ||
154
+ (code >= 0x10000 && code <= 0x10ffff)
155
+ );
156
+ }
157
+
158
+ /**
159
+ * Decode the five XML named entities plus numeric (`&#65;` / `&#x41;`)
160
+ * references. Decimal references admit ONLY decimal digits and hex digits only
161
+ * after `#x` — a malformed `&#12A;` must stay literal text, not be consumed
162
+ * with `parseInt` silently stopping at the `A` and emitting U+000C.
163
+ */
164
+ export function decodeXmlEntities(s: string): string {
165
+ return s.replace(/&(amp|lt|gt|quot|apos|#(?:[0-9]+|x[0-9a-fA-F]+));/g, (whole, body: string) => {
166
+ switch (body) {
167
+ case 'amp':
168
+ return '&';
169
+ case 'lt':
170
+ return '<';
171
+ case 'gt':
172
+ return '>';
173
+ case 'quot':
174
+ return '"';
175
+ case 'apos':
176
+ return "'";
177
+ default: {
178
+ const code = body[1] === 'x' ? parseInt(body.slice(2), 16) : parseInt(body.slice(1), 10);
179
+ return isXmlChar(code) ? String.fromCodePoint(code) : whole;
180
+ }
181
+ }
182
+ });
183
+ }
184
+
185
+ /** One element found by {@link xmlElementBlocks}. */
186
+ /** One element found by {@link xmlElementBlocks}. */
187
+ export interface XmlElementBlock {
188
+ /** The qualified name as written, e.g. `w:p` — which of `names` matched. */
189
+ name: string;
190
+ /** The element's attributes. Values are RAW (entities not decoded). */
191
+ attributes: Record<string, string>;
192
+ /** The body between `>` and the matching close tag; undefined when self-closing. */
193
+ body: string | undefined;
194
+ /** Index of the element's opening `<` in the scanned string. */
195
+ start: number;
196
+ /** One past the element's final `>` (as far as the parse got, for a block cut short by the depth cap). */
197
+ end: number;
198
+ }
199
+
200
+ /**
201
+ * How deep the element stack may go before a part is given up on.
202
+ *
203
+ * Real office XML nests a few dozen levels. A crafted part can nest as deep as
204
+ * it has bytes, and the parser's own cost climbs faster than linearly once the
205
+ * stack is enormous: measured on unclosed `<a:p>` openers, 40 k deep took
206
+ * 136 ms and 80 k took 2.7 s. A 50 MB part could spell millions. So the depth
207
+ * is bounded far above any real document and far below where that curve bites.
208
+ *
209
+ * On reaching it the parse stops and what was found is returned — including
210
+ * the element still open, whose body is taken as far as the parse got, so a
211
+ * paragraph holding real text before the crafted tail still yields that text.
212
+ */
213
+ export const MAX_ELEMENT_DEPTH = 1_000;
214
+
215
+ /**
216
+ * Thrown to stop the parse at {@link MAX_ELEMENT_DEPTH}; never escapes an
217
+ * extractor (also shared by `email-text.ts`'s HTML strip).
218
+ */
219
+ export const TOO_DEEP = Symbol('too deep');
220
+
221
+ /** Thrown to stop the scan when a `visit` callback returns true; never escapes. */
222
+ const STOP_SCAN = Symbol('stop scan');
223
+
224
+ /** The match currently being collected by {@link xmlElementBlocks}. */
225
+ interface OpenMatch {
226
+ name: string;
227
+ attributes: Record<string, string>;
228
+ depth: number;
229
+ /** Index of the open tag's `<`. */
230
+ start: number;
231
+ /** Index of the `>` that ends the open tag — the body starts one past it. */
232
+ tagEnd: number;
233
+ }
234
+
235
+ /**
236
+ * The `<name …>…</name>` and self-closing `<name …/>` elements named by
237
+ * `names`, in document order, at ANY depth — but never descending into a
238
+ * match, since a match nested inside another is part of that one's body
239
+ * rather than a block of its own.
240
+ *
241
+ * Parsing is `htmlparser2` in XML mode, which is the point of this function.
242
+ * What stood here before was a hand-rolled scanner, and the lexical rules it
243
+ * had to know kept turning out to be one rule short: a `>` inside a quoted
244
+ * attribute value, a `/` that only ends a name as part of `/>`, a `</w:p>`
245
+ * written inside a comment or a CDATA section, a namespace prefix outside
246
+ * ASCII. Each gap was a real defect, several were reachable from an uploaded
247
+ * file, and the fixes needed their own bookkeeping — a tag-end memo, a section
248
+ * index, caps on both — which then had defects of their own. All of that is
249
+ * the parser's job here, and it is code with far more mileage than ours.
250
+ *
251
+ * Bodies are RAW slices of `xml` (entities not decoded), because the callers
252
+ * re-scan them for nested elements and decode only the text they keep.
253
+ */
254
+ /** The part of a qualified XML name after its namespace prefix. */
255
+ export function localName(qualified: string): string {
256
+ return qualified.slice(qualified.lastIndexOf(':') + 1);
257
+ }
258
+
259
+ /**
260
+ * The value of the attribute whose LOCAL name is `want`, or undefined.
261
+ * Namespace DECLARATIONS are not attributes and never answer for one.
262
+ */
263
+ export function attrByLocalName(
264
+ attributes: Record<string, string>,
265
+ want: string,
266
+ ): string | undefined {
267
+ for (const [key, value] of Object.entries(attributes)) {
268
+ if (key === 'xmlns' || key.startsWith('xmlns:')) continue;
269
+ if (localName(key) === want) return value;
270
+ }
271
+ return undefined;
272
+ }
273
+
274
+ /**
275
+ * {@link xmlElementBlocks}, matching each element's LOCAL name instead of the
276
+ * qualified one — `p` finds `<w:p>`, `<a:p>` and an unprefixed `<p>` alike.
277
+ *
278
+ * Prefixes are a document's own choice: XML binds them to namespace URIs, and
279
+ * a producer may bind any prefix it likes or default the namespace and use
280
+ * none. Naming `a:p` or `text:p` literally therefore read only the documents
281
+ * whose authors happened to pick the usual prefix — a valid deck using `d:p`
282
+ * for DrawingML extracted as EMPTY, and an ODT that defaulted the text
283
+ * namespace found no paragraphs at all. Matching the local name reads both,
284
+ * and replaces the prefix-rewriting pass the ODF readers used to run over
285
+ * every document to paper over the same problem.
286
+ */
287
+ export function localElementBlocks(xml: string, localNames: readonly string[]): XmlElementBlock[] {
288
+ const wanted = new Set(localNames);
289
+ return xmlElementBlocks(xml, wanted, (name) => wanted.has(localName(name)));
290
+ }
291
+
292
+ /**
293
+ * {@link localElementBlocks} as a WALK: `visit` receives each element as its
294
+ * close tag is reached, and returning true STOPS the scan — the input past
295
+ * that element is never parsed and no block is materialized beyond it. For
296
+ * callers with a cap (the ods row walk): collecting every block into an array
297
+ * before consulting the cap let an accepted document allocate its whole
298
+ * expansion first.
299
+ */
300
+ export function walkLocalElementBlocks(
301
+ xml: string,
302
+ localNames: readonly string[],
303
+ visit: (block: XmlElementBlock) => boolean | void,
304
+ ): void {
305
+ const wanted = new Set(localNames);
306
+ xmlElementBlocks(xml, wanted, (name) => wanted.has(localName(name)), visit);
307
+ }
308
+
309
+ /**
310
+ * `xml` with every element named by `localNames` (matched on its LOCAL name)
311
+ * removed WHOLE — open tag through matching close tag — by the parsed block
312
+ * boundaries. The structural counterpart to string replacement, which deleted
313
+ * every occurrence of a block's serialized BODY: a slide whose visible text
314
+ * happened to serialize identically to its notes lost that text too.
315
+ */
316
+ export function removeLocalElements(xml: string, localNames: readonly string[]): string {
317
+ const blocks = localElementBlocks(xml, localNames);
318
+ if (blocks.length === 0) return xml;
319
+ let out = '';
320
+ let at = 0;
321
+ // Blocks arrive in document order and never overlap (the scan does not
322
+ // descend into a match); the max() is belt and braces.
323
+ for (const { start, end } of blocks) {
324
+ out += xml.slice(at, Math.max(at, start));
325
+ at = Math.max(at, end);
326
+ }
327
+ return out + xml.slice(at);
328
+ }
329
+
330
+ export function xmlElementBlocks(
331
+ xml: string,
332
+ names: Iterable<string>,
333
+ /** Overrides name matching — {@link localElementBlocks} matches local names with it. */
334
+ matches?: (name: string) => boolean,
335
+ /**
336
+ * Streaming hook — see {@link walkLocalElementBlocks}. When given, each
337
+ * block is handed to it INSTEAD of being accumulated (the return value is
338
+ * then an empty array), and returning true stops the scan.
339
+ */
340
+ visit?: (block: XmlElementBlock) => boolean | void,
341
+ ): XmlElementBlock[] {
342
+ const wanted = new Set(names);
343
+ const isWanted = matches ?? ((name: string): boolean => wanted.has(name));
344
+ const out: XmlElementBlock[] = [];
345
+ /** Hand a completed block over; true means the visitor asked to stop. */
346
+ const emit = (block: XmlElementBlock): boolean => {
347
+ if (visit !== undefined) return visit(block) === true;
348
+ out.push(block);
349
+ return false;
350
+ };
351
+ let depth = 0;
352
+ let open: OpenMatch | null = null;
353
+ const parser: Parser = new Parser(
354
+ {
355
+ onopentag(name, attributes) {
356
+ depth++;
357
+ if (open === null && isWanted(name)) {
358
+ open = { name, attributes, depth, start: parser.startIndex, tagEnd: parser.endIndex };
359
+ }
360
+ if (depth > MAX_ELEMENT_DEPTH) throw TOO_DEEP;
361
+ },
362
+ onclosetag(name) {
363
+ if (open !== null && depth === open.depth && name === open.name) {
364
+ // A self-closing `<w:p/>` reports its close on the SAME token as its
365
+ // open tag; anything else — including the close the parser implies
366
+ // for an element left open at end of input — ends further on.
367
+ const selfClosing = parser.endIndex === open.tagEnd;
368
+ const stop = emit({
369
+ name,
370
+ attributes: open.attributes,
371
+ body: selfClosing ? undefined : xml.slice(open.tagEnd + 1, parser.startIndex),
372
+ start: open.start,
373
+ end: parser.endIndex + 1,
374
+ });
375
+ open = null;
376
+ if (stop) throw STOP_SCAN;
377
+ }
378
+ depth--;
379
+ },
380
+ },
381
+ { xmlMode: true, decodeEntities: false },
382
+ );
383
+ try {
384
+ parser.write(xml);
385
+ parser.end();
386
+ } catch (err) {
387
+ if (err !== TOO_DEEP && err !== STOP_SCAN) throw err;
388
+ // On the depth bound, the element still open keeps the body it had
389
+ // reached — the text before the crafted tail is real and there is no
390
+ // reason to discard it. (Read through an alias: the assignments happen
391
+ // inside the parser's callbacks, which control-flow analysis cannot see
392
+ // from here.) A STOP is the visitor's own choice mid-document, so nothing
393
+ // is pending by construction (the throw follows a completed block).
394
+ const pending = open as OpenMatch | null;
395
+ if (err === TOO_DEEP && pending !== null) {
396
+ emit({
397
+ name: pending.name,
398
+ attributes: pending.attributes,
399
+ body: xml.slice(pending.tagEnd + 1, parser.startIndex),
400
+ start: pending.start,
401
+ end: parser.startIndex,
402
+ });
403
+ }
404
+ parser.reset(); // drop the stack this part built before giving up on it
405
+ }
406
+ return out;
407
+ }
408
+
409
+ /**
410
+ * The text of one OOXML paragraph: every `<w:t>`/`<a:t>` run's character
411
+ * content, concatenated with NO separator — Word/PowerPoint split runs
412
+ * mid-word on formatting boundaries, so any separator would break words apart.
413
+ * `tag` is the run tag ('w:t' for docx, 'a:t' for pptx). A self-closing
414
+ * `<w:t/>` is an empty run and contributes nothing, exactly as it did when
415
+ * the pattern simply failed to match it.
416
+ */
417
+ export function paragraphRunText(paragraphXml: string, localTag: string): string {
418
+ let out = '';
419
+ let inRun = 0;
420
+ let inCdata = false;
421
+ let depth = 0;
422
+ const parser = new Parser(
423
+ {
424
+ onopentag(name) {
425
+ depth++;
426
+ if (localName(name) === localTag) inRun++;
427
+ if (depth > MAX_ELEMENT_DEPTH) throw TOO_DEEP;
428
+ },
429
+ // TEXT, never the raw body: a run's body is markup as well as characters
430
+ // when the part is malformed enough to nest runs, and slicing it wholesale
431
+ // put `<w:t xml:space="preserve">` into the document's extracted text.
432
+ //
433
+ // Entities decode through the module's STRICT decoder, not the parser's:
434
+ // the parser turns a numeric reference outside XML's `Char` production
435
+ // (`&#0;`, a lone surrogate) into replacement/control characters, where
436
+ // `decodeXmlEntities` keeps such a reference literal. CDATA text is
437
+ // already literal and must not decode at all.
438
+ ontext(text) {
439
+ if (inRun > 0) out += inCdata ? text : decodeXmlEntities(text);
440
+ },
441
+ oncdatastart() {
442
+ inCdata = true;
443
+ },
444
+ oncdataend() {
445
+ inCdata = false;
446
+ },
447
+ onclosetag(name) {
448
+ if (localName(name) === localTag && inRun > 0) inRun--;
449
+ depth--;
450
+ },
451
+ },
452
+ { xmlMode: true, decodeEntities: false },
453
+ );
454
+ try {
455
+ parser.write(paragraphXml);
456
+ parser.end();
457
+ } catch (err) {
458
+ if (err !== TOO_DEEP) throw err;
459
+ parser.reset();
460
+ }
461
+ return out;
462
+ }
463
+
464
+ /**
465
+ * Split an XML fragment into its `<{tag}>…</{tag}>` blocks, in document order.
466
+ * A self-closing `<{tag}/>` yields '' — an EMPTY block, which is what an empty
467
+ * `<w:p/>` paragraph or `<w:tc/>` cell means. See {@link xmlElementBlocks} for
468
+ * the non-nesting and quoting guarantees.
469
+ */
470
+ /** {@link xmlBlocks} by LOCAL name — see {@link localElementBlocks} for why. */
471
+ export function localBlocks(xml: string, local: string): string[] {
472
+ return localElementBlocks(xml, [local]).map((e) => e.body ?? '');
473
+ }
474
+
475
+ export function xmlBlocks(xml: string, tag: string): string[] {
476
+ return xmlElementBlocks(xml, [tag]).map((e) => e.body ?? '');
477
+ }
@@ -0,0 +1,131 @@
1
+ import { isUtf8 } from 'node:buffer';
2
+ import { fileExtension } from './doc-extract.types.js';
3
+ import { displayPath, type FileReader, type ReadResult } from './file-reader.js';
4
+
5
+ /** Minimal extension→mime map for the binary-read notice (fallback: octet-stream). */
6
+ const MIME_BY_EXT: Record<string, string> = {
7
+ '.png': 'image/png',
8
+ '.jpg': 'image/jpeg',
9
+ '.jpeg': 'image/jpeg',
10
+ '.gif': 'image/gif',
11
+ '.webp': 'image/webp',
12
+ '.bmp': 'image/bmp',
13
+ '.ico': 'image/x-icon',
14
+ '.zip': 'application/zip',
15
+ '.gz': 'application/gzip',
16
+ '.tar': 'application/x-tar',
17
+ '.7z': 'application/x-7z-compressed',
18
+ '.rar': 'application/vnd.rar',
19
+ '.doc': 'application/msword',
20
+ '.ppt': 'application/vnd.ms-powerpoint',
21
+ '.xls': 'application/vnd.ms-excel',
22
+ '.mp3': 'audio/mpeg',
23
+ '.wav': 'audio/wav',
24
+ '.mp4': 'video/mp4',
25
+ '.mov': 'video/quicktime',
26
+ '.woff': 'font/woff',
27
+ '.woff2': 'font/woff2',
28
+ '.ttf': 'font/ttf',
29
+ '.pdf': 'application/pdf',
30
+ '.wasm': 'application/wasm',
31
+ '.exe': 'application/vnd.microsoft.portable-executable',
32
+ };
33
+
34
+ /**
35
+ * Text is what the fallback reader may hand to the text tools: no NUL byte
36
+ * AND valid UTF-8. A NUL-free file that does not decode as UTF-8 is still
37
+ * binary here — the decode would be lossy, so text written back could never
38
+ * round-trip the original bytes. Checked on the raw bytes, so a large binary
39
+ * is refused without allocating its full decoded string first.
40
+ */
41
+ function isTextBytes(bytes: Buffer): boolean {
42
+ return !bytes.includes(0) && isUtf8(bytes);
43
+ }
44
+
45
+ /**
46
+ * The DEFAULT reader — the registry's fallback for every extension no other
47
+ * reader owns. Decodes utf8 text; content carrying a NUL byte or invalid
48
+ * UTF-8 is not (round-trippable) text at all, so the read answers with an
49
+ * honest one-line notice INSTEAD of raw bytes: what the file is (mime by
50
+ * extension + size), plus the actionable hint where one exists (unzip for
51
+ * archives). Same binary test on the grep path: binary content is simply not
52
+ * searchable.
53
+ */
54
+ export class TextReader implements FileReader {
55
+ /** Fallback reader: matched by the registry's default, not by extension. */
56
+ readonly extensions: readonly string[] = [];
57
+ readonly textEditable: boolean = true;
58
+
59
+ async read(bytes: Buffer, path: string): Promise<ReadResult> {
60
+ return isTextBytes(bytes)
61
+ ? { kind: 'text', text: bytes.toString('utf8') }
62
+ : { kind: 'refusal', message: this.binaryNotice(path, bytes.length) };
63
+ }
64
+
65
+ /**
66
+ * Binary content is not editable, whatever its extension. `read` answers a
67
+ * binary file with a notice INSTEAD of its bytes, so an agent asking to
68
+ * write text over one would be overwriting something it could not read. The
69
+ * write tools refuse instead. (Uploads and the HTTP write routes are
70
+ * untouched: a human replacing a file is exactly the right move.)
71
+ */
72
+ editRefusalForExisting(bytes: Buffer, path: string): string | null {
73
+ return isTextBytes(bytes)
74
+ ? null
75
+ : `"${displayPath(path)}" holds binary content, which read_file cannot show as text. Writing text over it would ` +
76
+ `destroy those bytes — replace the file by uploading a new version instead.`;
77
+ }
78
+
79
+ async greppableText(bytes: Buffer): Promise<string | null> {
80
+ return isTextBytes(bytes) ? bytes.toString('utf8') : null; // skip binary (NUL / invalid UTF-8)
81
+ }
82
+
83
+ /** The honest one-line notice returned INSTEAD of raw bytes for unreadable binary content. */
84
+ protected binaryNotice(path: string, sizeBytes: number): string {
85
+ const ext = fileExtension(path);
86
+ const mime = MIME_BY_EXT[ext] ?? 'application/octet-stream';
87
+ const zipHint = ext === '.zip' ? ' Use the unzip tool to extract its contents.' : '';
88
+ return `[${displayPath(path)} is a binary file (${mime}, ${sizeBytes} bytes) — not readable as text.${zipHint}]`;
89
+ }
90
+ }
91
+
92
+ /** The modern OOXML counterpart of each legacy binary office extension. */
93
+ const MODERN_BY_LEGACY: Record<string, string> = { '.doc': '.docx', '.ppt': '.pptx', '.xls': '.xlsx' };
94
+
95
+ /**
96
+ * Legacy binary office formats (.doc/.ppt/.xls) — NOT extractable (text
97
+ * extraction supports only the modern formats), so a binary read gets the
98
+ * convert-to-modern hint instead of the generic notice. Reads and greps are
99
+ * otherwise the text reader's behaviour: a legacy-named file that happens to
100
+ * hold plain text still reads (and greps) as text.
101
+ *
102
+ * NOT text-editable: read_file cannot extract a binary legacy document, so an
103
+ * edit_file/write_file on one would overwrite the real document with text
104
+ * that never round-tripped — the write tools refuse with the convert-or-
105
+ * replace message below.
106
+ */
107
+ export class LegacyOfficeReader extends TextReader {
108
+ override readonly extensions: readonly string[] = ['.doc', '.ppt', '.xls'];
109
+ override readonly textEditable: boolean = false;
110
+
111
+ /** The write-refusal for the agent text-editing tools (see `assertNotDocumentEdit`). */
112
+ editRefusal(path: string): string {
113
+ const ext = fileExtension(path);
114
+ const modern = MODERN_BY_LEGACY[ext] ?? 'the modern format';
115
+ return (
116
+ `"${displayPath(path)}" is a legacy binary office format (${ext}). read_file cannot extract its text, and text ` +
117
+ 'written by the editing tools would destroy the binary document. Convert the document to ' +
118
+ `${modern} and upload that, or replace the file by uploading a new version.`
119
+ );
120
+ }
121
+
122
+ protected override binaryNotice(path: string, sizeBytes: number): string {
123
+ const ext = fileExtension(path);
124
+ const modern = MODERN_BY_LEGACY[ext];
125
+ return (
126
+ `[${displayPath(path)} is a legacy office format (${ext}, ${sizeBytes} bytes) — text extraction supports only the ` +
127
+ `modern format. Convert the document to ${modern} and upload that to read its text, or replace it by ` +
128
+ 'uploading a new version.]'
129
+ );
130
+ }
131
+ }