@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
@@ -0,0 +1,356 @@
1
+ import { Buffer } from 'node:buffer'
2
+
3
+ /**
4
+ * Ports of the SheetJS 0.20.3 XML helpers (`xlsx.mjs`) the extractor needs to read a part the way
5
+ * SheetJS would: its tag pattern, `parsexmltag`, `strip_ns`, `utf8read`, and `unescapexml`.
6
+ * Everything here is linear in the text it scans.
7
+ */
8
+
9
+ const swapUtf16ByteOrder = (bytes: Buffer) =>
10
+ Buffer.from(bytes.subarray(0, bytes.length - (bytes.length % 2))).swap16()
11
+
12
+ /**
13
+ * The texts SheetJS can read from a part: its Latin-1 ("binary") view and, for BOM-marked parts,
14
+ * the UTF-16 decodings of `cc2str` (little- and big-endian from byte 2, including its
15
+ * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
16
+ */
17
+ export const sheetJsTextViews = (content: Uint8Array): ReadonlyArray<string> => {
18
+ const bytes = Buffer.from(content.buffer, content.byteOffset, content.byteLength)
19
+ const latin1 = bytes.toString('latin1')
20
+ const littleEndian = bytes[0] === 0xff && bytes[1] === 0xfe
21
+ const bigEndian = bytes[0] === 0xfe && bytes[1] === 0xff
22
+ const offsetBigEndian = bytes[1] === 0xfe && bytes[2] === 0xff
23
+
24
+ if (!littleEndian && !bigEndian && !offsetBigEndian) return [latin1]
25
+
26
+ return [
27
+ latin1,
28
+ bytes.subarray(2).toString('utf16le'),
29
+ swapUtf16ByteOrder(bytes.subarray(2)).toString('utf16le'),
30
+ swapUtf16ByteOrder(bytes.subarray(3)).toString('utf16le')
31
+ ]
32
+ }
33
+
34
+ /**
35
+ * SheetJS's own tag pattern (`tagregex1`, used for every part it parses): quoted values may hold
36
+ * `<` and `>`. Each attempt stops at the next quote of its kind, so a scan stays linear.
37
+ */
38
+ const sheetJsTagPattern =
39
+ /<[/?]?[a-zA-Z0-9:_-]+(?:\s+[^"\s?<>/]+\s*=\s*(?:"[^"]*"|'[^']*'|[^'"<>\s=]+))*\s*[/?]?>/gm
40
+
41
+ /** SheetJS `attregexg`. */
42
+ const sheetJsAttribute = /\s([^"\s?>/]+)\s*=\s*((?:")([^"]*)(?:")|(?:')([^']*)(?:')|([^'">\s]+))/g
43
+
44
+ export type SheetJsTag = {
45
+ /** The tag up to its first space, line feed, or carriage return (SheetJS `y[0]`). */
46
+ readonly head: string
47
+ /** Raw (still escaped) attribute values by SheetJS key, plus lower-cased copies. */
48
+ readonly attributes: ReadonlyMap<string, string>
49
+ }
50
+
51
+ /**
52
+ * Port of SheetJS `parsexmltag`: exact-case keys (plus lower-cased copies), a namespace prefix
53
+ * dropped, an unprefixed name cut at its first `_`, the last value winning. Values are raw.
54
+ */
55
+ export const parseSheetJsTag = (tag: string): SheetJsTag => {
56
+ let end = 0
57
+
58
+ for (; end < tag.length; end += 1) {
59
+ const code = tag.charCodeAt(end)
60
+
61
+ if (code === 32 || code === 10 || code === 13) break
62
+ }
63
+
64
+ const attributes = new Map<string, string>()
65
+
66
+ if (end === tag.length) return { head: tag, attributes }
67
+
68
+ for (const [match] of tag.matchAll(sheetJsAttribute)) {
69
+ const text = match.slice(1)
70
+ let equals = text.indexOf('=')
71
+ let name = text.slice(0, equals).trim()
72
+
73
+ while (text.charCodeAt(equals + 1) === 32) equals += 1
74
+
75
+ const quoteCode = text.charCodeAt(equals + 1)
76
+ const quoted = quoteCode === 34 || quoteCode === 39 ? 1 : 0
77
+ const value = text.slice(equals + 1 + quoted, text.length - quoted)
78
+ const colon = name.indexOf(':')
79
+
80
+ if (colon < 0) {
81
+ if (name.indexOf('_') > 0) name = name.slice(0, name.indexOf('_'))
82
+ } else {
83
+ const local = (colon === 5 && name.startsWith('xmlns') ? 'xmlns' : '') + name.slice(colon + 1)
84
+
85
+ if (attributes.has(local) && name.slice(colon - 3, colon) === 'ext') continue
86
+
87
+ name = local
88
+ }
89
+
90
+ attributes.set(name, value)
91
+ attributes.set(name.toLowerCase(), value)
92
+ }
93
+
94
+ return { head: tag.slice(0, end), attributes }
95
+ }
96
+
97
+ /** Every tag SheetJS's pattern finds in `text`, in document order. */
98
+ export function* sheetJsTags(text: string): Generator<SheetJsTag> {
99
+ for (const [tag] of text.matchAll(sheetJsTagPattern)) yield parseSheetJsTag(tag)
100
+ }
101
+
102
+ /** SheetJS `strip_ns`: the first `<prefix:` (or `</prefix:`) loses its prefix. */
103
+ export const stripSheetJsNamespace = (head: string) => head.replace(/<(\/?)\w+:/, '<$1')
104
+
105
+ /** SheetJS `utf8read` in Node: the Latin-1 ("binary") string read back as UTF-8. */
106
+ export const sheetJsUtf8Read = (binary: string) => Buffer.from(binary, 'latin1').toString('utf8')
107
+
108
+ const encodings: ReadonlyMap<string, string> = new Map([
109
+ ['&quot;', '"'],
110
+ ['&apos;', "'"],
111
+ ['&gt;', '>'],
112
+ ['&lt;', '<'],
113
+ ['&amp;', '&']
114
+ ])
115
+
116
+ /**
117
+ * SheetJS `unescapexml` for text without CDATA, quirks included: entity names match ignoring
118
+ * case but only lower-case ones map (`&QUOT;` becomes U+0000), `&#X41;` is read as decimal, and
119
+ * numeric references wrap at U+FFFF. Returns `undefined` for text with a CDATA marker, which
120
+ * SheetJS splits recursively (and never ends for an unterminated one).
121
+ */
122
+ export const sheetJsUnescapeXml = (text: string): string | undefined => {
123
+ if (text.includes('<![CDATA[')) return undefined
124
+
125
+ return text
126
+ .replace(
127
+ /&(?:quot|apos|gt|lt|amp|#x?([\da-fA-F]+));/gi,
128
+ (entity, code: string | undefined) =>
129
+ encodings.get(entity) ??
130
+ String.fromCharCode(Number.parseInt(code ?? '', entity.includes('x') ? 16 : 10))
131
+ )
132
+ .replace(/_x([\da-fA-F]{4})_/gi, (_, code: string) =>
133
+ String.fromCharCode(Number.parseInt(code, 16))
134
+ )
135
+ }
136
+
137
+ /** An attribute value as SheetJS reads text attributes: `unescapexml(utf8read(raw))`. */
138
+ export const sheetJsAttributeText = (raw: string) => sheetJsUnescapeXml(sheetJsUtf8Read(raw))
139
+
140
+ const isHighSurrogate = (code: number) => code >= 0xd800 && code <= 0xdbff
141
+
142
+ const isLowSurrogate = (code: number) => code >= 0xdc00 && code <= 0xdfff
143
+
144
+ const codeEscape = (code: number) => `_x${code.toString(16).toUpperCase().padStart(4, '0')}_`
145
+
146
+ const escapedCharacters: ReadonlyMap<string, string> = new Map([
147
+ ['&', '&amp;'],
148
+ ['<', '&lt;'],
149
+ ['>', '&gt;'],
150
+ ['"', '&quot;']
151
+ ])
152
+
153
+ // SheetJS matches `_xHHHH_` codes ignoring case (`coderegex`), so `_X0041_` is a code too.
154
+ const startsCodeEscape = /_x[\da-fA-F]{4}_/iy
155
+
156
+ /**
157
+ * Escape `text` for a double-quoted attribute of a generated UTF-8 part so that SheetJS's
158
+ * `unescapexml(utf8read(…))` returns `text` exactly: markup characters become entities, an `_`
159
+ * that would start an `_xHHHH_` code becomes `_x005F_`, and control characters, U+FFFE, U+FFFF,
160
+ * and lone surrogates become `_xHHHH_` codes.
161
+ */
162
+ export const sheetJsAttributeEscape = (text: string) => {
163
+ let output = ''
164
+
165
+ for (let index = 0; index < text.length; index += 1) {
166
+ const character = text.charAt(index)
167
+ const code = text.charCodeAt(index)
168
+ const escaped = escapedCharacters.get(character)
169
+
170
+ if (escaped !== undefined) {
171
+ output += escaped
172
+ continue
173
+ }
174
+
175
+ if (character === '_') {
176
+ startsCodeEscape.lastIndex = index
177
+ output += startsCodeEscape.test(text) ? codeEscape(code) : character
178
+ continue
179
+ }
180
+
181
+ if (isHighSurrogate(code) && isLowSurrogate(text.charCodeAt(index + 1))) {
182
+ output += text.slice(index, index + 2)
183
+ index += 1
184
+ continue
185
+ }
186
+
187
+ output +=
188
+ code < 0x20 ||
189
+ code === 0xfffe ||
190
+ code === 0xffff ||
191
+ isHighSurrogate(code) ||
192
+ isLowSurrogate(code)
193
+ ? codeEscape(code)
194
+ : character
195
+ }
196
+
197
+ return output
198
+ }
199
+
200
+ /** SheetJS's own `XML_HEADER`, used for every generated part. */
201
+ export const sheetJsXmlHeader = '<?xml version="1.0" encoding="UTF-8" standalone="yes"?>\r\n'
202
+
203
+ export const spreadsheetMainNamespace = 'http://schemas.openxmlformats.org/spreadsheetml/2006/main'
204
+
205
+ export const officeDocumentRelationshipsNamespace =
206
+ 'http://schemas.openxmlformats.org/officeDocument/2006/relationships'
207
+
208
+ const cdataMarker = '<![CDATA['
209
+
210
+ type TextStep = (text: string) => string | undefined
211
+
212
+ /** The two conversions SheetJS chains over cell and shared-string text. */
213
+ const sheetJsTextSteps: ReadonlyArray<TextStep> = [sheetJsUnescapeXml, sheetJsUtf8Read]
214
+
215
+ /** Depth first, so at most `steps + 1` derived strings are alive at once. */
216
+ const meetsCdata = (text: string, steps: number): boolean => {
217
+ if (text.includes(cdataMarker)) return true
218
+
219
+ if (steps === 0) return false
220
+
221
+ return sheetJsTextSteps.some(step => {
222
+ const next = step(text)
223
+
224
+ return next !== undefined && next !== text && meetsCdata(next, steps - 1)
225
+ })
226
+ }
227
+
228
+ const isNameCode = (code: number) =>
229
+ (code >= 48 && code <= 57) || // 0-9
230
+ (code >= 65 && code <= 90) || // A-Z
231
+ (code >= 97 && code <= 122) || // a-z
232
+ code === 95 || // _
233
+ code === 46 || // .
234
+ code === 45 // -
235
+
236
+ /** The end of the run of name characters (`[\w.-]`) starting at `start`. */
237
+ const nameEnd = (text: string, start: number) => {
238
+ let end = start
239
+
240
+ while (end < text.length && isNameCode(text.charCodeAt(end))) end += 1
241
+
242
+ return end
243
+ }
244
+
245
+ /**
246
+ * `text` without its simple opening tags, `<(?:[\w.-]+:)?[\w.-]+>`: a superset of the tags SheetJS
247
+ * removes before decoding (`<(?:\w+:)?(?:si|sstItem)>` in `parse_sst_xml`, `<(?:\w+:)?r>` in
248
+ * `parse_rs`). One forward scan: a `<` that does not start such a tag is kept and the scan resumes
249
+ * at the next `<`, so every character is read at most twice.
250
+ */
251
+ export const withoutSimpleTags = (text: string) => {
252
+ let output = ''
253
+ let kept = 0
254
+ let index = text.indexOf('<')
255
+
256
+ while (index >= 0) {
257
+ let end = nameEnd(text, index + 1)
258
+
259
+ if (end > index + 1 && text.charCodeAt(end) === 58) {
260
+ const local = nameEnd(text, end + 1)
261
+
262
+ end = local > end + 1 ? local : -1
263
+ }
264
+
265
+ if (end > index + 1 && text.charCodeAt(end) === 62) {
266
+ output += text.slice(kept, index)
267
+ kept = end + 1
268
+ index = text.indexOf('<', kept)
269
+ } else {
270
+ index = text.indexOf('<', index + 1)
271
+ }
272
+ }
273
+
274
+ return output + text.slice(kept)
275
+ }
276
+
277
+ /** `<<` or `<!`: never written in worksheets or shared strings by Excel, LibreOffice, or Sheets. */
278
+ const hasMarkupOpener = (text: string) => text.includes('<<') || text.includes('<!')
279
+
280
+ /**
281
+ * Whether SheetJS could meet a CDATA marker in this part. SheetJS's `unescapexml` handles CDATA by
282
+ * recursing on a string two characters shorter and copying the whole tail at every level, so an
283
+ * unterminated marker costs quadratic time and memory.
284
+ *
285
+ * What SheetJS hands to `unescapexml` in a worksheet or shared-strings part comes from the part
286
+ * text through two kinds of transformation:
287
+ *
288
+ * - Decodes: raw (`<v>` of every cell), `utf8read(raw)` (shared and inline strings), and
289
+ * `utf8read(unescapexml(raw))` (cells of type `str`, decoded again after `utf8read`).
290
+ * `utf8read` keeps only the low byte of each character, so U+013C from `_x013C_`, `&#x13C;`, or
291
+ * `&#316;` becomes `<`.
292
+ * - Tag removal before decoding: `parse_sst_xml` removes every `<si>`/`<sstItem>` opening tag
293
+ * from the whole shared-strings table, and `parse_rs` removes every `<r>` opening tag from rich
294
+ * text (after `utf8read`). Inline strings (`t="inlineStr"`) call `parse_si` without options,
295
+ * so their rich text is processed even with `cellHTML: false`. `A<<r>![CDATA[B` thus reaches
296
+ * `unescapexml` as `A<![CDATA[B`.
297
+ *
298
+ * So the check rejects a text view (`sheetJsTextViews`) when:
299
+ *
300
+ * - the view, or `utf8read` of it, contains `<<` or `<!`. A marker assembled by removing tags
301
+ * needs a literal `<` (in the view, or from `utf8read`) followed by a removed tag, or by `!`,
302
+ * and every removed tag starts with `<`;
303
+ * - the view, or the view without any simple opening tag (`withoutSimpleTags`, a superset of
304
+ * SheetJS's removals), meets the marker raw or after any chain of up to two steps of
305
+ * `unescapexml` and `utf8read`, in any order (a superset of SheetJS's decode sequences).
306
+ *
307
+ * Every step is a linear pass, at most a few per view. The first rule deliberately fails closed:
308
+ * it also rejects XML comments, `<!DOCTYPE`, and any other `<!…` declaration, which Excel,
309
+ * LibreOffice, and Google Sheets never write in worksheets or shared strings (nor a literal `<<`).
310
+ */
311
+ export const sheetJsCouldReadCdata = (content: Uint8Array) =>
312
+ sheetJsTextViews(content).some(
313
+ view =>
314
+ hasMarkupOpener(view) ||
315
+ hasMarkupOpener(sheetJsUtf8Read(view)) ||
316
+ meetsCdata(view, 2) ||
317
+ meetsCdata(withoutSimpleTags(view), 2)
318
+ )
319
+
320
+ const xmlBoundary = new Set([' ', '\t', '\r', '\n', '>'])
321
+
322
+ /** SheetJS `str_match_xml`: the first `<tag` element's inner text (exact prefix and case). */
323
+ const sheetJsElementText = (text: string, tag: string) => {
324
+ const width = tag.length + 1
325
+ let start = text.indexOf(`<${tag}`)
326
+
327
+ while (start >= 0 && start <= text.length - width && !xmlBoundary.has(text.charAt(start + width)))
328
+ start = text.indexOf(`<${tag}`, start + 1)
329
+
330
+ if (start === -1) return undefined
331
+
332
+ const contentStart = text.indexOf('>', start + tag.length)
333
+
334
+ if (contentStart === -1) return undefined
335
+
336
+ const end = text.indexOf(`</${tag}>`, contentStart)
337
+
338
+ return end === -1 ? undefined : text.slice(contentStart + 1, end)
339
+ }
340
+
341
+ /**
342
+ * The `dc:title` of a core-properties part, read as SheetJS `parse_core_props` reads it, without
343
+ * handing the part to SheetJS. A title holding CDATA is ignored.
344
+ */
345
+ export const coreTitle = (content: Uint8Array | undefined) => {
346
+ if (content === undefined) return undefined
347
+
348
+ const raw = sheetJsElementText(
349
+ Buffer.from(content.buffer, content.byteOffset, content.byteLength).toString('utf8'),
350
+ 'dc:title'
351
+ )
352
+
353
+ const title = raw === undefined ? undefined : sheetJsUnescapeXml(raw)
354
+
355
+ return title !== undefined && title.trim().length > 0 ? title : undefined
356
+ }
@@ -0,0 +1,177 @@
1
+ import { Effect, Predicate } from 'effect'
2
+ import { minimumSheetJsVersion, SheetJsUnavailableError } from '../errors.ts'
3
+ import type { XlsxWorkbook } from './xlsx-text.ts'
4
+
5
+ /** Loads the SheetJS module. The default is a lazy `import('xlsx')`. */
6
+ export type SheetJsLoader = () => Promise<unknown>
7
+
8
+ export const defaultSheetJsLoader: SheetJsLoader = () => import('xlsx')
9
+
10
+ export type SheetJs = {
11
+ readonly version: string
12
+ /** Parse workbook bytes with `sheetJsReadOptions`; the result is checked before use. */
13
+ readonly read: (bytes: Uint8Array) => unknown
14
+ }
15
+
16
+ /**
17
+ * SheetJS `read` options: cell values and display text (`cell.w`) only.
18
+ *
19
+ * - `cellFormula: false`: no formula text. SheetJS otherwise copies a shifted master formula onto
20
+ * every shared-formula dependent and scans every earlier array formula for each cell, before
21
+ * any extractor budget runs. Cached values still render; formula-only cells render empty.
22
+ * - `cellHTML: false`: no rich-text HTML (`cell.h`) we never read. Inline strings ignore it
23
+ * (SheetJS calls `parse_si` without options), so their rich text is still rendered; the CDATA
24
+ * check (`sheetJsCouldReadCdata`) does not rely on this option.
25
+ * - `cellText: true`: keep the formatted display text (`cell.w`) the CSV uses. SheetJS formats
26
+ * every styled cell inside `read`, with work proportional to the format code, so it only ever
27
+ * reads the generated `xl/styles.xml` (codes of at most 255 characters, `xlsx-styles.ts`).
28
+ * - `cellNF`, `cellStyles`, `cellDates: false`: no format strings or style objects; dates stay
29
+ * serial numbers whose `cell.w` carries the formatted date (`cellStyles` would also force
30
+ * `sheetStubs`).
31
+ * - `sheetStubs: false`: no objects for empty cells.
32
+ * - `bookDeps`, `bookFiles`, `bookProps`, `bookSheets`, `bookVBA: false`: no calculation chain,
33
+ * raw archive, or VBA blob, and a full parse (not the properties-only or names-only modes).
34
+ * - `dense: false`: sheets keyed by address, as `xlsx-text.ts` reads them.
35
+ * - `WTF: false`: per-sheet parse errors skip the sheet instead of throwing.
36
+ *
37
+ * A fresh copy is passed on every call because SheetJS writes defaults into the options object.
38
+ */
39
+ export const sheetJsReadOptions = {
40
+ type: 'array',
41
+ cellFormula: false,
42
+ cellHTML: false,
43
+ cellText: true,
44
+ cellNF: false,
45
+ cellStyles: false,
46
+ cellDates: false,
47
+ sheetStubs: false,
48
+ bookDeps: false,
49
+ bookFiles: false,
50
+ bookProps: false,
51
+ bookSheets: false,
52
+ bookVBA: false,
53
+ dense: false,
54
+ WTF: false
55
+ } as const
56
+
57
+ /** SemVer 2.0.0: `major.minor.patch`, optional `-prerelease`, optional `+build`. */
58
+ const semver =
59
+ /^(0|[1-9]\d*)\.(0|[1-9]\d*)\.(0|[1-9]\d*)(?:-((?:0|[1-9]\d*|\d*[a-zA-Z-][0-9a-zA-Z-]*)(?:\.(?:0|[1-9]\d*|\d*[a-zA-Z-][0-9a-zA-Z-]*))*))?(?:\+[0-9a-zA-Z-]+(?:\.[0-9a-zA-Z-]+)*)?$/
60
+
61
+ type ParsedVersion = {
62
+ readonly release: readonly [number, number, number]
63
+ readonly prerelease: boolean
64
+ }
65
+
66
+ const parseVersion = (version: string): ParsedVersion | undefined => {
67
+ const match = semver.exec(version)
68
+
69
+ if (match === null) return undefined
70
+
71
+ const release = [Number(match[1]), Number(match[2]), Number(match[3])] as const
72
+
73
+ return release.every(Number.isSafeInteger)
74
+ ? { release, prerelease: match[4] !== undefined }
75
+ : undefined
76
+ }
77
+
78
+ /**
79
+ * Compare by SemVer precedence against the minimum release: a prerelease of 0.20.3 is below it,
80
+ * prereleases of later releases are above it, and build metadata is ignored.
81
+ */
82
+ const meetsMinimumVersion = (installed: ParsedVersion, minimum: ParsedVersion) => {
83
+ for (const [index, part] of installed.release.entries()) {
84
+ const required = minimum.release[index] ?? 0
85
+
86
+ if (part !== required) return part > required
87
+ }
88
+
89
+ return !installed.prerelease
90
+ }
91
+
92
+ const isMissingModule = (cause: unknown) =>
93
+ Predicate.hasProperty(cause, 'code') &&
94
+ (cause.code === 'ERR_MODULE_NOT_FOUND' || cause.code === 'MODULE_NOT_FOUND')
95
+
96
+ /** ESM namespace, or a CommonJS interop namespace whose `default` is the module. */
97
+ const sheetJsExports = (namespace: unknown): object | undefined => {
98
+ if (Predicate.hasProperty(namespace, 'read')) return namespace
99
+
100
+ if (
101
+ Predicate.hasProperty(namespace, 'default') &&
102
+ Predicate.hasProperty(namespace.default, 'read')
103
+ )
104
+ return namespace.default
105
+
106
+ return undefined
107
+ }
108
+
109
+ /**
110
+ * Load SheetJS lazily and refuse anything that is not SheetJS 0.20.3 or newer. The version must be
111
+ * strict SemVer; anything else fails closed as `invalid`.
112
+ */
113
+ export const loadSheetJs = (loader: SheetJsLoader) =>
114
+ Effect.gen(function* () {
115
+ const namespace = yield* Effect.tryPromise({
116
+ try: loader,
117
+ catch: cause =>
118
+ new SheetJsUnavailableError({
119
+ reason: isMissingModule(cause) ? 'missing' : 'invalid',
120
+ cause
121
+ })
122
+ })
123
+
124
+ const exports = sheetJsExports(namespace)
125
+
126
+ if (
127
+ exports === undefined ||
128
+ !Predicate.hasProperty(exports, 'read') ||
129
+ !Predicate.isFunction(exports.read) ||
130
+ !Predicate.hasProperty(exports, 'version') ||
131
+ !Predicate.isString(exports.version)
132
+ )
133
+ return yield* Effect.fail(new SheetJsUnavailableError({ reason: 'invalid' }))
134
+
135
+ const { read, version } = exports
136
+ const installed = parseVersion(version)
137
+ const minimum = parseVersion(minimumSheetJsVersion)
138
+
139
+ // A version that is not strict SemVer is not a SheetJS release we can vouch for.
140
+ if (installed === undefined || minimum === undefined)
141
+ return yield* Effect.fail(
142
+ new SheetJsUnavailableError({ reason: 'invalid', installedVersion: version })
143
+ )
144
+
145
+ if (!meetsMinimumVersion(installed, minimum))
146
+ return yield* Effect.fail(
147
+ new SheetJsUnavailableError({ reason: 'outdated', installedVersion: version })
148
+ )
149
+
150
+ const sheetJs: SheetJs = {
151
+ version,
152
+ read: bytes => read(bytes, { ...sheetJsReadOptions })
153
+ }
154
+
155
+ return sheetJs
156
+ })
157
+
158
+ /** Check the parsed workbook shape before reading it. */
159
+ export const asXlsxWorkbook = (parsed: unknown): XlsxWorkbook | undefined => {
160
+ if (
161
+ !Predicate.hasProperty(parsed, 'SheetNames') ||
162
+ !Predicate.hasProperty(parsed, 'Sheets') ||
163
+ !Array.isArray(parsed.SheetNames) ||
164
+ !Predicate.isObject(parsed.Sheets)
165
+ )
166
+ return undefined
167
+
168
+ const sheetNames: Array<string> = []
169
+
170
+ for (const name of parsed.SheetNames) {
171
+ if (!Predicate.isString(name)) return undefined
172
+
173
+ sheetNames.push(name)
174
+ }
175
+
176
+ return { SheetNames: sheetNames, Sheets: parsed.Sheets }
177
+ }
@@ -0,0 +1,162 @@
1
+ import { Effect } from 'effect'
2
+ import * as Schema from 'effect/Schema'
3
+
4
+ /**
5
+ * Worker admission. Every Node `FileExtractor` layer in a JavaScript realm (the main thread, or
6
+ * each worker thread or `vm` context that builds the layer) shares one pool of
7
+ * `processWorkerLimit` worker slots, however often the layer is built (a per-request
8
+ * `Effect.provide` builds a new one each time). The pool lives on `globalThis` under a versioned
9
+ * `Symbol.for` key and holds only plain data and plain callbacks, so duplicated copies of this
10
+ * package (and of Effect) in one realm share it. Each layer also has its own pool of
11
+ * `maxConcurrentWorkers` slots (at most `processWorkerLimit`), which can only lower that layer's
12
+ * share. An extraction takes its layer's slot, then a realm slot, and waits for both until its
13
+ * admission deadline; a slot is released only after its worker has terminated.
14
+ *
15
+ * Admission is first come, first served: a freed slot is handed to the longest-waiting extraction
16
+ * whose deadline has not passed, and a new extraction takes a free slot only when nobody waits.
17
+ */
18
+
19
+ /** Worker slots shared by every layer in the realm. */
20
+ export const processWorkerLimit = 4
21
+
22
+ /**
23
+ * Takes a freed slot for its waiter and returns `true`, or returns `false` when the waiter's
24
+ * deadline has passed (it then fails with `busy`). Either way it leaves the queue. It never runs
25
+ * the waiter's fiber: the resume is deferred to a microtask, so a waiter that finishes at once
26
+ * cannot release (and hand off) again inside this call, and a long queue drains in constant stack.
27
+ */
28
+ export type SlotHandOff = () => boolean
29
+
30
+ /**
31
+ * A counting pool of worker slots with a FIFO queue (a `Set` iterates in insertion order). While
32
+ * anyone waits, every slot is taken: a released slot passes to a waiter without being counted
33
+ * free.
34
+ */
35
+ export type SlotPool = {
36
+ readonly capacity: number
37
+ active: number
38
+ readonly waiters: Set<SlotHandOff>
39
+ }
40
+
41
+ export const makeSlotPool = (capacity: number): SlotPool => ({
42
+ capacity,
43
+ active: 0,
44
+ waiters: new Set()
45
+ })
46
+
47
+ /** The version names the hand-off protocol above; a change to it needs a new key. */
48
+ const processPoolKey = Symbol.for('@yolk-sdk/extractors/worker-admission/v2')
49
+
50
+ /** Waiters are only ever added by `withSlot`, in this module or a copy of it. */
51
+ const WaiterSet = Schema.declare((value): value is Set<SlotHandOff> => value instanceof Set)
52
+
53
+ /** The pool's shape, checked when another copy of this module may have created it. */
54
+ const isSlotPool = Schema.is(
55
+ Schema.Struct({ capacity: Schema.Number, active: Schema.Number, waiters: WaiterSet })
56
+ )
57
+
58
+ /** The realm-wide pool, created by the first copy of this module that asks for it. */
59
+ export const processSlotPool = (): SlotPool => {
60
+ const existing: unknown = Object.getOwnPropertyDescriptor(globalThis, processPoolKey)?.value
61
+
62
+ if (isSlotPool(existing)) return existing
63
+
64
+ const pool = makeSlotPool(processWorkerLimit)
65
+
66
+ Object.defineProperty(globalThis, processPoolKey, { value: pool, configurable: false })
67
+
68
+ return pool
69
+ }
70
+
71
+ /** Running and waiting extractions in the realm-wide pool. */
72
+ export const processAdmissionSnapshot = () => {
73
+ const pool = processSlotPool()
74
+
75
+ return { active: pool.active, waiting: pool.waiters.size }
76
+ }
77
+
78
+ /** Hand the slot to the first waiter that can still take it; otherwise it becomes free. */
79
+ const releaseSlot = (pool: SlotPool) => {
80
+ for (const handOff of pool.waiters) if (handOff()) return
81
+
82
+ pool.active -= 1
83
+ }
84
+
85
+ /** Whether `deadline` (epoch ms) has passed. */
86
+ const expired = (deadline: number) => Date.now() > deadline
87
+
88
+ /**
89
+ * Run `self` holding one slot of `pool`, waiting for it until `deadline` (epoch ms) and failing
90
+ * with `busy()` after that. The deadline is checked again whenever a slot would be taken, by a new
91
+ * extraction or by a hand-off, so a late wake-up cannot turn an expired wait into admission.
92
+ *
93
+ * Taking a slot and registering its release happen without an interruption point between them,
94
+ * so a slot is never leaked; only the wait is interruptible. A slot handed to a waiter whose
95
+ * fiber is interrupted before it resumes is released again. The wait uses a real timer, not the
96
+ * Effect `Clock`, so a test clock cannot stall it.
97
+ */
98
+ export const withSlot =
99
+ <E2>(pool: SlotPool, deadline: number, busy: () => E2) =>
100
+ <A, E, R>(self: Effect.Effect<A, E, R>): Effect.Effect<A, E | E2, R> =>
101
+ Effect.uninterruptibleMask(restore =>
102
+ Effect.suspend(() => {
103
+ // Set once this fiber owns a slot, before its release is registered.
104
+ let owned = false
105
+
106
+ const admission = Effect.callback<boolean>(resume => {
107
+ if (expired(deadline)) return resume(Effect.succeed(false))
108
+
109
+ if (pool.active < pool.capacity && pool.waiters.size === 0) {
110
+ pool.active += 1
111
+ owned = true
112
+
113
+ return resume(Effect.succeed(true))
114
+ }
115
+
116
+ const handOff: SlotHandOff = () => {
117
+ stop()
118
+
119
+ if (expired(deadline)) {
120
+ queueMicrotask(() => resume(Effect.succeed(false)))
121
+
122
+ return false
123
+ }
124
+
125
+ // The slot is this fiber's from here on. If it is interrupted before the deferred
126
+ // resume runs, Effect ignores that resume and `onInterrupt` passes the slot on.
127
+ owned = true
128
+ queueMicrotask(() => resume(Effect.succeed(true)))
129
+
130
+ return true
131
+ }
132
+
133
+ const timer = setTimeout(() => {
134
+ stop()
135
+ resume(Effect.succeed(false))
136
+ }, deadline - Date.now())
137
+
138
+ const stop = () => {
139
+ clearTimeout(timer)
140
+ pool.waiters.delete(handOff)
141
+ }
142
+
143
+ pool.waiters.add(handOff)
144
+
145
+ return Effect.sync(stop)
146
+ })
147
+
148
+ return restore(admission).pipe(
149
+ // Interrupted after a hand-off but before resuming: pass the slot on.
150
+ Effect.onInterrupt(() =>
151
+ Effect.sync(() => {
152
+ if (owned) releaseSlot(pool)
153
+ })
154
+ ),
155
+ Effect.flatMap((admitted): Effect.Effect<A, E | E2, R> =>
156
+ admitted
157
+ ? restore(self).pipe(Effect.ensuring(Effect.sync(() => releaseSlot(pool))))
158
+ : Effect.fail(busy())
159
+ )
160
+ )
161
+ })
162
+ )