@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
@@ -0,0 +1,253 @@
1
+ import { Buffer } from "node:buffer";
2
+ //#region src/node/sheetjs-xml.ts
3
+ /**
4
+ * Ports of the SheetJS 0.20.3 XML helpers (`xlsx.mjs`) the extractor needs to read a part the way
5
+ * SheetJS would: its tag pattern, `parsexmltag`, `strip_ns`, `utf8read`, and `unescapexml`.
6
+ * Everything here is linear in the text it scans.
7
+ */
8
+ const swapUtf16ByteOrder = (bytes) => Buffer.from(bytes.subarray(0, bytes.length - bytes.length % 2)).swap16();
9
+ /**
10
+ * The texts SheetJS can read from a part: its Latin-1 ("binary") view and, for BOM-marked parts,
11
+ * the UTF-16 decodings of `cc2str` (little- and big-endian from byte 2, including its
12
+ * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
13
+ */
14
+ const sheetJsTextViews = (content) => {
15
+ const bytes = Buffer.from(content.buffer, content.byteOffset, content.byteLength);
16
+ const latin1 = bytes.toString("latin1");
17
+ const littleEndian = bytes[0] === 255 && bytes[1] === 254;
18
+ const bigEndian = bytes[0] === 254 && bytes[1] === 255;
19
+ const offsetBigEndian = bytes[1] === 254 && bytes[2] === 255;
20
+ if (!littleEndian && !bigEndian && !offsetBigEndian) return [latin1];
21
+ return [
22
+ latin1,
23
+ bytes.subarray(2).toString("utf16le"),
24
+ swapUtf16ByteOrder(bytes.subarray(2)).toString("utf16le"),
25
+ swapUtf16ByteOrder(bytes.subarray(3)).toString("utf16le")
26
+ ];
27
+ };
28
+ /**
29
+ * SheetJS's own tag pattern (`tagregex1`, used for every part it parses): quoted values may hold
30
+ * `<` and `>`. Each attempt stops at the next quote of its kind, so a scan stays linear.
31
+ */
32
+ const sheetJsTagPattern = /<[/?]?[a-zA-Z0-9:_-]+(?:\s+[^"\s?<>/]+\s*=\s*(?:"[^"]*"|'[^']*'|[^'"<>\s=]+))*\s*[/?]?>/gm;
33
+ /** SheetJS `attregexg`. */
34
+ const sheetJsAttribute = /\s([^"\s?>/]+)\s*=\s*((?:")([^"]*)(?:")|(?:')([^']*)(?:')|([^'">\s]+))/g;
35
+ /**
36
+ * Port of SheetJS `parsexmltag`: exact-case keys (plus lower-cased copies), a namespace prefix
37
+ * dropped, an unprefixed name cut at its first `_`, the last value winning. Values are raw.
38
+ */
39
+ const parseSheetJsTag = (tag) => {
40
+ let end = 0;
41
+ for (; end < tag.length; end += 1) {
42
+ const code = tag.charCodeAt(end);
43
+ if (code === 32 || code === 10 || code === 13) break;
44
+ }
45
+ const attributes = /* @__PURE__ */ new Map();
46
+ if (end === tag.length) return {
47
+ head: tag,
48
+ attributes
49
+ };
50
+ for (const [match] of tag.matchAll(sheetJsAttribute)) {
51
+ const text = match.slice(1);
52
+ let equals = text.indexOf("=");
53
+ let name = text.slice(0, equals).trim();
54
+ while (text.charCodeAt(equals + 1) === 32) equals += 1;
55
+ const quoteCode = text.charCodeAt(equals + 1);
56
+ const quoted = quoteCode === 34 || quoteCode === 39 ? 1 : 0;
57
+ const value = text.slice(equals + 1 + quoted, text.length - quoted);
58
+ const colon = name.indexOf(":");
59
+ if (colon < 0) {
60
+ if (name.indexOf("_") > 0) name = name.slice(0, name.indexOf("_"));
61
+ } else {
62
+ const local = (colon === 5 && name.startsWith("xmlns") ? "xmlns" : "") + name.slice(colon + 1);
63
+ if (attributes.has(local) && name.slice(colon - 3, colon) === "ext") continue;
64
+ name = local;
65
+ }
66
+ attributes.set(name, value);
67
+ attributes.set(name.toLowerCase(), value);
68
+ }
69
+ return {
70
+ head: tag.slice(0, end),
71
+ attributes
72
+ };
73
+ };
74
+ /** Every tag SheetJS's pattern finds in `text`, in document order. */
75
+ function* sheetJsTags(text) {
76
+ for (const [tag] of text.matchAll(sheetJsTagPattern)) yield parseSheetJsTag(tag);
77
+ }
78
+ /** SheetJS `strip_ns`: the first `<prefix:` (or `</prefix:`) loses its prefix. */
79
+ const stripSheetJsNamespace = (head) => head.replace(/<(\/?)\w+:/, "<$1");
80
+ /** SheetJS `utf8read` in Node: the Latin-1 ("binary") string read back as UTF-8. */
81
+ const sheetJsUtf8Read = (binary) => Buffer.from(binary, "latin1").toString("utf8");
82
+ const encodings = new Map([
83
+ ["&quot;", "\""],
84
+ ["&apos;", "'"],
85
+ ["&gt;", ">"],
86
+ ["&lt;", "<"],
87
+ ["&amp;", "&"]
88
+ ]);
89
+ /**
90
+ * SheetJS `unescapexml` for text without CDATA, quirks included: entity names match ignoring
91
+ * case but only lower-case ones map (`&QUOT;` becomes U+0000), `&#X41;` is read as decimal, and
92
+ * numeric references wrap at U+FFFF. Returns `undefined` for text with a CDATA marker, which
93
+ * SheetJS splits recursively (and never ends for an unterminated one).
94
+ */
95
+ const sheetJsUnescapeXml = (text) => {
96
+ if (text.includes("<![CDATA[")) return void 0;
97
+ return text.replace(/&(?:quot|apos|gt|lt|amp|#x?([\da-fA-F]+));/gi, (entity, code) => encodings.get(entity) ?? String.fromCharCode(Number.parseInt(code ?? "", entity.includes("x") ? 16 : 10))).replace(/_x([\da-fA-F]{4})_/gi, (_, code) => String.fromCharCode(Number.parseInt(code, 16)));
98
+ };
99
+ /** An attribute value as SheetJS reads text attributes: `unescapexml(utf8read(raw))`. */
100
+ const sheetJsAttributeText = (raw) => sheetJsUnescapeXml(sheetJsUtf8Read(raw));
101
+ const isHighSurrogate = (code) => code >= 55296 && code <= 56319;
102
+ const isLowSurrogate = (code) => code >= 56320 && code <= 57343;
103
+ const codeEscape = (code) => `_x${code.toString(16).toUpperCase().padStart(4, "0")}_`;
104
+ const escapedCharacters = new Map([
105
+ ["&", "&amp;"],
106
+ ["<", "&lt;"],
107
+ [">", "&gt;"],
108
+ ["\"", "&quot;"]
109
+ ]);
110
+ const startsCodeEscape = /_x[\da-fA-F]{4}_/iy;
111
+ /**
112
+ * Escape `text` for a double-quoted attribute of a generated UTF-8 part so that SheetJS's
113
+ * `unescapexml(utf8read(…))` returns `text` exactly: markup characters become entities, an `_`
114
+ * that would start an `_xHHHH_` code becomes `_x005F_`, and control characters, U+FFFE, U+FFFF,
115
+ * and lone surrogates become `_xHHHH_` codes.
116
+ */
117
+ const sheetJsAttributeEscape = (text) => {
118
+ let output = "";
119
+ for (let index = 0; index < text.length; index += 1) {
120
+ const character = text.charAt(index);
121
+ const code = text.charCodeAt(index);
122
+ const escaped = escapedCharacters.get(character);
123
+ if (escaped !== void 0) {
124
+ output += escaped;
125
+ continue;
126
+ }
127
+ if (character === "_") {
128
+ startsCodeEscape.lastIndex = index;
129
+ output += startsCodeEscape.test(text) ? codeEscape(code) : character;
130
+ continue;
131
+ }
132
+ if (isHighSurrogate(code) && isLowSurrogate(text.charCodeAt(index + 1))) {
133
+ output += text.slice(index, index + 2);
134
+ index += 1;
135
+ continue;
136
+ }
137
+ output += code < 32 || code === 65534 || code === 65535 || isHighSurrogate(code) || isLowSurrogate(code) ? codeEscape(code) : character;
138
+ }
139
+ return output;
140
+ };
141
+ /** SheetJS's own `XML_HEADER`, used for every generated part. */
142
+ const sheetJsXmlHeader = "<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"yes\"?>\r\n";
143
+ const spreadsheetMainNamespace = "http://schemas.openxmlformats.org/spreadsheetml/2006/main";
144
+ const officeDocumentRelationshipsNamespace = "http://schemas.openxmlformats.org/officeDocument/2006/relationships";
145
+ const cdataMarker = "<![CDATA[";
146
+ /** The two conversions SheetJS chains over cell and shared-string text. */
147
+ const sheetJsTextSteps = [sheetJsUnescapeXml, sheetJsUtf8Read];
148
+ /** Depth first, so at most `steps + 1` derived strings are alive at once. */
149
+ const meetsCdata = (text, steps) => {
150
+ if (text.includes(cdataMarker)) return true;
151
+ if (steps === 0) return false;
152
+ return sheetJsTextSteps.some((step) => {
153
+ const next = step(text);
154
+ return next !== void 0 && next !== text && meetsCdata(next, steps - 1);
155
+ });
156
+ };
157
+ const isNameCode = (code) => code >= 48 && code <= 57 || code >= 65 && code <= 90 || code >= 97 && code <= 122 || code === 95 || code === 46 || code === 45;
158
+ /** The end of the run of name characters (`[\w.-]`) starting at `start`. */
159
+ const nameEnd = (text, start) => {
160
+ let end = start;
161
+ while (end < text.length && isNameCode(text.charCodeAt(end))) end += 1;
162
+ return end;
163
+ };
164
+ /**
165
+ * `text` without its simple opening tags, `<(?:[\w.-]+:)?[\w.-]+>`: a superset of the tags SheetJS
166
+ * removes before decoding (`<(?:\w+:)?(?:si|sstItem)>` in `parse_sst_xml`, `<(?:\w+:)?r>` in
167
+ * `parse_rs`). One forward scan: a `<` that does not start such a tag is kept and the scan resumes
168
+ * at the next `<`, so every character is read at most twice.
169
+ */
170
+ const withoutSimpleTags = (text) => {
171
+ let output = "";
172
+ let kept = 0;
173
+ let index = text.indexOf("<");
174
+ while (index >= 0) {
175
+ let end = nameEnd(text, index + 1);
176
+ if (end > index + 1 && text.charCodeAt(end) === 58) {
177
+ const local = nameEnd(text, end + 1);
178
+ end = local > end + 1 ? local : -1;
179
+ }
180
+ if (end > index + 1 && text.charCodeAt(end) === 62) {
181
+ output += text.slice(kept, index);
182
+ kept = end + 1;
183
+ index = text.indexOf("<", kept);
184
+ } else index = text.indexOf("<", index + 1);
185
+ }
186
+ return output + text.slice(kept);
187
+ };
188
+ /** `<<` or `<!`: never written in worksheets or shared strings by Excel, LibreOffice, or Sheets. */
189
+ const hasMarkupOpener = (text) => text.includes("<<") || text.includes("<!");
190
+ /**
191
+ * Whether SheetJS could meet a CDATA marker in this part. SheetJS's `unescapexml` handles CDATA by
192
+ * recursing on a string two characters shorter and copying the whole tail at every level, so an
193
+ * unterminated marker costs quadratic time and memory.
194
+ *
195
+ * What SheetJS hands to `unescapexml` in a worksheet or shared-strings part comes from the part
196
+ * text through two kinds of transformation:
197
+ *
198
+ * - Decodes: raw (`<v>` of every cell), `utf8read(raw)` (shared and inline strings), and
199
+ * `utf8read(unescapexml(raw))` (cells of type `str`, decoded again after `utf8read`).
200
+ * `utf8read` keeps only the low byte of each character, so U+013C from `_x013C_`, `&#x13C;`, or
201
+ * `&#316;` becomes `<`.
202
+ * - Tag removal before decoding: `parse_sst_xml` removes every `<si>`/`<sstItem>` opening tag
203
+ * from the whole shared-strings table, and `parse_rs` removes every `<r>` opening tag from rich
204
+ * text (after `utf8read`). Inline strings (`t="inlineStr"`) call `parse_si` without options,
205
+ * so their rich text is processed even with `cellHTML: false`. `A<<r>![CDATA[B` thus reaches
206
+ * `unescapexml` as `A<![CDATA[B`.
207
+ *
208
+ * So the check rejects a text view (`sheetJsTextViews`) when:
209
+ *
210
+ * - the view, or `utf8read` of it, contains `<<` or `<!`. A marker assembled by removing tags
211
+ * needs a literal `<` (in the view, or from `utf8read`) followed by a removed tag, or by `!`,
212
+ * and every removed tag starts with `<`;
213
+ * - the view, or the view without any simple opening tag (`withoutSimpleTags`, a superset of
214
+ * SheetJS's removals), meets the marker raw or after any chain of up to two steps of
215
+ * `unescapexml` and `utf8read`, in any order (a superset of SheetJS's decode sequences).
216
+ *
217
+ * Every step is a linear pass, at most a few per view. The first rule deliberately fails closed:
218
+ * it also rejects XML comments, `<!DOCTYPE`, and any other `<!…` declaration, which Excel,
219
+ * LibreOffice, and Google Sheets never write in worksheets or shared strings (nor a literal `<<`).
220
+ */
221
+ const sheetJsCouldReadCdata = (content) => sheetJsTextViews(content).some((view) => hasMarkupOpener(view) || hasMarkupOpener(sheetJsUtf8Read(view)) || meetsCdata(view, 2) || meetsCdata(withoutSimpleTags(view), 2));
222
+ const xmlBoundary = new Set([
223
+ " ",
224
+ " ",
225
+ "\r",
226
+ "\n",
227
+ ">"
228
+ ]);
229
+ /** SheetJS `str_match_xml`: the first `<tag` element's inner text (exact prefix and case). */
230
+ const sheetJsElementText = (text, tag) => {
231
+ const width = tag.length + 1;
232
+ let start = text.indexOf(`<${tag}`);
233
+ while (start >= 0 && start <= text.length - width && !xmlBoundary.has(text.charAt(start + width))) start = text.indexOf(`<${tag}`, start + 1);
234
+ if (start === -1) return void 0;
235
+ const contentStart = text.indexOf(">", start + tag.length);
236
+ if (contentStart === -1) return void 0;
237
+ const end = text.indexOf(`</${tag}>`, contentStart);
238
+ return end === -1 ? void 0 : text.slice(contentStart + 1, end);
239
+ };
240
+ /**
241
+ * The `dc:title` of a core-properties part, read as SheetJS `parse_core_props` reads it, without
242
+ * handing the part to SheetJS. A title holding CDATA is ignored.
243
+ */
244
+ const coreTitle = (content) => {
245
+ if (content === void 0) return void 0;
246
+ const raw = sheetJsElementText(Buffer.from(content.buffer, content.byteOffset, content.byteLength).toString("utf8"), "dc:title");
247
+ const title = raw === void 0 ? void 0 : sheetJsUnescapeXml(raw);
248
+ return title !== void 0 && title.trim().length > 0 ? title : void 0;
249
+ };
250
+ //#endregion
251
+ export { coreTitle, officeDocumentRelationshipsNamespace, parseSheetJsTag, sheetJsAttributeEscape, sheetJsAttributeText, sheetJsCouldReadCdata, sheetJsTags, sheetJsTextViews, sheetJsUnescapeXml, sheetJsUtf8Read, sheetJsXmlHeader, spreadsheetMainNamespace, stripSheetJsNamespace, withoutSimpleTags };
252
+
253
+ //# sourceMappingURL=sheetjs-xml.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sheetjs-xml.mjs","names":[],"sources":["../../src/node/sheetjs-xml.ts"],"sourcesContent":["import { Buffer } from 'node:buffer'\n\n/**\n * Ports of the SheetJS 0.20.3 XML helpers (`xlsx.mjs`) the extractor needs to read a part the way\n * SheetJS would: its tag pattern, `parsexmltag`, `strip_ns`, `utf8read`, and `unescapexml`.\n * Everything here is linear in the text it scans.\n */\n\nconst swapUtf16ByteOrder = (bytes: Buffer) =>\n Buffer.from(bytes.subarray(0, bytes.length - (bytes.length % 2))).swap16()\n\n/**\n * The texts SheetJS can read from a part: its Latin-1 (\"binary\") view and, for BOM-marked parts,\n * the UTF-16 decodings of `cc2str` (little- and big-endian from byte 2, including its\n * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.\n */\nexport const sheetJsTextViews = (content: Uint8Array): ReadonlyArray<string> => {\n const bytes = Buffer.from(content.buffer, content.byteOffset, content.byteLength)\n const latin1 = bytes.toString('latin1')\n const littleEndian = bytes[0] === 0xff && bytes[1] === 0xfe\n const bigEndian = bytes[0] === 0xfe && bytes[1] === 0xff\n const offsetBigEndian = bytes[1] === 0xfe && bytes[2] === 0xff\n\n if (!littleEndian && !bigEndian && !offsetBigEndian) return [latin1]\n\n return [\n latin1,\n bytes.subarray(2).toString('utf16le'),\n swapUtf16ByteOrder(bytes.subarray(2)).toString('utf16le'),\n swapUtf16ByteOrder(bytes.subarray(3)).toString('utf16le')\n ]\n}\n\n/**\n * SheetJS's own tag pattern (`tagregex1`, used for every part it parses): quoted values may hold\n * `<` and `>`. Each attempt stops at the next quote of its kind, so a scan stays linear.\n */\nconst sheetJsTagPattern =\n /<[/?]?[a-zA-Z0-9:_-]+(?:\\s+[^\"\\s?<>/]+\\s*=\\s*(?:\"[^\"]*\"|'[^']*'|[^'\"<>\\s=]+))*\\s*[/?]?>/gm\n\n/** SheetJS `attregexg`. */\nconst sheetJsAttribute = /\\s([^\"\\s?>/]+)\\s*=\\s*((?:\")([^\"]*)(?:\")|(?:')([^']*)(?:')|([^'\">\\s]+))/g\n\nexport type SheetJsTag = {\n /** The tag up to its first space, line feed, or carriage return (SheetJS `y[0]`). */\n readonly head: string\n /** Raw (still escaped) attribute values by SheetJS key, plus lower-cased copies. */\n readonly attributes: ReadonlyMap<string, string>\n}\n\n/**\n * Port of SheetJS `parsexmltag`: exact-case keys (plus lower-cased copies), a namespace prefix\n * dropped, an unprefixed name cut at its first `_`, the last value winning. Values are raw.\n */\nexport const parseSheetJsTag = (tag: string): SheetJsTag => {\n let end = 0\n\n for (; end < tag.length; end += 1) {\n const code = tag.charCodeAt(end)\n\n if (code === 32 || code === 10 || code === 13) break\n }\n\n const attributes = new Map<string, string>()\n\n if (end === tag.length) return { head: tag, attributes }\n\n for (const [match] of tag.matchAll(sheetJsAttribute)) {\n const text = match.slice(1)\n let equals = text.indexOf('=')\n let name = text.slice(0, equals).trim()\n\n while (text.charCodeAt(equals + 1) === 32) equals += 1\n\n const quoteCode = text.charCodeAt(equals + 1)\n const quoted = quoteCode === 34 || quoteCode === 39 ? 1 : 0\n const value = text.slice(equals + 1 + quoted, text.length - quoted)\n const colon = name.indexOf(':')\n\n if (colon < 0) {\n if (name.indexOf('_') > 0) name = name.slice(0, name.indexOf('_'))\n } else {\n const local = (colon === 5 && name.startsWith('xmlns') ? 'xmlns' : '') + name.slice(colon + 1)\n\n if (attributes.has(local) && name.slice(colon - 3, colon) === 'ext') continue\n\n name = local\n }\n\n attributes.set(name, value)\n attributes.set(name.toLowerCase(), value)\n }\n\n return { head: tag.slice(0, end), attributes }\n}\n\n/** Every tag SheetJS's pattern finds in `text`, in document order. */\nexport function* sheetJsTags(text: string): Generator<SheetJsTag> {\n for (const [tag] of text.matchAll(sheetJsTagPattern)) yield parseSheetJsTag(tag)\n}\n\n/** SheetJS `strip_ns`: the first `<prefix:` (or `</prefix:`) loses its prefix. */\nexport const stripSheetJsNamespace = (head: string) => head.replace(/<(\\/?)\\w+:/, '<$1')\n\n/** SheetJS `utf8read` in Node: the Latin-1 (\"binary\") string read back as UTF-8. */\nexport const sheetJsUtf8Read = (binary: string) => Buffer.from(binary, 'latin1').toString('utf8')\n\nconst encodings: ReadonlyMap<string, string> = new Map([\n ['&quot;', '\"'],\n ['&apos;', \"'\"],\n ['&gt;', '>'],\n ['&lt;', '<'],\n ['&amp;', '&']\n])\n\n/**\n * SheetJS `unescapexml` for text without CDATA, quirks included: entity names match ignoring\n * case but only lower-case ones map (`&QUOT;` becomes U+0000), `&#X41;` is read as decimal, and\n * numeric references wrap at U+FFFF. Returns `undefined` for text with a CDATA marker, which\n * SheetJS splits recursively (and never ends for an unterminated one).\n */\nexport const sheetJsUnescapeXml = (text: string): string | undefined => {\n if (text.includes('<![CDATA[')) return undefined\n\n return text\n .replace(\n /&(?:quot|apos|gt|lt|amp|#x?([\\da-fA-F]+));/gi,\n (entity, code: string | undefined) =>\n encodings.get(entity) ??\n String.fromCharCode(Number.parseInt(code ?? '', entity.includes('x') ? 16 : 10))\n )\n .replace(/_x([\\da-fA-F]{4})_/gi, (_, code: string) =>\n String.fromCharCode(Number.parseInt(code, 16))\n )\n}\n\n/** An attribute value as SheetJS reads text attributes: `unescapexml(utf8read(raw))`. */\nexport const sheetJsAttributeText = (raw: string) => sheetJsUnescapeXml(sheetJsUtf8Read(raw))\n\nconst isHighSurrogate = (code: number) => code >= 0xd800 && code <= 0xdbff\n\nconst isLowSurrogate = (code: number) => code >= 0xdc00 && code <= 0xdfff\n\nconst codeEscape = (code: number) => `_x${code.toString(16).toUpperCase().padStart(4, '0')}_`\n\nconst escapedCharacters: ReadonlyMap<string, string> = new Map([\n ['&', '&amp;'],\n ['<', '&lt;'],\n ['>', '&gt;'],\n ['\"', '&quot;']\n])\n\n// SheetJS matches `_xHHHH_` codes ignoring case (`coderegex`), so `_X0041_` is a code too.\nconst startsCodeEscape = /_x[\\da-fA-F]{4}_/iy\n\n/**\n * Escape `text` for a double-quoted attribute of a generated UTF-8 part so that SheetJS's\n * `unescapexml(utf8read(…))` returns `text` exactly: markup characters become entities, an `_`\n * that would start an `_xHHHH_` code becomes `_x005F_`, and control characters, U+FFFE, U+FFFF,\n * and lone surrogates become `_xHHHH_` codes.\n */\nexport const sheetJsAttributeEscape = (text: string) => {\n let output = ''\n\n for (let index = 0; index < text.length; index += 1) {\n const character = text.charAt(index)\n const code = text.charCodeAt(index)\n const escaped = escapedCharacters.get(character)\n\n if (escaped !== undefined) {\n output += escaped\n continue\n }\n\n if (character === '_') {\n startsCodeEscape.lastIndex = index\n output += startsCodeEscape.test(text) ? codeEscape(code) : character\n continue\n }\n\n if (isHighSurrogate(code) && isLowSurrogate(text.charCodeAt(index + 1))) {\n output += text.slice(index, index + 2)\n index += 1\n continue\n }\n\n output +=\n code < 0x20 ||\n code === 0xfffe ||\n code === 0xffff ||\n isHighSurrogate(code) ||\n isLowSurrogate(code)\n ? codeEscape(code)\n : character\n }\n\n return output\n}\n\n/** SheetJS's own `XML_HEADER`, used for every generated part. */\nexport const sheetJsXmlHeader = '<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"yes\"?>\\r\\n'\n\nexport const spreadsheetMainNamespace = 'http://schemas.openxmlformats.org/spreadsheetml/2006/main'\n\nexport const officeDocumentRelationshipsNamespace =\n 'http://schemas.openxmlformats.org/officeDocument/2006/relationships'\n\nconst cdataMarker = '<![CDATA['\n\ntype TextStep = (text: string) => string | undefined\n\n/** The two conversions SheetJS chains over cell and shared-string text. */\nconst sheetJsTextSteps: ReadonlyArray<TextStep> = [sheetJsUnescapeXml, sheetJsUtf8Read]\n\n/** Depth first, so at most `steps + 1` derived strings are alive at once. */\nconst meetsCdata = (text: string, steps: number): boolean => {\n if (text.includes(cdataMarker)) return true\n\n if (steps === 0) return false\n\n return sheetJsTextSteps.some(step => {\n const next = step(text)\n\n return next !== undefined && next !== text && meetsCdata(next, steps - 1)\n })\n}\n\nconst isNameCode = (code: number) =>\n (code >= 48 && code <= 57) || // 0-9\n (code >= 65 && code <= 90) || // A-Z\n (code >= 97 && code <= 122) || // a-z\n code === 95 || // _\n code === 46 || // .\n code === 45 // -\n\n/** The end of the run of name characters (`[\\w.-]`) starting at `start`. */\nconst nameEnd = (text: string, start: number) => {\n let end = start\n\n while (end < text.length && isNameCode(text.charCodeAt(end))) end += 1\n\n return end\n}\n\n/**\n * `text` without its simple opening tags, `<(?:[\\w.-]+:)?[\\w.-]+>`: a superset of the tags SheetJS\n * removes before decoding (`<(?:\\w+:)?(?:si|sstItem)>` in `parse_sst_xml`, `<(?:\\w+:)?r>` in\n * `parse_rs`). One forward scan: a `<` that does not start such a tag is kept and the scan resumes\n * at the next `<`, so every character is read at most twice.\n */\nexport const withoutSimpleTags = (text: string) => {\n let output = ''\n let kept = 0\n let index = text.indexOf('<')\n\n while (index >= 0) {\n let end = nameEnd(text, index + 1)\n\n if (end > index + 1 && text.charCodeAt(end) === 58) {\n const local = nameEnd(text, end + 1)\n\n end = local > end + 1 ? local : -1\n }\n\n if (end > index + 1 && text.charCodeAt(end) === 62) {\n output += text.slice(kept, index)\n kept = end + 1\n index = text.indexOf('<', kept)\n } else {\n index = text.indexOf('<', index + 1)\n }\n }\n\n return output + text.slice(kept)\n}\n\n/** `<<` or `<!`: never written in worksheets or shared strings by Excel, LibreOffice, or Sheets. */\nconst hasMarkupOpener = (text: string) => text.includes('<<') || text.includes('<!')\n\n/**\n * Whether SheetJS could meet a CDATA marker in this part. SheetJS's `unescapexml` handles CDATA by\n * recursing on a string two characters shorter and copying the whole tail at every level, so an\n * unterminated marker costs quadratic time and memory.\n *\n * What SheetJS hands to `unescapexml` in a worksheet or shared-strings part comes from the part\n * text through two kinds of transformation:\n *\n * - Decodes: raw (`<v>` of every cell), `utf8read(raw)` (shared and inline strings), and\n * `utf8read(unescapexml(raw))` (cells of type `str`, decoded again after `utf8read`).\n * `utf8read` keeps only the low byte of each character, so U+013C from `_x013C_`, `&#x13C;`, or\n * `&#316;` becomes `<`.\n * - Tag removal before decoding: `parse_sst_xml` removes every `<si>`/`<sstItem>` opening tag\n * from the whole shared-strings table, and `parse_rs` removes every `<r>` opening tag from rich\n * text (after `utf8read`). Inline strings (`t=\"inlineStr\"`) call `parse_si` without options,\n * so their rich text is processed even with `cellHTML: false`. `A<<r>![CDATA[B` thus reaches\n * `unescapexml` as `A<![CDATA[B`.\n *\n * So the check rejects a text view (`sheetJsTextViews`) when:\n *\n * - the view, or `utf8read` of it, contains `<<` or `<!`. A marker assembled by removing tags\n * needs a literal `<` (in the view, or from `utf8read`) followed by a removed tag, or by `!`,\n * and every removed tag starts with `<`;\n * - the view, or the view without any simple opening tag (`withoutSimpleTags`, a superset of\n * SheetJS's removals), meets the marker raw or after any chain of up to two steps of\n * `unescapexml` and `utf8read`, in any order (a superset of SheetJS's decode sequences).\n *\n * Every step is a linear pass, at most a few per view. The first rule deliberately fails closed:\n * it also rejects XML comments, `<!DOCTYPE`, and any other `<!…` declaration, which Excel,\n * LibreOffice, and Google Sheets never write in worksheets or shared strings (nor a literal `<<`).\n */\nexport const sheetJsCouldReadCdata = (content: Uint8Array) =>\n sheetJsTextViews(content).some(\n view =>\n hasMarkupOpener(view) ||\n hasMarkupOpener(sheetJsUtf8Read(view)) ||\n meetsCdata(view, 2) ||\n meetsCdata(withoutSimpleTags(view), 2)\n )\n\nconst xmlBoundary = new Set([' ', '\\t', '\\r', '\\n', '>'])\n\n/** SheetJS `str_match_xml`: the first `<tag` element's inner text (exact prefix and case). */\nconst sheetJsElementText = (text: string, tag: string) => {\n const width = tag.length + 1\n let start = text.indexOf(`<${tag}`)\n\n while (start >= 0 && start <= text.length - width && !xmlBoundary.has(text.charAt(start + width)))\n start = text.indexOf(`<${tag}`, start + 1)\n\n if (start === -1) return undefined\n\n const contentStart = text.indexOf('>', start + tag.length)\n\n if (contentStart === -1) return undefined\n\n const end = text.indexOf(`</${tag}>`, contentStart)\n\n return end === -1 ? undefined : text.slice(contentStart + 1, end)\n}\n\n/**\n * The `dc:title` of a core-properties part, read as SheetJS `parse_core_props` reads it, without\n * handing the part to SheetJS. A title holding CDATA is ignored.\n */\nexport const coreTitle = (content: Uint8Array | undefined) => {\n if (content === undefined) return undefined\n\n const raw = sheetJsElementText(\n Buffer.from(content.buffer, content.byteOffset, content.byteLength).toString('utf8'),\n 'dc:title'\n )\n\n const title = raw === undefined ? undefined : sheetJsUnescapeXml(raw)\n\n return title !== undefined && title.trim().length > 0 ? title : undefined\n}\n"],"mappings":";;;;;;;AAQA,MAAM,sBAAsB,UAC1B,OAAO,KAAK,MAAM,SAAS,GAAG,MAAM,SAAU,MAAM,SAAS,CAAE,CAAC,EAAE,OAAO;;;;;;AAO3E,MAAa,oBAAoB,YAA+C;CAC9E,MAAM,QAAQ,OAAO,KAAK,QAAQ,QAAQ,QAAQ,YAAY,QAAQ,UAAU;CAChF,MAAM,SAAS,MAAM,SAAS,QAAQ;CACtC,MAAM,eAAe,MAAM,OAAO,OAAQ,MAAM,OAAO;CACvD,MAAM,YAAY,MAAM,OAAO,OAAQ,MAAM,OAAO;CACpD,MAAM,kBAAkB,MAAM,OAAO,OAAQ,MAAM,OAAO;CAE1D,IAAI,CAAC,gBAAgB,CAAC,aAAa,CAAC,iBAAiB,OAAO,CAAC,MAAM;CAEnE,OAAO;EACL;EACA,MAAM,SAAS,CAAC,EAAE,SAAS,SAAS;EACpC,mBAAmB,MAAM,SAAS,CAAC,CAAC,EAAE,SAAS,SAAS;EACxD,mBAAmB,MAAM,SAAS,CAAC,CAAC,EAAE,SAAS,SAAS;CAC1D;AACF;;;;;AAMA,MAAM,oBACJ;;AAGF,MAAM,mBAAmB;;;;;AAazB,MAAa,mBAAmB,QAA4B;CAC1D,IAAI,MAAM;CAEV,OAAO,MAAM,IAAI,QAAQ,OAAO,GAAG;EACjC,MAAM,OAAO,IAAI,WAAW,GAAG;EAE/B,IAAI,SAAS,MAAM,SAAS,MAAM,SAAS,IAAI;CACjD;CAEA,MAAM,6BAAa,IAAI,IAAoB;CAE3C,IAAI,QAAQ,IAAI,QAAQ,OAAO;EAAE,MAAM;EAAK;CAAW;CAEvD,KAAK,MAAM,CAAC,UAAU,IAAI,SAAS,gBAAgB,GAAG;EACpD,MAAM,OAAO,MAAM,MAAM,CAAC;EAC1B,IAAI,SAAS,KAAK,QAAQ,GAAG;EAC7B,IAAI,OAAO,KAAK,MAAM,GAAG,MAAM,EAAE,KAAK;EAEtC,OAAO,KAAK,WAAW,SAAS,CAAC,MAAM,IAAI,UAAU;EAErD,MAAM,YAAY,KAAK,WAAW,SAAS,CAAC;EAC5C,MAAM,SAAS,cAAc,MAAM,cAAc,KAAK,IAAI;EAC1D,MAAM,QAAQ,KAAK,MAAM,SAAS,IAAI,QAAQ,KAAK,SAAS,MAAM;EAClE,MAAM,QAAQ,KAAK,QAAQ,GAAG;EAE9B,IAAI,QAAQ;OACN,KAAK,QAAQ,GAAG,IAAI,GAAG,OAAO,KAAK,MAAM,GAAG,KAAK,QAAQ,GAAG,CAAC;EAAA,OAC5D;GACL,MAAM,SAAS,UAAU,KAAK,KAAK,WAAW,OAAO,IAAI,UAAU,MAAM,KAAK,MAAM,QAAQ,CAAC;GAE7F,IAAI,WAAW,IAAI,KAAK,KAAK,KAAK,MAAM,QAAQ,GAAG,KAAK,MAAM,OAAO;GAErE,OAAO;EACT;EAEA,WAAW,IAAI,MAAM,KAAK;EAC1B,WAAW,IAAI,KAAK,YAAY,GAAG,KAAK;CAC1C;CAEA,OAAO;EAAE,MAAM,IAAI,MAAM,GAAG,GAAG;EAAG;CAAW;AAC/C;;AAGA,UAAiB,YAAY,MAAqC;CAChE,KAAK,MAAM,CAAC,QAAQ,KAAK,SAAS,iBAAiB,GAAG,MAAM,gBAAgB,GAAG;AACjF;;AAGA,MAAa,yBAAyB,SAAiB,KAAK,QAAQ,cAAc,KAAK;;AAGvF,MAAa,mBAAmB,WAAmB,OAAO,KAAK,QAAQ,QAAQ,EAAE,SAAS,MAAM;AAEhG,MAAM,YAAyC,IAAI,IAAI;CACrD,CAAC,UAAU,IAAG;CACd,CAAC,UAAU,GAAG;CACd,CAAC,QAAQ,GAAG;CACZ,CAAC,QAAQ,GAAG;CACZ,CAAC,SAAS,GAAG;AACf,CAAC;;;;;;;AAQD,MAAa,sBAAsB,SAAqC;CACtE,IAAI,KAAK,SAAS,WAAW,GAAG,OAAO,KAAA;CAEvC,OAAO,KACJ,QACC,iDACC,QAAQ,SACP,UAAU,IAAI,MAAM,KACpB,OAAO,aAAa,OAAO,SAAS,QAAQ,IAAI,OAAO,SAAS,GAAG,IAAI,KAAK,EAAE,CAAC,CACnF,EACC,QAAQ,yBAAyB,GAAG,SACnC,OAAO,aAAa,OAAO,SAAS,MAAM,EAAE,CAAC,CAC/C;AACJ;;AAGA,MAAa,wBAAwB,QAAgB,mBAAmB,gBAAgB,GAAG,CAAC;AAE5F,MAAM,mBAAmB,SAAiB,QAAQ,SAAU,QAAQ;AAEpE,MAAM,kBAAkB,SAAiB,QAAQ,SAAU,QAAQ;AAEnE,MAAM,cAAc,SAAiB,KAAK,KAAK,SAAS,EAAE,EAAE,YAAY,EAAE,SAAS,GAAG,GAAG,EAAE;AAE3F,MAAM,oBAAiD,IAAI,IAAI;CAC7D,CAAC,KAAK,OAAO;CACb,CAAC,KAAK,MAAM;CACZ,CAAC,KAAK,MAAM;CACZ,CAAC,MAAK,QAAQ;AAChB,CAAC;AAGD,MAAM,mBAAmB;;;;;;;AAQzB,MAAa,0BAA0B,SAAiB;CACtD,IAAI,SAAS;CAEb,KAAK,IAAI,QAAQ,GAAG,QAAQ,KAAK,QAAQ,SAAS,GAAG;EACnD,MAAM,YAAY,KAAK,OAAO,KAAK;EACnC,MAAM,OAAO,KAAK,WAAW,KAAK;EAClC,MAAM,UAAU,kBAAkB,IAAI,SAAS;EAE/C,IAAI,YAAY,KAAA,GAAW;GACzB,UAAU;GACV;EACF;EAEA,IAAI,cAAc,KAAK;GACrB,iBAAiB,YAAY;GAC7B,UAAU,iBAAiB,KAAK,IAAI,IAAI,WAAW,IAAI,IAAI;GAC3D;EACF;EAEA,IAAI,gBAAgB,IAAI,KAAK,eAAe,KAAK,WAAW,QAAQ,CAAC,CAAC,GAAG;GACvE,UAAU,KAAK,MAAM,OAAO,QAAQ,CAAC;GACrC,SAAS;GACT;EACF;EAEA,UACE,OAAO,MACP,SAAS,SACT,SAAS,SACT,gBAAgB,IAAI,KACpB,eAAe,IAAI,IACf,WAAW,IAAI,IACf;CACR;CAEA,OAAO;AACT;;AAGA,MAAa,mBAAmB;AAEhC,MAAa,2BAA2B;AAExC,MAAa,uCACX;AAEF,MAAM,cAAc;;AAKpB,MAAM,mBAA4C,CAAC,oBAAoB,eAAe;;AAGtF,MAAM,cAAc,MAAc,UAA2B;CAC3D,IAAI,KAAK,SAAS,WAAW,GAAG,OAAO;CAEvC,IAAI,UAAU,GAAG,OAAO;CAExB,OAAO,iBAAiB,MAAK,SAAQ;EACnC,MAAM,OAAO,KAAK,IAAI;EAEtB,OAAO,SAAS,KAAA,KAAa,SAAS,QAAQ,WAAW,MAAM,QAAQ,CAAC;CAC1E,CAAC;AACH;AAEA,MAAM,cAAc,SACjB,QAAQ,MAAM,QAAQ,MACtB,QAAQ,MAAM,QAAQ,MACtB,QAAQ,MAAM,QAAQ,OACvB,SAAS,MACT,SAAS,MACT,SAAS;;AAGX,MAAM,WAAW,MAAc,UAAkB;CAC/C,IAAI,MAAM;CAEV,OAAO,MAAM,KAAK,UAAU,WAAW,KAAK,WAAW,GAAG,CAAC,GAAG,OAAO;CAErE,OAAO;AACT;;;;;;;AAQA,MAAa,qBAAqB,SAAiB;CACjD,IAAI,SAAS;CACb,IAAI,OAAO;CACX,IAAI,QAAQ,KAAK,QAAQ,GAAG;CAE5B,OAAO,SAAS,GAAG;EACjB,IAAI,MAAM,QAAQ,MAAM,QAAQ,CAAC;EAEjC,IAAI,MAAM,QAAQ,KAAK,KAAK,WAAW,GAAG,MAAM,IAAI;GAClD,MAAM,QAAQ,QAAQ,MAAM,MAAM,CAAC;GAEnC,MAAM,QAAQ,MAAM,IAAI,QAAQ;EAClC;EAEA,IAAI,MAAM,QAAQ,KAAK,KAAK,WAAW,GAAG,MAAM,IAAI;GAClD,UAAU,KAAK,MAAM,MAAM,KAAK;GAChC,OAAO,MAAM;GACb,QAAQ,KAAK,QAAQ,KAAK,IAAI;EAChC,OACE,QAAQ,KAAK,QAAQ,KAAK,QAAQ,CAAC;CAEvC;CAEA,OAAO,SAAS,KAAK,MAAM,IAAI;AACjC;;AAGA,MAAM,mBAAmB,SAAiB,KAAK,SAAS,IAAI,KAAK,KAAK,SAAS,IAAI;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiCnF,MAAa,yBAAyB,YACpC,iBAAiB,OAAO,EAAE,MACxB,SACE,gBAAgB,IAAI,KACpB,gBAAgB,gBAAgB,IAAI,CAAC,KACrC,WAAW,MAAM,CAAC,KAClB,WAAW,kBAAkB,IAAI,GAAG,CAAC,CACzC;AAEF,MAAM,cAAc,IAAI,IAAI;CAAC;CAAK;CAAM;CAAM;CAAM;AAAG,CAAC;;AAGxD,MAAM,sBAAsB,MAAc,QAAgB;CACxD,MAAM,QAAQ,IAAI,SAAS;CAC3B,IAAI,QAAQ,KAAK,QAAQ,IAAI,KAAK;CAElC,OAAO,SAAS,KAAK,SAAS,KAAK,SAAS,SAAS,CAAC,YAAY,IAAI,KAAK,OAAO,QAAQ,KAAK,CAAC,GAC9F,QAAQ,KAAK,QAAQ,IAAI,OAAO,QAAQ,CAAC;CAE3C,IAAI,UAAU,IAAI,OAAO,KAAA;CAEzB,MAAM,eAAe,KAAK,QAAQ,KAAK,QAAQ,IAAI,MAAM;CAEzD,IAAI,iBAAiB,IAAI,OAAO,KAAA;CAEhC,MAAM,MAAM,KAAK,QAAQ,KAAK,IAAI,IAAI,YAAY;CAElD,OAAO,QAAQ,KAAK,KAAA,IAAY,KAAK,MAAM,eAAe,GAAG,GAAG;AAClE;;;;;AAMA,MAAa,aAAa,YAAoC;CAC5D,IAAI,YAAY,KAAA,GAAW,OAAO,KAAA;CAElC,MAAM,MAAM,mBACV,OAAO,KAAK,QAAQ,QAAQ,QAAQ,YAAY,QAAQ,UAAU,EAAE,SAAS,MAAM,GACnF,UACF;CAEA,MAAM,QAAQ,QAAQ,KAAA,IAAY,KAAA,IAAY,mBAAmB,GAAG;CAEpE,OAAO,UAAU,KAAA,KAAa,MAAM,KAAK,EAAE,SAAS,IAAI,QAAQ,KAAA;AAClE"}
@@ -0,0 +1,62 @@
1
+ import { SheetJsUnavailableError } from "../errors.mjs";
2
+ import { XlsxWorkbook } from "./xlsx-text.mjs";
3
+ import { Effect } from "effect";
4
+
5
+ //#region src/node/sheetjs.d.ts
6
+ /** Loads the SheetJS module. The default is a lazy `import('xlsx')`. */
7
+ type SheetJsLoader = () => Promise<unknown>;
8
+ declare const defaultSheetJsLoader: SheetJsLoader;
9
+ type SheetJs = {
10
+ readonly version: string; /** Parse workbook bytes with `sheetJsReadOptions`; the result is checked before use. */
11
+ readonly read: (bytes: Uint8Array) => unknown;
12
+ };
13
+ /**
14
+ * SheetJS `read` options: cell values and display text (`cell.w`) only.
15
+ *
16
+ * - `cellFormula: false`: no formula text. SheetJS otherwise copies a shifted master formula onto
17
+ * every shared-formula dependent and scans every earlier array formula for each cell, before
18
+ * any extractor budget runs. Cached values still render; formula-only cells render empty.
19
+ * - `cellHTML: false`: no rich-text HTML (`cell.h`) we never read. Inline strings ignore it
20
+ * (SheetJS calls `parse_si` without options), so their rich text is still rendered; the CDATA
21
+ * check (`sheetJsCouldReadCdata`) does not rely on this option.
22
+ * - `cellText: true`: keep the formatted display text (`cell.w`) the CSV uses. SheetJS formats
23
+ * every styled cell inside `read`, with work proportional to the format code, so it only ever
24
+ * reads the generated `xl/styles.xml` (codes of at most 255 characters, `xlsx-styles.ts`).
25
+ * - `cellNF`, `cellStyles`, `cellDates: false`: no format strings or style objects; dates stay
26
+ * serial numbers whose `cell.w` carries the formatted date (`cellStyles` would also force
27
+ * `sheetStubs`).
28
+ * - `sheetStubs: false`: no objects for empty cells.
29
+ * - `bookDeps`, `bookFiles`, `bookProps`, `bookSheets`, `bookVBA: false`: no calculation chain,
30
+ * raw archive, or VBA blob, and a full parse (not the properties-only or names-only modes).
31
+ * - `dense: false`: sheets keyed by address, as `xlsx-text.ts` reads them.
32
+ * - `WTF: false`: per-sheet parse errors skip the sheet instead of throwing.
33
+ *
34
+ * A fresh copy is passed on every call because SheetJS writes defaults into the options object.
35
+ */
36
+ declare const sheetJsReadOptions: {
37
+ readonly type: "array";
38
+ readonly cellFormula: false;
39
+ readonly cellHTML: false;
40
+ readonly cellText: true;
41
+ readonly cellNF: false;
42
+ readonly cellStyles: false;
43
+ readonly cellDates: false;
44
+ readonly sheetStubs: false;
45
+ readonly bookDeps: false;
46
+ readonly bookFiles: false;
47
+ readonly bookProps: false;
48
+ readonly bookSheets: false;
49
+ readonly bookVBA: false;
50
+ readonly dense: false;
51
+ readonly WTF: false;
52
+ };
53
+ /**
54
+ * Load SheetJS lazily and refuse anything that is not SheetJS 0.20.3 or newer. The version must be
55
+ * strict SemVer; anything else fails closed as `invalid`.
56
+ */
57
+ declare const loadSheetJs: (loader: SheetJsLoader) => Effect.Effect<SheetJs, SheetJsUnavailableError, never>;
58
+ /** Check the parsed workbook shape before reading it. */
59
+ declare const asXlsxWorkbook: (parsed: unknown) => XlsxWorkbook | undefined;
60
+ //#endregion
61
+ export { SheetJs, SheetJsLoader, asXlsxWorkbook, defaultSheetJsLoader, loadSheetJs, sheetJsReadOptions };
62
+ //# sourceMappingURL=sheetjs.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sheetjs.d.mts","names":[],"sources":["../../src/node/sheetjs.ts"],"mappings":";;;;;;KAKY,aAAA,SAAsB,OAAO;AAAA,cAE5B,oBAAA,EAAsB,aAAoC;AAAA,KAE3D,OAAA;EAAA,SACD,OAAA,UAL8B;EAAA,SAO9B,IAAA,GAAO,KAAA,EAAO,UAAU;AAAA;;;;AALoC;AAEvE;;;;;;;;;AAGmC;AA0BnC;;;;;;;;;cAAa,kBAAA;EAAA;;;;;;;;;;;;;;;;;;;;cA0EA,WAAA,GAAe,MAAA,EAAQ,aAAA,KAAa,MAAA,CAAA,MAAA,CAAA,OAAA,EAAA,uBAAA;;cA8CpC,cAAA,GAAkB,MAAA,cAAkB,YAAY"}
@@ -0,0 +1,122 @@
1
+ import { SheetJsUnavailableError, minimumSheetJsVersion } from "../errors.mjs";
2
+ import { Effect, Predicate } from "effect";
3
+ //#region src/node/sheetjs.ts
4
+ const defaultSheetJsLoader = () => import("xlsx");
5
+ /**
6
+ * SheetJS `read` options: cell values and display text (`cell.w`) only.
7
+ *
8
+ * - `cellFormula: false`: no formula text. SheetJS otherwise copies a shifted master formula onto
9
+ * every shared-formula dependent and scans every earlier array formula for each cell, before
10
+ * any extractor budget runs. Cached values still render; formula-only cells render empty.
11
+ * - `cellHTML: false`: no rich-text HTML (`cell.h`) we never read. Inline strings ignore it
12
+ * (SheetJS calls `parse_si` without options), so their rich text is still rendered; the CDATA
13
+ * check (`sheetJsCouldReadCdata`) does not rely on this option.
14
+ * - `cellText: true`: keep the formatted display text (`cell.w`) the CSV uses. SheetJS formats
15
+ * every styled cell inside `read`, with work proportional to the format code, so it only ever
16
+ * reads the generated `xl/styles.xml` (codes of at most 255 characters, `xlsx-styles.ts`).
17
+ * - `cellNF`, `cellStyles`, `cellDates: false`: no format strings or style objects; dates stay
18
+ * serial numbers whose `cell.w` carries the formatted date (`cellStyles` would also force
19
+ * `sheetStubs`).
20
+ * - `sheetStubs: false`: no objects for empty cells.
21
+ * - `bookDeps`, `bookFiles`, `bookProps`, `bookSheets`, `bookVBA: false`: no calculation chain,
22
+ * raw archive, or VBA blob, and a full parse (not the properties-only or names-only modes).
23
+ * - `dense: false`: sheets keyed by address, as `xlsx-text.ts` reads them.
24
+ * - `WTF: false`: per-sheet parse errors skip the sheet instead of throwing.
25
+ *
26
+ * A fresh copy is passed on every call because SheetJS writes defaults into the options object.
27
+ */
28
+ const sheetJsReadOptions = {
29
+ type: "array",
30
+ cellFormula: false,
31
+ cellHTML: false,
32
+ cellText: true,
33
+ cellNF: false,
34
+ cellStyles: false,
35
+ cellDates: false,
36
+ sheetStubs: false,
37
+ bookDeps: false,
38
+ bookFiles: false,
39
+ bookProps: false,
40
+ bookSheets: false,
41
+ bookVBA: false,
42
+ dense: false,
43
+ WTF: false
44
+ };
45
+ /** SemVer 2.0.0: `major.minor.patch`, optional `-prerelease`, optional `+build`. */
46
+ const semver = /^(0|[1-9]\d*)\.(0|[1-9]\d*)\.(0|[1-9]\d*)(?:-((?:0|[1-9]\d*|\d*[a-zA-Z-][0-9a-zA-Z-]*)(?:\.(?:0|[1-9]\d*|\d*[a-zA-Z-][0-9a-zA-Z-]*))*))?(?:\+[0-9a-zA-Z-]+(?:\.[0-9a-zA-Z-]+)*)?$/;
47
+ const parseVersion = (version) => {
48
+ const match = semver.exec(version);
49
+ if (match === null) return void 0;
50
+ const release = [
51
+ Number(match[1]),
52
+ Number(match[2]),
53
+ Number(match[3])
54
+ ];
55
+ return release.every(Number.isSafeInteger) ? {
56
+ release,
57
+ prerelease: match[4] !== void 0
58
+ } : void 0;
59
+ };
60
+ /**
61
+ * Compare by SemVer precedence against the minimum release: a prerelease of 0.20.3 is below it,
62
+ * prereleases of later releases are above it, and build metadata is ignored.
63
+ */
64
+ const meetsMinimumVersion = (installed, minimum) => {
65
+ for (const [index, part] of installed.release.entries()) {
66
+ const required = minimum.release[index] ?? 0;
67
+ if (part !== required) return part > required;
68
+ }
69
+ return !installed.prerelease;
70
+ };
71
+ const isMissingModule = (cause) => Predicate.hasProperty(cause, "code") && (cause.code === "ERR_MODULE_NOT_FOUND" || cause.code === "MODULE_NOT_FOUND");
72
+ /** ESM namespace, or a CommonJS interop namespace whose `default` is the module. */
73
+ const sheetJsExports = (namespace) => {
74
+ if (Predicate.hasProperty(namespace, "read")) return namespace;
75
+ if (Predicate.hasProperty(namespace, "default") && Predicate.hasProperty(namespace.default, "read")) return namespace.default;
76
+ };
77
+ /**
78
+ * Load SheetJS lazily and refuse anything that is not SheetJS 0.20.3 or newer. The version must be
79
+ * strict SemVer; anything else fails closed as `invalid`.
80
+ */
81
+ const loadSheetJs = (loader) => Effect.gen(function* () {
82
+ const exports = sheetJsExports(yield* Effect.tryPromise({
83
+ try: loader,
84
+ catch: (cause) => new SheetJsUnavailableError({
85
+ reason: isMissingModule(cause) ? "missing" : "invalid",
86
+ cause
87
+ })
88
+ }));
89
+ if (exports === void 0 || !Predicate.hasProperty(exports, "read") || !Predicate.isFunction(exports.read) || !Predicate.hasProperty(exports, "version") || !Predicate.isString(exports.version)) return yield* Effect.fail(new SheetJsUnavailableError({ reason: "invalid" }));
90
+ const { read, version } = exports;
91
+ const installed = parseVersion(version);
92
+ const minimum = parseVersion(minimumSheetJsVersion);
93
+ if (installed === void 0 || minimum === void 0) return yield* Effect.fail(new SheetJsUnavailableError({
94
+ reason: "invalid",
95
+ installedVersion: version
96
+ }));
97
+ if (!meetsMinimumVersion(installed, minimum)) return yield* Effect.fail(new SheetJsUnavailableError({
98
+ reason: "outdated",
99
+ installedVersion: version
100
+ }));
101
+ return {
102
+ version,
103
+ read: (bytes) => read(bytes, { ...sheetJsReadOptions })
104
+ };
105
+ });
106
+ /** Check the parsed workbook shape before reading it. */
107
+ const asXlsxWorkbook = (parsed) => {
108
+ if (!Predicate.hasProperty(parsed, "SheetNames") || !Predicate.hasProperty(parsed, "Sheets") || !Array.isArray(parsed.SheetNames) || !Predicate.isObject(parsed.Sheets)) return void 0;
109
+ const sheetNames = [];
110
+ for (const name of parsed.SheetNames) {
111
+ if (!Predicate.isString(name)) return void 0;
112
+ sheetNames.push(name);
113
+ }
114
+ return {
115
+ SheetNames: sheetNames,
116
+ Sheets: parsed.Sheets
117
+ };
118
+ };
119
+ //#endregion
120
+ export { asXlsxWorkbook, defaultSheetJsLoader, loadSheetJs, sheetJsReadOptions };
121
+
122
+ //# sourceMappingURL=sheetjs.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sheetjs.mjs","names":[],"sources":["../../src/node/sheetjs.ts"],"sourcesContent":["import { Effect, Predicate } from 'effect'\nimport { minimumSheetJsVersion, SheetJsUnavailableError } from '../errors.ts'\nimport type { XlsxWorkbook } from './xlsx-text.ts'\n\n/** Loads the SheetJS module. The default is a lazy `import('xlsx')`. */\nexport type SheetJsLoader = () => Promise<unknown>\n\nexport const defaultSheetJsLoader: SheetJsLoader = () => import('xlsx')\n\nexport type SheetJs = {\n readonly version: string\n /** Parse workbook bytes with `sheetJsReadOptions`; the result is checked before use. */\n readonly read: (bytes: Uint8Array) => unknown\n}\n\n/**\n * SheetJS `read` options: cell values and display text (`cell.w`) only.\n *\n * - `cellFormula: false`: no formula text. SheetJS otherwise copies a shifted master formula onto\n * every shared-formula dependent and scans every earlier array formula for each cell, before\n * any extractor budget runs. Cached values still render; formula-only cells render empty.\n * - `cellHTML: false`: no rich-text HTML (`cell.h`) we never read. Inline strings ignore it\n * (SheetJS calls `parse_si` without options), so their rich text is still rendered; the CDATA\n * check (`sheetJsCouldReadCdata`) does not rely on this option.\n * - `cellText: true`: keep the formatted display text (`cell.w`) the CSV uses. SheetJS formats\n * every styled cell inside `read`, with work proportional to the format code, so it only ever\n * reads the generated `xl/styles.xml` (codes of at most 255 characters, `xlsx-styles.ts`).\n * - `cellNF`, `cellStyles`, `cellDates: false`: no format strings or style objects; dates stay\n * serial numbers whose `cell.w` carries the formatted date (`cellStyles` would also force\n * `sheetStubs`).\n * - `sheetStubs: false`: no objects for empty cells.\n * - `bookDeps`, `bookFiles`, `bookProps`, `bookSheets`, `bookVBA: false`: no calculation chain,\n * raw archive, or VBA blob, and a full parse (not the properties-only or names-only modes).\n * - `dense: false`: sheets keyed by address, as `xlsx-text.ts` reads them.\n * - `WTF: false`: per-sheet parse errors skip the sheet instead of throwing.\n *\n * A fresh copy is passed on every call because SheetJS writes defaults into the options object.\n */\nexport const sheetJsReadOptions = {\n type: 'array',\n cellFormula: false,\n cellHTML: false,\n cellText: true,\n cellNF: false,\n cellStyles: false,\n cellDates: false,\n sheetStubs: false,\n bookDeps: false,\n bookFiles: false,\n bookProps: false,\n bookSheets: false,\n bookVBA: false,\n dense: false,\n WTF: false\n} as const\n\n/** SemVer 2.0.0: `major.minor.patch`, optional `-prerelease`, optional `+build`. */\nconst semver =\n /^(0|[1-9]\\d*)\\.(0|[1-9]\\d*)\\.(0|[1-9]\\d*)(?:-((?:0|[1-9]\\d*|\\d*[a-zA-Z-][0-9a-zA-Z-]*)(?:\\.(?:0|[1-9]\\d*|\\d*[a-zA-Z-][0-9a-zA-Z-]*))*))?(?:\\+[0-9a-zA-Z-]+(?:\\.[0-9a-zA-Z-]+)*)?$/\n\ntype ParsedVersion = {\n readonly release: readonly [number, number, number]\n readonly prerelease: boolean\n}\n\nconst parseVersion = (version: string): ParsedVersion | undefined => {\n const match = semver.exec(version)\n\n if (match === null) return undefined\n\n const release = [Number(match[1]), Number(match[2]), Number(match[3])] as const\n\n return release.every(Number.isSafeInteger)\n ? { release, prerelease: match[4] !== undefined }\n : undefined\n}\n\n/**\n * Compare by SemVer precedence against the minimum release: a prerelease of 0.20.3 is below it,\n * prereleases of later releases are above it, and build metadata is ignored.\n */\nconst meetsMinimumVersion = (installed: ParsedVersion, minimum: ParsedVersion) => {\n for (const [index, part] of installed.release.entries()) {\n const required = minimum.release[index] ?? 0\n\n if (part !== required) return part > required\n }\n\n return !installed.prerelease\n}\n\nconst isMissingModule = (cause: unknown) =>\n Predicate.hasProperty(cause, 'code') &&\n (cause.code === 'ERR_MODULE_NOT_FOUND' || cause.code === 'MODULE_NOT_FOUND')\n\n/** ESM namespace, or a CommonJS interop namespace whose `default` is the module. */\nconst sheetJsExports = (namespace: unknown): object | undefined => {\n if (Predicate.hasProperty(namespace, 'read')) return namespace\n\n if (\n Predicate.hasProperty(namespace, 'default') &&\n Predicate.hasProperty(namespace.default, 'read')\n )\n return namespace.default\n\n return undefined\n}\n\n/**\n * Load SheetJS lazily and refuse anything that is not SheetJS 0.20.3 or newer. The version must be\n * strict SemVer; anything else fails closed as `invalid`.\n */\nexport const loadSheetJs = (loader: SheetJsLoader) =>\n Effect.gen(function* () {\n const namespace = yield* Effect.tryPromise({\n try: loader,\n catch: cause =>\n new SheetJsUnavailableError({\n reason: isMissingModule(cause) ? 'missing' : 'invalid',\n cause\n })\n })\n\n const exports = sheetJsExports(namespace)\n\n if (\n exports === undefined ||\n !Predicate.hasProperty(exports, 'read') ||\n !Predicate.isFunction(exports.read) ||\n !Predicate.hasProperty(exports, 'version') ||\n !Predicate.isString(exports.version)\n )\n return yield* Effect.fail(new SheetJsUnavailableError({ reason: 'invalid' }))\n\n const { read, version } = exports\n const installed = parseVersion(version)\n const minimum = parseVersion(minimumSheetJsVersion)\n\n // A version that is not strict SemVer is not a SheetJS release we can vouch for.\n if (installed === undefined || minimum === undefined)\n return yield* Effect.fail(\n new SheetJsUnavailableError({ reason: 'invalid', installedVersion: version })\n )\n\n if (!meetsMinimumVersion(installed, minimum))\n return yield* Effect.fail(\n new SheetJsUnavailableError({ reason: 'outdated', installedVersion: version })\n )\n\n const sheetJs: SheetJs = {\n version,\n read: bytes => read(bytes, { ...sheetJsReadOptions })\n }\n\n return sheetJs\n })\n\n/** Check the parsed workbook shape before reading it. */\nexport const asXlsxWorkbook = (parsed: unknown): XlsxWorkbook | undefined => {\n if (\n !Predicate.hasProperty(parsed, 'SheetNames') ||\n !Predicate.hasProperty(parsed, 'Sheets') ||\n !Array.isArray(parsed.SheetNames) ||\n !Predicate.isObject(parsed.Sheets)\n )\n return undefined\n\n const sheetNames: Array<string> = []\n\n for (const name of parsed.SheetNames) {\n if (!Predicate.isString(name)) return undefined\n\n sheetNames.push(name)\n }\n\n return { SheetNames: sheetNames, Sheets: parsed.Sheets }\n}\n"],"mappings":";;;AAOA,MAAa,6BAA4C,OAAO;;;;;;;;;;;;;;;;;;;;;;;;AA+BhE,MAAa,qBAAqB;CAChC,MAAM;CACN,aAAa;CACb,UAAU;CACV,UAAU;CACV,QAAQ;CACR,YAAY;CACZ,WAAW;CACX,YAAY;CACZ,UAAU;CACV,WAAW;CACX,WAAW;CACX,YAAY;CACZ,SAAS;CACT,OAAO;CACP,KAAK;AACP;;AAGA,MAAM,SACJ;AAOF,MAAM,gBAAgB,YAA+C;CACnE,MAAM,QAAQ,OAAO,KAAK,OAAO;CAEjC,IAAI,UAAU,MAAM,OAAO,KAAA;CAE3B,MAAM,UAAU;EAAC,OAAO,MAAM,EAAE;EAAG,OAAO,MAAM,EAAE;EAAG,OAAO,MAAM,EAAE;CAAC;CAErE,OAAO,QAAQ,MAAM,OAAO,aAAa,IACrC;EAAE;EAAS,YAAY,MAAM,OAAO,KAAA;CAAU,IAC9C,KAAA;AACN;;;;;AAMA,MAAM,uBAAuB,WAA0B,YAA2B;CAChF,KAAK,MAAM,CAAC,OAAO,SAAS,UAAU,QAAQ,QAAQ,GAAG;EACvD,MAAM,WAAW,QAAQ,QAAQ,UAAU;EAE3C,IAAI,SAAS,UAAU,OAAO,OAAO;CACvC;CAEA,OAAO,CAAC,UAAU;AACpB;AAEA,MAAM,mBAAmB,UACvB,UAAU,YAAY,OAAO,MAAM,MAClC,MAAM,SAAS,0BAA0B,MAAM,SAAS;;AAG3D,MAAM,kBAAkB,cAA2C;CACjE,IAAI,UAAU,YAAY,WAAW,MAAM,GAAG,OAAO;CAErD,IACE,UAAU,YAAY,WAAW,SAAS,KAC1C,UAAU,YAAY,UAAU,SAAS,MAAM,GAE/C,OAAO,UAAU;AAGrB;;;;;AAMA,MAAa,eAAe,WAC1B,OAAO,IAAI,aAAa;CAUtB,MAAM,UAAU,eAAe,OATN,OAAO,WAAW;EACzC,KAAK;EACL,QAAO,UACL,IAAI,wBAAwB;GAC1B,QAAQ,gBAAgB,KAAK,IAAI,YAAY;GAC7C;EACF,CAAC;CACL,CAAC,CAEuC;CAExC,IACE,YAAY,KAAA,KACZ,CAAC,UAAU,YAAY,SAAS,MAAM,KACtC,CAAC,UAAU,WAAW,QAAQ,IAAI,KAClC,CAAC,UAAU,YAAY,SAAS,SAAS,KACzC,CAAC,UAAU,SAAS,QAAQ,OAAO,GAEnC,OAAO,OAAO,OAAO,KAAK,IAAI,wBAAwB,EAAE,QAAQ,UAAU,CAAC,CAAC;CAE9E,MAAM,EAAE,MAAM,YAAY;CAC1B,MAAM,YAAY,aAAa,OAAO;CACtC,MAAM,UAAU,aAAa,qBAAqB;CAGlD,IAAI,cAAc,KAAA,KAAa,YAAY,KAAA,GACzC,OAAO,OAAO,OAAO,KACnB,IAAI,wBAAwB;EAAE,QAAQ;EAAW,kBAAkB;CAAQ,CAAC,CAC9E;CAEF,IAAI,CAAC,oBAAoB,WAAW,OAAO,GACzC,OAAO,OAAO,OAAO,KACnB,IAAI,wBAAwB;EAAE,QAAQ;EAAY,kBAAkB;CAAQ,CAAC,CAC/E;CAOF,OAAO;EAJL;EACA,OAAM,UAAS,KAAK,OAAO,EAAE,GAAG,mBAAmB,CAAC;CAGzC;AACf,CAAC;;AAGH,MAAa,kBAAkB,WAA8C;CAC3E,IACE,CAAC,UAAU,YAAY,QAAQ,YAAY,KAC3C,CAAC,UAAU,YAAY,QAAQ,QAAQ,KACvC,CAAC,MAAM,QAAQ,OAAO,UAAU,KAChC,CAAC,UAAU,SAAS,OAAO,MAAM,GAEjC,OAAO,KAAA;CAET,MAAM,aAA4B,CAAC;CAEnC,KAAK,MAAM,QAAQ,OAAO,YAAY;EACpC,IAAI,CAAC,UAAU,SAAS,IAAI,GAAG,OAAO,KAAA;EAEtC,WAAW,KAAK,IAAI;CACtB;CAEA,OAAO;EAAE,YAAY;EAAY,QAAQ,OAAO;CAAO;AACzD"}
@@ -0,0 +1,58 @@
1
+ import { Effect } from "effect";
2
+
3
+ //#region src/node/worker-admission.d.ts
4
+ /**
5
+ * Worker admission. Every Node `FileExtractor` layer in a JavaScript realm (the main thread, or
6
+ * each worker thread or `vm` context that builds the layer) shares one pool of
7
+ * `processWorkerLimit` worker slots, however often the layer is built (a per-request
8
+ * `Effect.provide` builds a new one each time). The pool lives on `globalThis` under a versioned
9
+ * `Symbol.for` key and holds only plain data and plain callbacks, so duplicated copies of this
10
+ * package (and of Effect) in one realm share it. Each layer also has its own pool of
11
+ * `maxConcurrentWorkers` slots (at most `processWorkerLimit`), which can only lower that layer's
12
+ * share. An extraction takes its layer's slot, then a realm slot, and waits for both until its
13
+ * admission deadline; a slot is released only after its worker has terminated.
14
+ *
15
+ * Admission is first come, first served: a freed slot is handed to the longest-waiting extraction
16
+ * whose deadline has not passed, and a new extraction takes a free slot only when nobody waits.
17
+ */
18
+ /** Worker slots shared by every layer in the realm. */
19
+ declare const processWorkerLimit = 4;
20
+ /**
21
+ * Takes a freed slot for its waiter and returns `true`, or returns `false` when the waiter's
22
+ * deadline has passed (it then fails with `busy`). Either way it leaves the queue. It never runs
23
+ * the waiter's fiber: the resume is deferred to a microtask, so a waiter that finishes at once
24
+ * cannot release (and hand off) again inside this call, and a long queue drains in constant stack.
25
+ */
26
+ type SlotHandOff = () => boolean;
27
+ /**
28
+ * A counting pool of worker slots with a FIFO queue (a `Set` iterates in insertion order). While
29
+ * anyone waits, every slot is taken: a released slot passes to a waiter without being counted
30
+ * free.
31
+ */
32
+ type SlotPool = {
33
+ readonly capacity: number;
34
+ active: number;
35
+ readonly waiters: Set<SlotHandOff>;
36
+ };
37
+ declare const makeSlotPool: (capacity: number) => SlotPool;
38
+ /** The realm-wide pool, created by the first copy of this module that asks for it. */
39
+ declare const processSlotPool: () => SlotPool;
40
+ /** Running and waiting extractions in the realm-wide pool. */
41
+ declare const processAdmissionSnapshot: () => {
42
+ active: number;
43
+ waiting: number;
44
+ };
45
+ /**
46
+ * Run `self` holding one slot of `pool`, waiting for it until `deadline` (epoch ms) and failing
47
+ * with `busy()` after that. The deadline is checked again whenever a slot would be taken, by a new
48
+ * extraction or by a hand-off, so a late wake-up cannot turn an expired wait into admission.
49
+ *
50
+ * Taking a slot and registering its release happen without an interruption point between them,
51
+ * so a slot is never leaked; only the wait is interruptible. A slot handed to a waiter whose
52
+ * fiber is interrupted before it resumes is released again. The wait uses a real timer, not the
53
+ * Effect `Clock`, so a test clock cannot stall it.
54
+ */
55
+ declare const withSlot: <E2>(pool: SlotPool, deadline: number, busy: () => E2) => <A, E, R>(self: Effect.Effect<A, E, R>) => Effect.Effect<A, E | E2, R>;
56
+ //#endregion
57
+ export { SlotHandOff, SlotPool, makeSlotPool, processAdmissionSnapshot, processSlotPool, processWorkerLimit, withSlot };
58
+ //# sourceMappingURL=worker-admission.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"worker-admission.d.mts","names":[],"sources":["../../src/node/worker-admission.ts"],"mappings":";;;;;AAmBA;;;;AAA+B;AAQ/B;;;;AAAuB;AAOvB;;;cAfa,kBAAA;;;;;;;KAQD,WAAA;AAaZ;;;;AAIE;AAJF,KANY,QAAA;EAAA,SACD,QAAA;EACT,MAAA;EAAA,SACS,OAAA,EAAS,GAAG,CAAC,WAAA;AAAA;AAAA,cAGX,YAAA,GAAgB,QAAA,aAAmB,QAI9C;;cAcW,eAAA,QAAsB,QAUlC;;cAGY,wBAAA;EAIZ,MAAA;EAAA,OAAA;AAAA;;;;;;;;;;;cAsBY,QAAA,OACN,IAAA,EAAM,QAAA,EAAU,QAAA,UAAkB,IAAA,QAAY,EAAA,eACzC,IAAA,EAAM,MAAA,CAAO,MAAA,CAAO,CAAA,EAAG,CAAA,EAAG,CAAA,MAAK,MAAA,CAAO,MAAA,CAAO,CAAA,EAAG,CAAA,GAAI,EAAA,EAAI,CAAA"}