@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
@@ -0,0 +1,275 @@
1
+ import { strToU8, zipSync } from 'fflate'
2
+ import { FileExtractionError } from '../errors.ts'
3
+ import {
4
+ coreTitle,
5
+ officeDocumentRelationshipsNamespace,
6
+ sheetJsCouldReadCdata,
7
+ sheetJsXmlHeader
8
+ } from './sheetjs-xml.ts'
9
+ import { readStyles, stylesXml } from './xlsx-styles.ts'
10
+ import { readWorkbookModel, workbookXml } from './xlsx-workbook.ts'
11
+ import {
12
+ directoryOf,
13
+ indexParts,
14
+ relationships,
15
+ relationshipsPathFor,
16
+ resolvePartPath,
17
+ utf8Text,
18
+ workbookPartPath
19
+ } from './xlsx-parts.ts'
20
+ import type { PartIndex, Relationship } from './xlsx-parts.ts'
21
+
22
+ /**
23
+ * The archive SheetJS reads is built here from scratch, never passed through. Every part SheetJS
24
+ * reads is generated or checked:
25
+ *
26
+ * - generated: `[Content_Types].xml`, `_rels/.rels`, `xl/_rels/workbook.xml.rels`,
27
+ * `xl/workbook.xml` (`xlsx-workbook.ts`), and `xl/styles.xml` (`xlsx-styles.ts`);
28
+ * - copied after checks: worksheets (stored as `xl/worksheets/sheet<n>.xml`) and
29
+ * `xl/sharedStrings.xml`, validated, hyperlink-stripped, and free of anything SheetJS could
30
+ * turn into a CDATA marker (`sheetJsCouldReadCdata`).
31
+ *
32
+ * With these entries SheetJS 0.20.3 `parse_zip` can only take its XLSX path:
33
+ *
34
+ * - no `META-INF/manifest.xml`, `objectdata.xml`, or `Index/Document.iwa`, so it never reaches
35
+ * `parse_ods` or `parse_numbers_iwa`; `[Content_Types].xml` exists, so `Index.zip` is never read;
36
+ * - every entry ends in `.xml` or `.rels`, so no binary (XLSB) parser can receive data: SheetJS
37
+ * dispatches on the requested path ending in `.bin`, and `safegetzipfile` only returns an entry
38
+ * whose name equals that path ignoring case;
39
+ * - the generated content types name the XML workbook, so `xlsb` stays false, and no attacker
40
+ * `Override`, `PartName`, relationship `Type`, or `Target` reaches SheetJS;
41
+ * - the generated workbook lists exactly the sheets the extractor counted, each with its own
42
+ * relationship to its own `xl/worksheets/sheet<n>.xml`, so every part is parsed at most once.
43
+ *
44
+ * Comments, threaded comments, VML, drawings, worksheet relationships, external links, pivot
45
+ * caches, calculation chains, metadata, themes, `customXml`, and `docProps/*` (the title is read
46
+ * by the extractor) are left out; SheetJS reads none of them to produce cell values or display
47
+ * text.
48
+ */
49
+
50
+ const strictOfficeDocumentRelationships = 'http://purl.oclc.org/ooxml/officeDocument/relationships'
51
+
52
+ const corePropertiesTypes = new Set([
53
+ 'http://schemas.openxmlformats.org/package/2006/relationships/metadata/core-properties',
54
+ 'http://schemas.openxmlformats.org/officedocument/2006/relationships/metadata/core-properties'
55
+ ])
56
+
57
+ /** A relationship of `kind` in the transitional or strict namespace. */
58
+ const hasKind = (relationship: Relationship, kind: string) =>
59
+ relationship.type === `${officeDocumentRelationshipsNamespace}/${kind}` ||
60
+ relationship.type === `${strictOfficeDocumentRelationships}/${kind}`
61
+
62
+ /** Canonical names SheetJS can receive besides `xl/worksheets/sheet<n>.xml`. */
63
+ export const sheetJsFixedParts = [
64
+ '[Content_Types].xml',
65
+ '_rels/.rels',
66
+ 'xl/workbook.xml',
67
+ 'xl/_rels/workbook.xml.rels',
68
+ 'xl/sharedStrings.xml',
69
+ 'xl/styles.xml'
70
+ ] as const
71
+
72
+ const canonicalWorksheet = /^xl\/worksheets\/sheet[1-9]\d*\.xml$/
73
+
74
+ const fixedPartNames: ReadonlySet<string> = new Set(sheetJsFixedParts)
75
+
76
+ /** Whether `name` is one of the canonical names `buildSheetJsInput` emits. */
77
+ export const isSheetJsInputName = (name: string) =>
78
+ fixedPartNames.has(name) || canonicalWorksheet.test(name)
79
+
80
+ const contentType = {
81
+ workbook: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml',
82
+ worksheet: 'application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml',
83
+ sharedStrings: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sharedStrings+xml',
84
+ styles: 'application/vnd.openxmlformats-officedocument.spreadsheetml.styles+xml'
85
+ } as const
86
+
87
+ const contentTypesXml = (overrides: ReadonlyArray<readonly [string, string]>) =>
88
+ `${sheetJsXmlHeader}<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"><Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/><Default Extension="xml" ContentType="application/xml"/>${overrides
89
+ .map(([path, type]) => `<Override PartName="/${path}" ContentType="${type}"/>`)
90
+ .join('')}</Types>`
91
+
92
+ const relationshipsXml = (entries: ReadonlyArray<readonly [string, string, string]>) =>
93
+ `${sheetJsXmlHeader}<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">${entries
94
+ .map(([id, type, target]) => `<Relationship Id="${id}" Type="${type}" Target="${target}"/>`)
95
+ .join('')}</Relationships>`
96
+
97
+ /** An XML part found by name ignoring case; `.xml` only, so no binary bytes are renamed. */
98
+ const xmlPart = (index: PartIndex, path: string | undefined) => {
99
+ const part = path === undefined ? undefined : index.find(path)
100
+
101
+ return part !== undefined && part.name.toLowerCase().endsWith('.xml') ? part : undefined
102
+ }
103
+
104
+ /** The part a relationship of `kind` points at, else the conventional path. */
105
+ const relatedXmlPart = (
106
+ index: PartIndex,
107
+ related: ReadonlyMap<string, Relationship>,
108
+ baseDirectory: string,
109
+ matches: (relationship: Relationship) => boolean,
110
+ conventionalPath: string
111
+ ) => {
112
+ for (const relationship of related.values()) {
113
+ if (!relationship.external && matches(relationship))
114
+ return xmlPart(index, resolvePartPath(baseDirectory, relationship.target))
115
+ }
116
+
117
+ return xmlPart(index, conventionalPath)
118
+ }
119
+
120
+ const malformed = () =>
121
+ new FileExtractionError({ format: 'xlsx', message: 'XLSX workbook is malformed.' })
122
+
123
+ const cdataRejected = () =>
124
+ new FileExtractionError({
125
+ format: 'xlsx',
126
+ message:
127
+ 'XLSX contains unsupported markup (CDATA, comments or declarations) in worksheet or shared-strings parts.'
128
+ })
129
+
130
+ /** A copied part, unless SheetJS could meet a CDATA marker in it (decoded or tag-removed). */
131
+ const withoutCdata = (bytes: Uint8Array) => {
132
+ if (sheetJsCouldReadCdata(bytes)) throw cdataRejected()
133
+
134
+ return bytes
135
+ }
136
+
137
+ export type SheetJsInputSheet = {
138
+ /** The sheet name exactly as SheetJS reads it from the generated workbook. */
139
+ readonly name: string
140
+ /** The validated source part handed over as this sheet's worksheet, if any. */
141
+ readonly part: string | undefined
142
+ }
143
+
144
+ export type SheetJsInput = {
145
+ /** The stored-entry ZIP handed to SheetJS. */
146
+ readonly archive: Uint8Array
147
+ /** Entry names in `archive`, all canonical (see `isSheetJsInputName`). */
148
+ readonly names: ReadonlyArray<string>
149
+ /** Every sheet of the generated workbook, in order. */
150
+ readonly sheets: ReadonlyArray<SheetJsInputSheet>
151
+ /** The workbook title from its core properties, read without SheetJS. */
152
+ readonly title: string | undefined
153
+ }
154
+
155
+ /**
156
+ * Build SheetJS's input from validated parts. Sheet `n` of the workbook gets relationship
157
+ * `rId<n>` to `xl/worksheets/sheet<n>.xml`. That entry holds the sheet's worksheet when its
158
+ * workbook relationship is an internal worksheet resolving to an `.xml` part; other sheets (chart
159
+ * sheets, missing parts) keep their name and order but have no entry, so SheetJS skips them.
160
+ *
161
+ * Throws `FileExtractionError` when the workbook declares more than `maxSheets` sheets, when it
162
+ * is malformed (see `readWorkbookModel`; two sheets on one worksheet part count too), or when a
163
+ * copied part holds CDATA, all before SheetJS runs.
164
+ */
165
+ export const buildSheetJsInput = (
166
+ parts: Readonly<Record<string, Uint8Array>>,
167
+ maxSheets: number
168
+ ): SheetJsInput => {
169
+ const index = indexParts(parts)
170
+ const workbook = index.find(workbookPartPath)
171
+
172
+ if (workbook === undefined)
173
+ throw new FileExtractionError({ format: 'xlsx', message: 'Invalid Office archive.' })
174
+
175
+ const workbookDirectory = directoryOf(workbookPartPath)
176
+
177
+ const workbookRelationships = relationships(
178
+ utf8Text(index.find(relationshipsPathFor(workbookPartPath))?.bytes)
179
+ )
180
+
181
+ const model = readWorkbookModel(workbook.bytes, maxSheets)
182
+ const files: Record<string, Uint8Array> = Object.create(null)
183
+ const overrides: Array<readonly [string, string]> = []
184
+ const sheetRelationships: Array<readonly [string, string, string]> = []
185
+ const sheets: Array<SheetJsInputSheet> = []
186
+ const included = new Set<string>()
187
+
188
+ for (const [position, sheet] of model.sheets.entries()) {
189
+ const name = `worksheets/sheet${position + 1}.xml`
190
+
191
+ sheetRelationships.push([
192
+ `rId${position + 1}`,
193
+ `${officeDocumentRelationshipsNamespace}/worksheet`,
194
+ name
195
+ ])
196
+
197
+ const relationship =
198
+ sheet.relationshipId === undefined
199
+ ? undefined
200
+ : workbookRelationships.get(sheet.relationshipId)
201
+
202
+ const part =
203
+ relationship === undefined || relationship.external || !hasKind(relationship, 'worksheet')
204
+ ? undefined
205
+ : xmlPart(index, resolvePartPath(workbookDirectory, relationship.target))
206
+
207
+ if (part === undefined) {
208
+ sheets.push({ name: sheet.name, part: undefined })
209
+ continue
210
+ }
211
+
212
+ // One part per sheet: SheetJS would parse and keep a shared part once per declaration.
213
+ if (included.has(part.name)) throw malformed()
214
+
215
+ included.add(part.name)
216
+ files[`xl/${name}`] = withoutCdata(part.bytes)
217
+ overrides.push([`xl/${name}`, contentType.worksheet])
218
+ sheets.push({ name: sheet.name, part: part.name })
219
+ }
220
+
221
+ const sharedStrings = relatedXmlPart(
222
+ index,
223
+ workbookRelationships,
224
+ workbookDirectory,
225
+ relationship => hasKind(relationship, 'sharedStrings'),
226
+ 'xl/sharedStrings.xml'
227
+ )
228
+
229
+ if (sharedStrings !== undefined) {
230
+ files['xl/sharedStrings.xml'] = withoutCdata(sharedStrings.bytes)
231
+ overrides.push(['xl/sharedStrings.xml', contentType.sharedStrings])
232
+ }
233
+
234
+ const styles = relatedXmlPart(
235
+ index,
236
+ workbookRelationships,
237
+ workbookDirectory,
238
+ relationship => hasKind(relationship, 'styles'),
239
+ 'xl/styles.xml'
240
+ )
241
+
242
+ if (styles !== undefined) {
243
+ files['xl/styles.xml'] = strToU8(stylesXml(readStyles(styles.bytes)))
244
+ overrides.push(['xl/styles.xml', contentType.styles])
245
+ }
246
+
247
+ const core = relatedXmlPart(
248
+ index,
249
+ relationships(utf8Text(index.find('_rels/.rels')?.bytes)),
250
+ '',
251
+ relationship => relationship.type !== undefined && corePropertiesTypes.has(relationship.type),
252
+ 'docProps/core.xml'
253
+ )
254
+
255
+ const archive = {
256
+ '[Content_Types].xml': strToU8(
257
+ contentTypesXml([[workbookPartPath, contentType.workbook], ...overrides])
258
+ ),
259
+ '_rels/.rels': strToU8(
260
+ relationshipsXml([
261
+ ['rId1', `${officeDocumentRelationshipsNamespace}/officeDocument`, workbookPartPath]
262
+ ])
263
+ ),
264
+ [workbookPartPath]: strToU8(workbookXml(model)),
265
+ 'xl/_rels/workbook.xml.rels': strToU8(relationshipsXml(sheetRelationships)),
266
+ ...files
267
+ }
268
+
269
+ return {
270
+ archive: zipSync(archive, { level: 0 }),
271
+ names: Object.keys(archive),
272
+ sheets,
273
+ title: coreTitle(core?.bytes)
274
+ }
275
+ }
@@ -0,0 +1,160 @@
1
+ import { Buffer } from 'node:buffer'
2
+ import {
3
+ sheetJsAttributeEscape,
4
+ sheetJsAttributeText,
5
+ sheetJsTags,
6
+ sheetJsXmlHeader,
7
+ spreadsheetMainNamespace,
8
+ stripSheetJsNamespace
9
+ } from './sheetjs-xml.ts'
10
+
11
+ /**
12
+ * `xl/styles.xml` is never handed to SheetJS as uploaded. With `cellText: true`, SheetJS formats
13
+ * every styled cell by re-parsing its number format (`SSF_format`, work proportional to the
14
+ * format's length, a quoted literal built one character at a time), so one huge format reused by
15
+ * many cells amplifies before any budget runs. The extractor writes a minimal stylesheet with
16
+ * only what display text needs: custom number formats (`numFmtId`, `formatCode`) of at most
17
+ * `maxNumberFormatCharacters`, and one `<xf numFmtId>` per source cell format, in order, so cell
18
+ * `s` indexes keep their meaning. Fonts, fills, borders, cell styles, and dxfs are left out:
19
+ * SheetJS parses fonts, fills, and borders whatever the options but uses them only with
20
+ * `cellStyles`, so leaving them out also removes their parsers from the input.
21
+ */
22
+
23
+ /** Excel's own limit on a number format code, counted after unescaping. */
24
+ export const maxNumberFormatCharacters = 255
25
+
26
+ /** Custom number formats kept; later ones are dropped (their cells show General). */
27
+ export const maxCustomNumberFormats = 1000
28
+
29
+ /** Excel's limit on cell formats (`cellXfs`); later ones are dropped (General). */
30
+ export const maxCellFormats = 64_000
31
+
32
+ /** SheetJS `str_remove_ng(text, '<!--', '-->')`, including its unterminated-comment behavior. */
33
+ const removeComments = (text: string) => {
34
+ let start = text.indexOf('<!--')
35
+
36
+ if (start === -1) return text
37
+
38
+ const output: Array<string> = []
39
+ let last = 0
40
+
41
+ while (start > -1) {
42
+ output.push(text.slice(last, start))
43
+
44
+ const end = text.indexOf('-->', start + 4)
45
+
46
+ if (end === -1) break
47
+
48
+ last = end + 3
49
+ start = text.indexOf('<!--', last)
50
+
51
+ if (start === -1) output.push(text.slice(last))
52
+ }
53
+
54
+ return output.join('')
55
+ }
56
+
57
+ /** SheetJS `remove_doctype`. */
58
+ const removeDoctype = (text: string) => {
59
+ const doctype = text.slice(0, 1024).indexOf('<!DOCTYPE')
60
+
61
+ if (doctype === -1) return text
62
+
63
+ const element = /<\w/.exec(text)
64
+
65
+ return element === null ? text : text.slice(0, doctype) + text.slice(element.index)
66
+ }
67
+
68
+ /** SheetJS `str_match_xml_ns`: the first `<tag>` start tag through the next `</tag>` (any prefix). */
69
+ const sheetJsRegion = (text: string, tag: string) => {
70
+ const start = new RegExp(`<(?:\\w+:)?${tag}\\b[^<>]*>`, 'g')
71
+ const end = new RegExp(`</(?:\\w+:)?${tag}>`, 'g')
72
+ const open = start.exec(text)
73
+
74
+ if (open === null) return undefined
75
+
76
+ end.lastIndex = start.lastIndex
77
+
78
+ return end.exec(text) === null ? undefined : text.slice(open.index, end.lastIndex)
79
+ }
80
+
81
+ const numberFormatId = (raw: string | undefined) => {
82
+ const id = Number.parseInt(raw ?? '', 10)
83
+
84
+ return Number.isSafeInteger(id) && id >= 0 ? id : undefined
85
+ }
86
+
87
+ export type StylesSummary = {
88
+ /** Custom number formats in document order, as `[numFmtId, formatCode]`. */
89
+ readonly numberFormats: ReadonlyArray<readonly [number, string]>
90
+ /** The `numFmtId` of each cell format in order, or `undefined` without a `cellXfs` element. */
91
+ readonly cellFormats: ReadonlyArray<number> | undefined
92
+ }
93
+
94
+ /**
95
+ * Read the number formats and cell formats SheetJS would read from a stylesheet (same comment and
96
+ * doctype removal, same regions, same tag grammar), bounded by the caps above. A format longer
97
+ * than `maxNumberFormatCharacters` after unescaping, holding CDATA, or with an invalid id is
98
+ * dropped; a cell format without a valid `numFmtId` becomes General (0).
99
+ */
100
+ export const readStyles = (content: Uint8Array): StylesSummary => {
101
+ const text = removeDoctype(
102
+ removeComments(
103
+ Buffer.from(content.buffer, content.byteOffset, content.byteLength).toString('latin1')
104
+ )
105
+ )
106
+
107
+ const numberFormats: Array<readonly [number, string]> = []
108
+ const formatsRegion = sheetJsRegion(text, 'numFmts')
109
+
110
+ if (formatsRegion !== undefined) {
111
+ for (const tag of sheetJsTags(formatsRegion)) {
112
+ if (numberFormats.length >= maxCustomNumberFormats) break
113
+
114
+ if (stripSheetJsNamespace(tag.head) !== '<numFmt') continue
115
+
116
+ const id = numberFormatId(tag.attributes.get('numFmtId'))
117
+ const raw = tag.attributes.get('formatCode')
118
+ const code = raw === undefined ? undefined : sheetJsAttributeText(raw)
119
+
120
+ if (id !== undefined && code !== undefined && code.length <= maxNumberFormatCharacters)
121
+ numberFormats.push([id, code])
122
+ }
123
+ }
124
+
125
+ const cellFormatsRegion = sheetJsRegion(text, 'cellXfs')
126
+
127
+ if (cellFormatsRegion === undefined) return { numberFormats, cellFormats: undefined }
128
+
129
+ const cellFormats: Array<number> = []
130
+
131
+ for (const tag of sheetJsTags(cellFormatsRegion)) {
132
+ if (cellFormats.length >= maxCellFormats) break
133
+
134
+ const head = stripSheetJsNamespace(tag.head)
135
+
136
+ if (head === '<xf' || head === '<xf/>' || head === '<xf>')
137
+ cellFormats.push(numberFormatId(tag.attributes.get('numFmtId')) ?? 0)
138
+ }
139
+
140
+ return { numberFormats, cellFormats }
141
+ }
142
+
143
+ /** The generated stylesheet SheetJS reads instead of the uploaded one. */
144
+ export const stylesXml = ({ numberFormats, cellFormats }: StylesSummary) =>
145
+ `${sheetJsXmlHeader}<styleSheet xmlns="${spreadsheetMainNamespace}">${
146
+ numberFormats.length === 0
147
+ ? ''
148
+ : `<numFmts count="${numberFormats.length}">${numberFormats
149
+ .map(
150
+ ([id, code]) =>
151
+ `<numFmt numFmtId="${id}" formatCode="${sheetJsAttributeEscape(code)}"/>`
152
+ )
153
+ .join('')}</numFmts>`
154
+ }${
155
+ cellFormats === undefined
156
+ ? ''
157
+ : `<cellXfs count="${cellFormats.length}">${cellFormats
158
+ .map(id => `<xf numFmtId="${id}"/>`)
159
+ .join('')}</cellXfs>`
160
+ }</styleSheet>`