@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
@@ -0,0 +1,260 @@
1
+ import { Buffer } from 'node:buffer'
2
+ import type { StrippedHyperlinkTags } from './office-archive.ts'
3
+ import { indexParts, relationships, relationshipsPathFor, utf8Text } from './xlsx-parts.ts'
4
+ import type { Relationship } from './xlsx-parts.ts'
5
+ import { parseCellRange } from './xlsx-range.ts'
6
+ import type { CellRange } from './xlsx-range.ts'
7
+ import { prefixedAttribute, xmlAttributes } from './xml-text.ts'
8
+
9
+ export type XlsxHyperlink = {
10
+ readonly range: CellRange
11
+ /** Normalized `http:`, `https:`, or `mailto:` URL, at most 2,048 characters. */
12
+ readonly target: string
13
+ /** Label for cells without text, at most 1,024 characters (longer labels end in `…`). */
14
+ readonly display?: string
15
+ }
16
+
17
+ /** Hyperlinks per worksheet name, in document order (a later link wins on overlap). */
18
+ export type XlsxHyperlinks = ReadonlyMap<string, ReadonlyArray<XlsxHyperlink>>
19
+
20
+ /** Longer targets are dropped: a truncated URL would point somewhere else. */
21
+ const maxHyperlinkTargetCharacters = 2048
22
+
23
+ /** Longer display labels are cut to this many characters, ending with an ellipsis. */
24
+ const maxHyperlinkDisplayCharacters = 1024
25
+
26
+ const capDisplay = (display: string) => {
27
+ if (display.length <= maxHyperlinkDisplayCharacters) return display
28
+
29
+ let kept = display.slice(0, maxHyperlinkDisplayCharacters - 1)
30
+ const last = kept.charCodeAt(kept.length - 1)
31
+
32
+ // Never leave half of a surrogate pair before the ellipsis.
33
+ if (last >= 0xd800 && last <= 0xdbff) kept = kept.slice(0, -1)
34
+
35
+ return `${kept}\u2026`
36
+ }
37
+
38
+ const shownProtocols = new Set(['http:', 'https:', 'mailto:'])
39
+
40
+ const shownTarget = (target: string) => {
41
+ if (target.length > maxHyperlinkTargetCharacters || !URL.canParse(target)) return undefined
42
+
43
+ const url = new URL(target)
44
+
45
+ if (!shownProtocols.has(url.protocol) || url.href.length > maxHyperlinkTargetCharacters)
46
+ return undefined
47
+
48
+ return url.href
49
+ }
50
+
51
+ const hyperlinkFrom = (
52
+ tag: string,
53
+ sheetRelationships: ReadonlyMap<string, Relationship>
54
+ ): XlsxHyperlink | undefined => {
55
+ // Tags were captured as Latin-1 text of the original bytes; attribute values are UTF-8.
56
+ const attributes = xmlAttributes(Buffer.from(tag, 'latin1').toString('utf8'))
57
+ const ref = attributes.get('ref')
58
+ const range = ref === undefined ? undefined : parseCellRange(ref)
59
+ const id = prefixedAttribute(attributes, 'id')
60
+ const relationship = id === undefined ? undefined : sheetRelationships.get(id)
61
+
62
+ // Internal `location`-only links (`#Sheet2!A1`) are omitted; external links keep their
63
+ // location as a fragment, as SheetJS does.
64
+ if (range === undefined || relationship === undefined || !relationship.external) return undefined
65
+
66
+ const location = attributes.get('location')
67
+
68
+ const target = shownTarget(
69
+ location === undefined || location.length === 0
70
+ ? relationship.target
71
+ : `${relationship.target}#${location}`
72
+ )
73
+
74
+ if (target === undefined) return undefined
75
+
76
+ const display = attributes.get('display')?.trim()
77
+
78
+ return display === undefined || display.length === 0
79
+ ? { range, target }
80
+ : { range, target, display: capDisplay(display) }
81
+ }
82
+
83
+ /** A sheet of the workbook and the validated worksheet part SheetJS reads for it, if any. */
84
+ export type HyperlinkSheet = {
85
+ readonly name: string
86
+ readonly part: string | undefined
87
+ }
88
+
89
+ /**
90
+ * Map the hyperlink tags removed from a workbook to its sheets (the same sheet list and worksheet
91
+ * parts SheetJS reads, from `buildSheetJsInput`) through each worksheet's relationships.
92
+ */
93
+ export const resolveXlsxHyperlinks = (
94
+ parts: Readonly<Record<string, Uint8Array>>,
95
+ hyperlinkTags: StrippedHyperlinkTags,
96
+ sheets: ReadonlyArray<HyperlinkSheet>
97
+ ): XlsxHyperlinks => {
98
+ const bySheet = new Map<string, ReadonlyArray<XlsxHyperlink>>()
99
+
100
+ if (hyperlinkTags.size === 0) return bySheet
101
+
102
+ const index = indexParts(parts)
103
+ const text = (path: string) => utf8Text(index.find(path)?.bytes)
104
+
105
+ for (const { name, part } of sheets) {
106
+ const tags = part === undefined ? undefined : hyperlinkTags.get(part)
107
+
108
+ if (part === undefined || tags === undefined || bySheet.has(name)) continue
109
+
110
+ const sheetRelationships = relationships(text(relationshipsPathFor(part)))
111
+
112
+ const links = tags.flatMap(linkTag => {
113
+ const link = hyperlinkFrom(linkTag, sheetRelationships)
114
+
115
+ return link === undefined ? [] : [link]
116
+ })
117
+
118
+ if (links.length > 0) bySheet.set(name, links)
119
+ }
120
+
121
+ return bySheet
122
+ }
123
+
124
+ type ClippedLink = {
125
+ readonly order: number
126
+ readonly firstRow: number
127
+ readonly lastRow: number
128
+ readonly firstColumn: number
129
+ readonly lastColumn: number
130
+ readonly link: XlsxHyperlink
131
+ }
132
+
133
+ export type HyperlinkLookup = {
134
+ /** The winning link at a cell. Rows must be queried in ascending order. */
135
+ readonly at: (row: number, column: number) => XlsxHyperlink | undefined
136
+ }
137
+
138
+ /**
139
+ * Index links clipped to the visited range for row-major lookup without scanning every link per
140
+ * cell: a segment tree over columns whose nodes hold max-heaps (by document order) of the links
141
+ * covering them. Links enter when the sweep reaches their first row and are discarded lazily once
142
+ * it passes their last row. Work is O((links + cells) · log(columns) · log(links)).
143
+ */
144
+ export const makeHyperlinkLookup = (
145
+ links: ReadonlyArray<XlsxHyperlink>,
146
+ visited: CellRange
147
+ ): HyperlinkLookup => {
148
+ const clipped = links
149
+ .flatMap((link, order): ReadonlyArray<ClippedLink> => {
150
+ const firstRow = Math.max(link.range.start.r, visited.start.r)
151
+ const lastRow = Math.min(link.range.end.r, visited.end.r)
152
+ const firstColumn = Math.max(link.range.start.c, visited.start.c) - visited.start.c
153
+ const lastColumn = Math.min(link.range.end.c, visited.end.c) - visited.start.c
154
+
155
+ return firstRow > lastRow || firstColumn > lastColumn
156
+ ? []
157
+ : [{ order, firstRow, lastRow, firstColumn, lastColumn, link }]
158
+ })
159
+ .sort((left, right) => left.firstRow - right.firstRow || left.order - right.order)
160
+
161
+ if (clipped.length === 0) return { at: () => undefined }
162
+
163
+ const width = visited.end.c - visited.start.c + 1
164
+ let leaves = 1
165
+
166
+ while (leaves < width) leaves *= 2
167
+
168
+ const heaps: Array<Array<ClippedLink> | undefined> = []
169
+
170
+ const push = (node: number, entry: ClippedLink) => {
171
+ const heap = heaps[node] ?? []
172
+ heaps[node] = heap
173
+ heap.push(entry)
174
+
175
+ let index = heap.length - 1
176
+
177
+ while (index > 0) {
178
+ const parent = (index - 1) >> 1
179
+ const parentEntry = heap[parent]
180
+
181
+ if (parentEntry === undefined || parentEntry.order >= entry.order) break
182
+
183
+ heap[index] = parentEntry
184
+ heap[parent] = entry
185
+ index = parent
186
+ }
187
+ }
188
+
189
+ const pop = (heap: Array<ClippedLink>) => {
190
+ const last = heap.pop()
191
+
192
+ if (last === undefined || heap.length === 0) return
193
+
194
+ heap[0] = last
195
+
196
+ let index = 0
197
+
198
+ for (;;) {
199
+ const left = index * 2 + 1
200
+ const right = left + 1
201
+ let largest = index
202
+
203
+ const largestOrder = () => heap[largest]?.order ?? -1
204
+
205
+ if ((heap[left]?.order ?? -1) > largestOrder()) largest = left
206
+
207
+ if ((heap[right]?.order ?? -1) > largestOrder()) largest = right
208
+
209
+ const swapped = heap[largest]
210
+
211
+ if (largest === index || swapped === undefined) return
212
+
213
+ heap[largest] = last
214
+ heap[index] = swapped
215
+ index = largest
216
+ }
217
+ }
218
+
219
+ const insert = (entry: ClippedLink) => {
220
+ let low = entry.firstColumn + leaves
221
+ let high = entry.lastColumn + leaves + 1
222
+
223
+ while (low < high) {
224
+ if ((low & 1) === 1) push(low++, entry)
225
+
226
+ if ((high & 1) === 1) push(--high, entry)
227
+
228
+ low >>= 1
229
+ high >>= 1
230
+ }
231
+ }
232
+
233
+ let nextEntry = 0
234
+
235
+ return {
236
+ at: (row, column) => {
237
+ for (let entry = clipped[nextEntry]; entry !== undefined && entry.firstRow <= row;) {
238
+ insert(entry)
239
+ nextEntry += 1
240
+ entry = clipped[nextEntry]
241
+ }
242
+
243
+ let best: ClippedLink | undefined
244
+
245
+ for (let node = column - visited.start.c + leaves; node >= 1; node >>= 1) {
246
+ const heap = heaps[node]
247
+
248
+ if (heap === undefined) continue
249
+
250
+ while (heap[0] !== undefined && heap[0].lastRow < row) pop(heap)
251
+
252
+ const top = heap[0]
253
+
254
+ if (top !== undefined && (best === undefined || top.order > best.order)) best = top
255
+ }
256
+
257
+ return best?.link
258
+ }
259
+ }
260
+ }
@@ -0,0 +1,83 @@
1
+ import { Buffer } from 'node:buffer'
2
+ import { xmlAttributes } from './xml-text.ts'
3
+
4
+ /** The main workbook part every validated XLSX archive contains. */
5
+ export const workbookPartPath = 'xl/workbook.xml'
6
+
7
+ export const utf8Text = (bytes: Uint8Array | undefined) =>
8
+ bytes === undefined ? '' : Buffer.from(bytes).toString('utf8')
9
+
10
+ /** Resolve a relationship target against the directory of its source part (`xl/`, …). */
11
+ export const resolvePartPath = (baseDirectory: string, target: string) => {
12
+ const segments: Array<string> = []
13
+ const path = target.startsWith('/') ? target.slice(1) : `${baseDirectory}${target}`
14
+
15
+ for (const segment of path.split('/')) {
16
+ if (segment === '..') segments.pop()
17
+ else if (segment !== '.' && segment !== '') segments.push(segment)
18
+ }
19
+
20
+ return segments.join('/')
21
+ }
22
+
23
+ export const directoryOf = (path: string) => path.slice(0, path.lastIndexOf('/') + 1)
24
+
25
+ export const relationshipsPathFor = (partPath: string) =>
26
+ `${directoryOf(partPath)}_rels/${partPath.slice(partPath.lastIndexOf('/') + 1)}.rels`
27
+
28
+ /**
29
+ * Validated parts by case-insensitive name, as SheetJS and OPC look them up. Archive validation
30
+ * rejects names that differ only in case, so each lower-cased name has one part.
31
+ */
32
+ export type PartIndex = {
33
+ /** The validated entry name and bytes of a part, matched ignoring case. */
34
+ readonly find: (path: string) => { readonly name: string; readonly bytes: Uint8Array } | undefined
35
+ }
36
+
37
+ export const indexParts = (parts: Readonly<Record<string, Uint8Array>>): PartIndex => {
38
+ const names = new Map<string, string>()
39
+
40
+ for (const name of Object.keys(parts)) names.set(name.toLowerCase(), name)
41
+
42
+ return {
43
+ find: path => {
44
+ const name = names.get(path.toLowerCase())
45
+ const bytes = name === undefined ? undefined : parts[name]
46
+
47
+ return name === undefined || bytes === undefined ? undefined : { name, bytes }
48
+ }
49
+ }
50
+ }
51
+
52
+ // `[^<>]` keeps each candidate inside one tag, so scans stay linear in the part size.
53
+ const startTagPattern = (localName: string) =>
54
+ new RegExp(`<(?:[\\w.-]+:)?${localName}(?=[\\s/>])[^<>]*>`, 'g')
55
+
56
+ const relationshipTag = startTagPattern('Relationship')
57
+
58
+ export type Relationship = {
59
+ readonly target: string
60
+ /** The `Type` attribute as written (relationship types are case-sensitive URIs). */
61
+ readonly type: string | undefined
62
+ readonly external: boolean
63
+ }
64
+
65
+ /** Relationships by `Id`; the first relationship with an `Id` wins. */
66
+ export const relationships = (xml: string): ReadonlyMap<string, Relationship> => {
67
+ const byId = new Map<string, Relationship>()
68
+
69
+ for (const [tag] of xml.matchAll(relationshipTag)) {
70
+ const attributes = xmlAttributes(tag)
71
+ const id = attributes.get('Id')
72
+ const target = attributes.get('Target')
73
+
74
+ if (id !== undefined && target !== undefined && !byId.has(id))
75
+ byId.set(id, {
76
+ target,
77
+ type: attributes.get('Type'),
78
+ external: attributes.get('TargetMode') === 'External'
79
+ })
80
+ }
81
+
82
+ return byId
83
+ }
@@ -0,0 +1,70 @@
1
+ /** Zero-based cell position. */
2
+ export type CellPosition = {
3
+ readonly r: number
4
+ readonly c: number
5
+ }
6
+
7
+ export type CellRange = {
8
+ readonly start: CellPosition
9
+ readonly end: CellPosition
10
+ }
11
+
12
+ const maxExcelRows = 1_048_576
13
+
14
+ const maxExcelColumns = 16_384
15
+
16
+ const cellAddressPattern = /^([A-Z]{1,3})([1-9][0-9]{0,6})$/
17
+
18
+ const cellAddress = (value: string): CellPosition | undefined => {
19
+ const match = cellAddressPattern.exec(value)
20
+
21
+ if (match?.[1] === undefined || match[2] === undefined) return undefined
22
+
23
+ let column = 0
24
+
25
+ for (const letter of match[1]) column = column * 26 + letter.charCodeAt(0) - 64
26
+
27
+ const row = Number(match[2])
28
+
29
+ if (!Number.isSafeInteger(row) || row > maxExcelRows || column > maxExcelColumns) return undefined
30
+
31
+ return { r: row - 1, c: column - 1 }
32
+ }
33
+
34
+ /**
35
+ * Strictly parse `A1` or `A1:B2` within Excel's grid. Permissive SheetJS range decoders wrap or
36
+ * normalize invalid and overflowed references, so they are never used on untrusted refs.
37
+ */
38
+ export const parseCellRange = (ref: string): CellRange | undefined => {
39
+ if (ref.length > 21) return undefined
40
+
41
+ const parts = ref.split(':')
42
+ const first = parts[0]
43
+
44
+ if (first === undefined || parts.length > 2) return undefined
45
+
46
+ const start = cellAddress(first)
47
+ const end = cellAddress(parts[1] ?? first)
48
+
49
+ if (start === undefined || end === undefined || end.r < start.r || end.c < start.c)
50
+ return undefined
51
+
52
+ return { start, end }
53
+ }
54
+
55
+ export const cellCount = (range: CellRange) =>
56
+ (range.end.r - range.start.r + 1) * (range.end.c - range.start.c + 1)
57
+
58
+ /** `A1`-style address of a zero-based position. */
59
+ export const encodeCellAddress = (position: CellPosition) => {
60
+ let column = ''
61
+ let remaining = position.c + 1
62
+
63
+ while (remaining > 0) {
64
+ const letter = (remaining - 1) % 26
65
+ column = String.fromCharCode(65 + letter) + column
66
+ remaining = Math.floor((remaining - 1) / 26)
67
+ }
68
+
69
+ return `${column}${position.r + 1}`
70
+ }
@@ -0,0 +1,171 @@
1
+ import { sheetJsTags, sheetJsTextViews } from './sheetjs-xml.ts'
2
+ import type { SheetJsTag } from './sheetjs-xml.ts'
3
+
4
+ /**
5
+ * Early, exact rejection of XLSX input that SheetJS 0.20.3 would hand to its ODS, Numbers, or
6
+ * binary (XLSB) parsers, so users get a clear error instead of "Could not read XLSX". These checks
7
+ * are not the security guarantee: SheetJS only ever receives the allowlisted archive built by
8
+ * `buildSheetJsInput` (`xlsx-sheetjs-input.ts`), which contains no marker entries and no `.bin`
9
+ * parts whatever these checks decide.
10
+ */
11
+
12
+ /**
13
+ * An entry name as SheetJS looks it up: its ZIP reader (`cfb_add`) keeps a name that already
14
+ * starts with `Root Entry/` and otherwise stores `"Root Entry/" + name` with the first `//`
15
+ * collapsed; `safegetzipfile` strips `Root Entry/`, treats `\` and `/` alike, and compares
16
+ * lower-cased names.
17
+ */
18
+ export const sheetJsEntryPath = (name: string) =>
19
+ (name.startsWith('Root Entry/') ? name : `Root Entry/${name}`.replace('//', '/'))
20
+ .replace(/^Root Entry\//, '')
21
+ .replaceAll('\\', '/')
22
+ .toLowerCase()
23
+
24
+ /** Entries `parse_zip` checks before content types (SheetJS-normalized paths). */
25
+ const alternateFormatEntries = new Set([
26
+ 'meta-inf/manifest.xml',
27
+ 'objectdata.xml',
28
+ 'index/document.iwa'
29
+ ])
30
+
31
+ /**
32
+ * Whether SheetJS would treat this entry as an ODS, UOC, or Numbers marker. `CFB.find` matches
33
+ * `Index.zip` by base name, and any `Root Entry/` name (any case) is rejected outright.
34
+ */
35
+ export const isAlternateFormatEntry = (name: string) => {
36
+ const path = sheetJsEntryPath(name)
37
+
38
+ return (
39
+ name.toLowerCase().startsWith('root entry/') ||
40
+ alternateFormatEntries.has(path) ||
41
+ path.slice(path.lastIndexOf('/') + 1) === 'index.zip'
42
+ )
43
+ }
44
+
45
+ /** Whether any tag SheetJS would parse from any view of `content` satisfies `test`. */
46
+ const someSheetJsTag = (content: Uint8Array, test: (tag: SheetJsTag) => boolean) =>
47
+ sheetJsTextViews(content).some(text => {
48
+ for (const tag of sheetJsTags(text)) {
49
+ if (test(tag)) return true
50
+ }
51
+
52
+ return false
53
+ })
54
+
55
+ const namedEntities: ReadonlyMap<string, string> = new Map([
56
+ ['quot', '"'],
57
+ ['apos', "'"],
58
+ ['gt', '>'],
59
+ ['lt', '<'],
60
+ ['amp', '&']
61
+ ])
62
+
63
+ /** SheetJS `unescapexml`: XML entities, numeric references, `_xHHHH_` codes, CDATA. */
64
+ const unescapeLikeSheetJs = (text: string): string =>
65
+ text
66
+ .replace(/<!\[CDATA\[|\]\]>/g, '')
67
+ .replace(/&(quot|apos|gt|lt|amp|#x?[\da-f]+);/gi, (raw, entity: string) => {
68
+ const named = namedEntities.get(entity.toLowerCase())
69
+
70
+ if (named !== undefined) return named
71
+
72
+ const hex = entity[1] === 'x' || entity[1] === 'X'
73
+ const code = Number.parseInt(entity.slice(hex ? 2 : 1), hex ? 16 : 10)
74
+
75
+ return Number.isInteger(code) && code <= 0xffff ? String.fromCharCode(code) : raw
76
+ })
77
+ .replace(/_x([\da-f]{4})_/gi, (_, code: string) =>
78
+ String.fromCharCode(Number.parseInt(code, 16))
79
+ )
80
+
81
+ /** SheetJS `resolve_path` from the directory of `xl/…` (only the final segment matters here). */
82
+ const resolveLikeSheetJs = (target: string) => {
83
+ if (target.startsWith('/')) return target.slice(1)
84
+
85
+ const segments = ['xl']
86
+
87
+ for (const step of target.split('/')) {
88
+ if (step === '..') segments.pop()
89
+ else if (step !== '.') segments.push(step)
90
+ }
91
+
92
+ return segments.join('/')
93
+ }
94
+
95
+ /**
96
+ * Whether a `Target` or `PartName` names a path ending in `.bin`, the only suffix SheetJS hands
97
+ * to a binary parser. Checked raw and SheetJS-unescaped, as written and resolved.
98
+ */
99
+ const namesBinaryPart = (value: string) =>
100
+ [value, unescapeLikeSheetJs(value)].some(
101
+ path =>
102
+ path.toLowerCase().endsWith('.bin') || resolveLikeSheetJs(path).toLowerCase().endsWith('.bin')
103
+ )
104
+
105
+ /**
106
+ * Relationship types whose `.bin` targets SheetJS never parses: it follows workbook
107
+ * relationships only as sheets (worksheet, chartsheet, dialogsheet, macrosheet, or no type) and
108
+ * worksheet relationships only as comments, drawings, and legacy drawings.
109
+ */
110
+ const binaryRelationshipTypes = new Set([
111
+ 'printerSettings',
112
+ 'oleObject',
113
+ 'activeXControlBinary',
114
+ 'customProperty',
115
+ 'attachedToolbars',
116
+ 'image',
117
+ 'hyperlink'
118
+ ])
119
+
120
+ /**
121
+ * Whether a relationships part has a `<Relationship>` whose `Target` ends in `.bin` and whose
122
+ * `Type` (read exactly as SheetJS reads it: case-sensitive, missing counts as a sheet) is not on
123
+ * the allowlist.
124
+ */
125
+ export const relationshipsRouteToBinary = (content: Uint8Array) =>
126
+ someSheetJsTag(content, ({ head, attributes }) => {
127
+ const target = attributes.get('Target')
128
+
129
+ if (head !== '<Relationship' || target === undefined || !namesBinaryPart(target)) return false
130
+
131
+ const type = attributes.get('Type')
132
+
133
+ return type === undefined || !binaryRelationshipTypes.has(type.slice(type.lastIndexOf('/') + 1))
134
+ })
135
+
136
+ /** Content types of `.bin` parts SheetJS never parses, lower-cased. */
137
+ const binaryPartContentTypes = new Set([
138
+ 'application/vnd.openxmlformats-officedocument.spreadsheetml.printersettings',
139
+ 'application/vnd.ms-office.activex',
140
+ 'application/vnd.openxmlformats-officedocument.oleobject',
141
+ 'application/vnd.openxmlformats-officedocument.spreadsheetml.customproperty',
142
+ 'application/vnd.ms-excel.attachedtoolbars'
143
+ ])
144
+
145
+ /** XLSB part types: `application/vnd.ms-excel.*` without an `+xml` suffix (toolbars excepted). */
146
+ const isBinarySpreadsheetType = (contentType: string) =>
147
+ contentType.startsWith('application/vnd.ms-excel.') &&
148
+ !contentType.endsWith('+xml') &&
149
+ !binaryPartContentTypes.has(contentType)
150
+
151
+ /**
152
+ * Whether `[Content_Types].xml` has an `<Override>` (any prefix, as SheetJS reads it) with an
153
+ * XLSB content type, or a `PartName` ending in `.bin` whose content type is not on the allowlist.
154
+ * `<Default>` entries are ignored: SheetJS does not route by them, and its own XLSX writer emits
155
+ * `<Default Extension="bin">` with the XLSB workbook type.
156
+ */
157
+ export const contentTypesRouteToBinary = (content: Uint8Array) =>
158
+ someSheetJsTag(content, ({ head, attributes }) => {
159
+ if (head.replace(/<\w*:/, '<') !== '<Override') return false
160
+
161
+ const contentType = attributes.get('ContentType')?.toLowerCase()
162
+ const partName = attributes.get('PartName')
163
+
164
+ if (contentType !== undefined && isBinarySpreadsheetType(contentType)) return true
165
+
166
+ return (
167
+ partName !== undefined &&
168
+ namesBinaryPart(partName) &&
169
+ (contentType === undefined || !binaryPartContentTypes.has(contentType))
170
+ )
171
+ })