@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
import { Buffer } from 'node:buffer'
|
|
2
|
+
import type { StrippedHyperlinkTags } from './office-archive.ts'
|
|
3
|
+
import { indexParts, relationships, relationshipsPathFor, utf8Text } from './xlsx-parts.ts'
|
|
4
|
+
import type { Relationship } from './xlsx-parts.ts'
|
|
5
|
+
import { parseCellRange } from './xlsx-range.ts'
|
|
6
|
+
import type { CellRange } from './xlsx-range.ts'
|
|
7
|
+
import { prefixedAttribute, xmlAttributes } from './xml-text.ts'
|
|
8
|
+
|
|
9
|
+
export type XlsxHyperlink = {
|
|
10
|
+
readonly range: CellRange
|
|
11
|
+
/** Normalized `http:`, `https:`, or `mailto:` URL, at most 2,048 characters. */
|
|
12
|
+
readonly target: string
|
|
13
|
+
/** Label for cells without text, at most 1,024 characters (longer labels end in `…`). */
|
|
14
|
+
readonly display?: string
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
/** Hyperlinks per worksheet name, in document order (a later link wins on overlap). */
|
|
18
|
+
export type XlsxHyperlinks = ReadonlyMap<string, ReadonlyArray<XlsxHyperlink>>
|
|
19
|
+
|
|
20
|
+
/** Longer targets are dropped: a truncated URL would point somewhere else. */
|
|
21
|
+
const maxHyperlinkTargetCharacters = 2048
|
|
22
|
+
|
|
23
|
+
/** Longer display labels are cut to this many characters, ending with an ellipsis. */
|
|
24
|
+
const maxHyperlinkDisplayCharacters = 1024
|
|
25
|
+
|
|
26
|
+
const capDisplay = (display: string) => {
|
|
27
|
+
if (display.length <= maxHyperlinkDisplayCharacters) return display
|
|
28
|
+
|
|
29
|
+
let kept = display.slice(0, maxHyperlinkDisplayCharacters - 1)
|
|
30
|
+
const last = kept.charCodeAt(kept.length - 1)
|
|
31
|
+
|
|
32
|
+
// Never leave half of a surrogate pair before the ellipsis.
|
|
33
|
+
if (last >= 0xd800 && last <= 0xdbff) kept = kept.slice(0, -1)
|
|
34
|
+
|
|
35
|
+
return `${kept}\u2026`
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const shownProtocols = new Set(['http:', 'https:', 'mailto:'])
|
|
39
|
+
|
|
40
|
+
const shownTarget = (target: string) => {
|
|
41
|
+
if (target.length > maxHyperlinkTargetCharacters || !URL.canParse(target)) return undefined
|
|
42
|
+
|
|
43
|
+
const url = new URL(target)
|
|
44
|
+
|
|
45
|
+
if (!shownProtocols.has(url.protocol) || url.href.length > maxHyperlinkTargetCharacters)
|
|
46
|
+
return undefined
|
|
47
|
+
|
|
48
|
+
return url.href
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const hyperlinkFrom = (
|
|
52
|
+
tag: string,
|
|
53
|
+
sheetRelationships: ReadonlyMap<string, Relationship>
|
|
54
|
+
): XlsxHyperlink | undefined => {
|
|
55
|
+
// Tags were captured as Latin-1 text of the original bytes; attribute values are UTF-8.
|
|
56
|
+
const attributes = xmlAttributes(Buffer.from(tag, 'latin1').toString('utf8'))
|
|
57
|
+
const ref = attributes.get('ref')
|
|
58
|
+
const range = ref === undefined ? undefined : parseCellRange(ref)
|
|
59
|
+
const id = prefixedAttribute(attributes, 'id')
|
|
60
|
+
const relationship = id === undefined ? undefined : sheetRelationships.get(id)
|
|
61
|
+
|
|
62
|
+
// Internal `location`-only links (`#Sheet2!A1`) are omitted; external links keep their
|
|
63
|
+
// location as a fragment, as SheetJS does.
|
|
64
|
+
if (range === undefined || relationship === undefined || !relationship.external) return undefined
|
|
65
|
+
|
|
66
|
+
const location = attributes.get('location')
|
|
67
|
+
|
|
68
|
+
const target = shownTarget(
|
|
69
|
+
location === undefined || location.length === 0
|
|
70
|
+
? relationship.target
|
|
71
|
+
: `${relationship.target}#${location}`
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
if (target === undefined) return undefined
|
|
75
|
+
|
|
76
|
+
const display = attributes.get('display')?.trim()
|
|
77
|
+
|
|
78
|
+
return display === undefined || display.length === 0
|
|
79
|
+
? { range, target }
|
|
80
|
+
: { range, target, display: capDisplay(display) }
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** A sheet of the workbook and the validated worksheet part SheetJS reads for it, if any. */
|
|
84
|
+
export type HyperlinkSheet = {
|
|
85
|
+
readonly name: string
|
|
86
|
+
readonly part: string | undefined
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Map the hyperlink tags removed from a workbook to its sheets (the same sheet list and worksheet
|
|
91
|
+
* parts SheetJS reads, from `buildSheetJsInput`) through each worksheet's relationships.
|
|
92
|
+
*/
|
|
93
|
+
export const resolveXlsxHyperlinks = (
|
|
94
|
+
parts: Readonly<Record<string, Uint8Array>>,
|
|
95
|
+
hyperlinkTags: StrippedHyperlinkTags,
|
|
96
|
+
sheets: ReadonlyArray<HyperlinkSheet>
|
|
97
|
+
): XlsxHyperlinks => {
|
|
98
|
+
const bySheet = new Map<string, ReadonlyArray<XlsxHyperlink>>()
|
|
99
|
+
|
|
100
|
+
if (hyperlinkTags.size === 0) return bySheet
|
|
101
|
+
|
|
102
|
+
const index = indexParts(parts)
|
|
103
|
+
const text = (path: string) => utf8Text(index.find(path)?.bytes)
|
|
104
|
+
|
|
105
|
+
for (const { name, part } of sheets) {
|
|
106
|
+
const tags = part === undefined ? undefined : hyperlinkTags.get(part)
|
|
107
|
+
|
|
108
|
+
if (part === undefined || tags === undefined || bySheet.has(name)) continue
|
|
109
|
+
|
|
110
|
+
const sheetRelationships = relationships(text(relationshipsPathFor(part)))
|
|
111
|
+
|
|
112
|
+
const links = tags.flatMap(linkTag => {
|
|
113
|
+
const link = hyperlinkFrom(linkTag, sheetRelationships)
|
|
114
|
+
|
|
115
|
+
return link === undefined ? [] : [link]
|
|
116
|
+
})
|
|
117
|
+
|
|
118
|
+
if (links.length > 0) bySheet.set(name, links)
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
return bySheet
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
type ClippedLink = {
|
|
125
|
+
readonly order: number
|
|
126
|
+
readonly firstRow: number
|
|
127
|
+
readonly lastRow: number
|
|
128
|
+
readonly firstColumn: number
|
|
129
|
+
readonly lastColumn: number
|
|
130
|
+
readonly link: XlsxHyperlink
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
export type HyperlinkLookup = {
|
|
134
|
+
/** The winning link at a cell. Rows must be queried in ascending order. */
|
|
135
|
+
readonly at: (row: number, column: number) => XlsxHyperlink | undefined
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* Index links clipped to the visited range for row-major lookup without scanning every link per
|
|
140
|
+
* cell: a segment tree over columns whose nodes hold max-heaps (by document order) of the links
|
|
141
|
+
* covering them. Links enter when the sweep reaches their first row and are discarded lazily once
|
|
142
|
+
* it passes their last row. Work is O((links + cells) · log(columns) · log(links)).
|
|
143
|
+
*/
|
|
144
|
+
export const makeHyperlinkLookup = (
|
|
145
|
+
links: ReadonlyArray<XlsxHyperlink>,
|
|
146
|
+
visited: CellRange
|
|
147
|
+
): HyperlinkLookup => {
|
|
148
|
+
const clipped = links
|
|
149
|
+
.flatMap((link, order): ReadonlyArray<ClippedLink> => {
|
|
150
|
+
const firstRow = Math.max(link.range.start.r, visited.start.r)
|
|
151
|
+
const lastRow = Math.min(link.range.end.r, visited.end.r)
|
|
152
|
+
const firstColumn = Math.max(link.range.start.c, visited.start.c) - visited.start.c
|
|
153
|
+
const lastColumn = Math.min(link.range.end.c, visited.end.c) - visited.start.c
|
|
154
|
+
|
|
155
|
+
return firstRow > lastRow || firstColumn > lastColumn
|
|
156
|
+
? []
|
|
157
|
+
: [{ order, firstRow, lastRow, firstColumn, lastColumn, link }]
|
|
158
|
+
})
|
|
159
|
+
.sort((left, right) => left.firstRow - right.firstRow || left.order - right.order)
|
|
160
|
+
|
|
161
|
+
if (clipped.length === 0) return { at: () => undefined }
|
|
162
|
+
|
|
163
|
+
const width = visited.end.c - visited.start.c + 1
|
|
164
|
+
let leaves = 1
|
|
165
|
+
|
|
166
|
+
while (leaves < width) leaves *= 2
|
|
167
|
+
|
|
168
|
+
const heaps: Array<Array<ClippedLink> | undefined> = []
|
|
169
|
+
|
|
170
|
+
const push = (node: number, entry: ClippedLink) => {
|
|
171
|
+
const heap = heaps[node] ?? []
|
|
172
|
+
heaps[node] = heap
|
|
173
|
+
heap.push(entry)
|
|
174
|
+
|
|
175
|
+
let index = heap.length - 1
|
|
176
|
+
|
|
177
|
+
while (index > 0) {
|
|
178
|
+
const parent = (index - 1) >> 1
|
|
179
|
+
const parentEntry = heap[parent]
|
|
180
|
+
|
|
181
|
+
if (parentEntry === undefined || parentEntry.order >= entry.order) break
|
|
182
|
+
|
|
183
|
+
heap[index] = parentEntry
|
|
184
|
+
heap[parent] = entry
|
|
185
|
+
index = parent
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
const pop = (heap: Array<ClippedLink>) => {
|
|
190
|
+
const last = heap.pop()
|
|
191
|
+
|
|
192
|
+
if (last === undefined || heap.length === 0) return
|
|
193
|
+
|
|
194
|
+
heap[0] = last
|
|
195
|
+
|
|
196
|
+
let index = 0
|
|
197
|
+
|
|
198
|
+
for (;;) {
|
|
199
|
+
const left = index * 2 + 1
|
|
200
|
+
const right = left + 1
|
|
201
|
+
let largest = index
|
|
202
|
+
|
|
203
|
+
const largestOrder = () => heap[largest]?.order ?? -1
|
|
204
|
+
|
|
205
|
+
if ((heap[left]?.order ?? -1) > largestOrder()) largest = left
|
|
206
|
+
|
|
207
|
+
if ((heap[right]?.order ?? -1) > largestOrder()) largest = right
|
|
208
|
+
|
|
209
|
+
const swapped = heap[largest]
|
|
210
|
+
|
|
211
|
+
if (largest === index || swapped === undefined) return
|
|
212
|
+
|
|
213
|
+
heap[largest] = last
|
|
214
|
+
heap[index] = swapped
|
|
215
|
+
index = largest
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
const insert = (entry: ClippedLink) => {
|
|
220
|
+
let low = entry.firstColumn + leaves
|
|
221
|
+
let high = entry.lastColumn + leaves + 1
|
|
222
|
+
|
|
223
|
+
while (low < high) {
|
|
224
|
+
if ((low & 1) === 1) push(low++, entry)
|
|
225
|
+
|
|
226
|
+
if ((high & 1) === 1) push(--high, entry)
|
|
227
|
+
|
|
228
|
+
low >>= 1
|
|
229
|
+
high >>= 1
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
let nextEntry = 0
|
|
234
|
+
|
|
235
|
+
return {
|
|
236
|
+
at: (row, column) => {
|
|
237
|
+
for (let entry = clipped[nextEntry]; entry !== undefined && entry.firstRow <= row;) {
|
|
238
|
+
insert(entry)
|
|
239
|
+
nextEntry += 1
|
|
240
|
+
entry = clipped[nextEntry]
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
let best: ClippedLink | undefined
|
|
244
|
+
|
|
245
|
+
for (let node = column - visited.start.c + leaves; node >= 1; node >>= 1) {
|
|
246
|
+
const heap = heaps[node]
|
|
247
|
+
|
|
248
|
+
if (heap === undefined) continue
|
|
249
|
+
|
|
250
|
+
while (heap[0] !== undefined && heap[0].lastRow < row) pop(heap)
|
|
251
|
+
|
|
252
|
+
const top = heap[0]
|
|
253
|
+
|
|
254
|
+
if (top !== undefined && (best === undefined || top.order > best.order)) best = top
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
return best?.link
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
}
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
import { Buffer } from 'node:buffer'
|
|
2
|
+
import { xmlAttributes } from './xml-text.ts'
|
|
3
|
+
|
|
4
|
+
/** The main workbook part every validated XLSX archive contains. */
|
|
5
|
+
export const workbookPartPath = 'xl/workbook.xml'
|
|
6
|
+
|
|
7
|
+
export const utf8Text = (bytes: Uint8Array | undefined) =>
|
|
8
|
+
bytes === undefined ? '' : Buffer.from(bytes).toString('utf8')
|
|
9
|
+
|
|
10
|
+
/** Resolve a relationship target against the directory of its source part (`xl/`, …). */
|
|
11
|
+
export const resolvePartPath = (baseDirectory: string, target: string) => {
|
|
12
|
+
const segments: Array<string> = []
|
|
13
|
+
const path = target.startsWith('/') ? target.slice(1) : `${baseDirectory}${target}`
|
|
14
|
+
|
|
15
|
+
for (const segment of path.split('/')) {
|
|
16
|
+
if (segment === '..') segments.pop()
|
|
17
|
+
else if (segment !== '.' && segment !== '') segments.push(segment)
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
return segments.join('/')
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export const directoryOf = (path: string) => path.slice(0, path.lastIndexOf('/') + 1)
|
|
24
|
+
|
|
25
|
+
export const relationshipsPathFor = (partPath: string) =>
|
|
26
|
+
`${directoryOf(partPath)}_rels/${partPath.slice(partPath.lastIndexOf('/') + 1)}.rels`
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Validated parts by case-insensitive name, as SheetJS and OPC look them up. Archive validation
|
|
30
|
+
* rejects names that differ only in case, so each lower-cased name has one part.
|
|
31
|
+
*/
|
|
32
|
+
export type PartIndex = {
|
|
33
|
+
/** The validated entry name and bytes of a part, matched ignoring case. */
|
|
34
|
+
readonly find: (path: string) => { readonly name: string; readonly bytes: Uint8Array } | undefined
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export const indexParts = (parts: Readonly<Record<string, Uint8Array>>): PartIndex => {
|
|
38
|
+
const names = new Map<string, string>()
|
|
39
|
+
|
|
40
|
+
for (const name of Object.keys(parts)) names.set(name.toLowerCase(), name)
|
|
41
|
+
|
|
42
|
+
return {
|
|
43
|
+
find: path => {
|
|
44
|
+
const name = names.get(path.toLowerCase())
|
|
45
|
+
const bytes = name === undefined ? undefined : parts[name]
|
|
46
|
+
|
|
47
|
+
return name === undefined || bytes === undefined ? undefined : { name, bytes }
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// `[^<>]` keeps each candidate inside one tag, so scans stay linear in the part size.
|
|
53
|
+
const startTagPattern = (localName: string) =>
|
|
54
|
+
new RegExp(`<(?:[\\w.-]+:)?${localName}(?=[\\s/>])[^<>]*>`, 'g')
|
|
55
|
+
|
|
56
|
+
const relationshipTag = startTagPattern('Relationship')
|
|
57
|
+
|
|
58
|
+
export type Relationship = {
|
|
59
|
+
readonly target: string
|
|
60
|
+
/** The `Type` attribute as written (relationship types are case-sensitive URIs). */
|
|
61
|
+
readonly type: string | undefined
|
|
62
|
+
readonly external: boolean
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** Relationships by `Id`; the first relationship with an `Id` wins. */
|
|
66
|
+
export const relationships = (xml: string): ReadonlyMap<string, Relationship> => {
|
|
67
|
+
const byId = new Map<string, Relationship>()
|
|
68
|
+
|
|
69
|
+
for (const [tag] of xml.matchAll(relationshipTag)) {
|
|
70
|
+
const attributes = xmlAttributes(tag)
|
|
71
|
+
const id = attributes.get('Id')
|
|
72
|
+
const target = attributes.get('Target')
|
|
73
|
+
|
|
74
|
+
if (id !== undefined && target !== undefined && !byId.has(id))
|
|
75
|
+
byId.set(id, {
|
|
76
|
+
target,
|
|
77
|
+
type: attributes.get('Type'),
|
|
78
|
+
external: attributes.get('TargetMode') === 'External'
|
|
79
|
+
})
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
return byId
|
|
83
|
+
}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
/** Zero-based cell position. */
|
|
2
|
+
export type CellPosition = {
|
|
3
|
+
readonly r: number
|
|
4
|
+
readonly c: number
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
export type CellRange = {
|
|
8
|
+
readonly start: CellPosition
|
|
9
|
+
readonly end: CellPosition
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
const maxExcelRows = 1_048_576
|
|
13
|
+
|
|
14
|
+
const maxExcelColumns = 16_384
|
|
15
|
+
|
|
16
|
+
const cellAddressPattern = /^([A-Z]{1,3})([1-9][0-9]{0,6})$/
|
|
17
|
+
|
|
18
|
+
const cellAddress = (value: string): CellPosition | undefined => {
|
|
19
|
+
const match = cellAddressPattern.exec(value)
|
|
20
|
+
|
|
21
|
+
if (match?.[1] === undefined || match[2] === undefined) return undefined
|
|
22
|
+
|
|
23
|
+
let column = 0
|
|
24
|
+
|
|
25
|
+
for (const letter of match[1]) column = column * 26 + letter.charCodeAt(0) - 64
|
|
26
|
+
|
|
27
|
+
const row = Number(match[2])
|
|
28
|
+
|
|
29
|
+
if (!Number.isSafeInteger(row) || row > maxExcelRows || column > maxExcelColumns) return undefined
|
|
30
|
+
|
|
31
|
+
return { r: row - 1, c: column - 1 }
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Strictly parse `A1` or `A1:B2` within Excel's grid. Permissive SheetJS range decoders wrap or
|
|
36
|
+
* normalize invalid and overflowed references, so they are never used on untrusted refs.
|
|
37
|
+
*/
|
|
38
|
+
export const parseCellRange = (ref: string): CellRange | undefined => {
|
|
39
|
+
if (ref.length > 21) return undefined
|
|
40
|
+
|
|
41
|
+
const parts = ref.split(':')
|
|
42
|
+
const first = parts[0]
|
|
43
|
+
|
|
44
|
+
if (first === undefined || parts.length > 2) return undefined
|
|
45
|
+
|
|
46
|
+
const start = cellAddress(first)
|
|
47
|
+
const end = cellAddress(parts[1] ?? first)
|
|
48
|
+
|
|
49
|
+
if (start === undefined || end === undefined || end.r < start.r || end.c < start.c)
|
|
50
|
+
return undefined
|
|
51
|
+
|
|
52
|
+
return { start, end }
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export const cellCount = (range: CellRange) =>
|
|
56
|
+
(range.end.r - range.start.r + 1) * (range.end.c - range.start.c + 1)
|
|
57
|
+
|
|
58
|
+
/** `A1`-style address of a zero-based position. */
|
|
59
|
+
export const encodeCellAddress = (position: CellPosition) => {
|
|
60
|
+
let column = ''
|
|
61
|
+
let remaining = position.c + 1
|
|
62
|
+
|
|
63
|
+
while (remaining > 0) {
|
|
64
|
+
const letter = (remaining - 1) % 26
|
|
65
|
+
column = String.fromCharCode(65 + letter) + column
|
|
66
|
+
remaining = Math.floor((remaining - 1) / 26)
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
return `${column}${position.r + 1}`
|
|
70
|
+
}
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
import { sheetJsTags, sheetJsTextViews } from './sheetjs-xml.ts'
|
|
2
|
+
import type { SheetJsTag } from './sheetjs-xml.ts'
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Early, exact rejection of XLSX input that SheetJS 0.20.3 would hand to its ODS, Numbers, or
|
|
6
|
+
* binary (XLSB) parsers, so users get a clear error instead of "Could not read XLSX". These checks
|
|
7
|
+
* are not the security guarantee: SheetJS only ever receives the allowlisted archive built by
|
|
8
|
+
* `buildSheetJsInput` (`xlsx-sheetjs-input.ts`), which contains no marker entries and no `.bin`
|
|
9
|
+
* parts whatever these checks decide.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* An entry name as SheetJS looks it up: its ZIP reader (`cfb_add`) keeps a name that already
|
|
14
|
+
* starts with `Root Entry/` and otherwise stores `"Root Entry/" + name` with the first `//`
|
|
15
|
+
* collapsed; `safegetzipfile` strips `Root Entry/`, treats `\` and `/` alike, and compares
|
|
16
|
+
* lower-cased names.
|
|
17
|
+
*/
|
|
18
|
+
export const sheetJsEntryPath = (name: string) =>
|
|
19
|
+
(name.startsWith('Root Entry/') ? name : `Root Entry/${name}`.replace('//', '/'))
|
|
20
|
+
.replace(/^Root Entry\//, '')
|
|
21
|
+
.replaceAll('\\', '/')
|
|
22
|
+
.toLowerCase()
|
|
23
|
+
|
|
24
|
+
/** Entries `parse_zip` checks before content types (SheetJS-normalized paths). */
|
|
25
|
+
const alternateFormatEntries = new Set([
|
|
26
|
+
'meta-inf/manifest.xml',
|
|
27
|
+
'objectdata.xml',
|
|
28
|
+
'index/document.iwa'
|
|
29
|
+
])
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Whether SheetJS would treat this entry as an ODS, UOC, or Numbers marker. `CFB.find` matches
|
|
33
|
+
* `Index.zip` by base name, and any `Root Entry/` name (any case) is rejected outright.
|
|
34
|
+
*/
|
|
35
|
+
export const isAlternateFormatEntry = (name: string) => {
|
|
36
|
+
const path = sheetJsEntryPath(name)
|
|
37
|
+
|
|
38
|
+
return (
|
|
39
|
+
name.toLowerCase().startsWith('root entry/') ||
|
|
40
|
+
alternateFormatEntries.has(path) ||
|
|
41
|
+
path.slice(path.lastIndexOf('/') + 1) === 'index.zip'
|
|
42
|
+
)
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** Whether any tag SheetJS would parse from any view of `content` satisfies `test`. */
|
|
46
|
+
const someSheetJsTag = (content: Uint8Array, test: (tag: SheetJsTag) => boolean) =>
|
|
47
|
+
sheetJsTextViews(content).some(text => {
|
|
48
|
+
for (const tag of sheetJsTags(text)) {
|
|
49
|
+
if (test(tag)) return true
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
return false
|
|
53
|
+
})
|
|
54
|
+
|
|
55
|
+
const namedEntities: ReadonlyMap<string, string> = new Map([
|
|
56
|
+
['quot', '"'],
|
|
57
|
+
['apos', "'"],
|
|
58
|
+
['gt', '>'],
|
|
59
|
+
['lt', '<'],
|
|
60
|
+
['amp', '&']
|
|
61
|
+
])
|
|
62
|
+
|
|
63
|
+
/** SheetJS `unescapexml`: XML entities, numeric references, `_xHHHH_` codes, CDATA. */
|
|
64
|
+
const unescapeLikeSheetJs = (text: string): string =>
|
|
65
|
+
text
|
|
66
|
+
.replace(/<!\[CDATA\[|\]\]>/g, '')
|
|
67
|
+
.replace(/&(quot|apos|gt|lt|amp|#x?[\da-f]+);/gi, (raw, entity: string) => {
|
|
68
|
+
const named = namedEntities.get(entity.toLowerCase())
|
|
69
|
+
|
|
70
|
+
if (named !== undefined) return named
|
|
71
|
+
|
|
72
|
+
const hex = entity[1] === 'x' || entity[1] === 'X'
|
|
73
|
+
const code = Number.parseInt(entity.slice(hex ? 2 : 1), hex ? 16 : 10)
|
|
74
|
+
|
|
75
|
+
return Number.isInteger(code) && code <= 0xffff ? String.fromCharCode(code) : raw
|
|
76
|
+
})
|
|
77
|
+
.replace(/_x([\da-f]{4})_/gi, (_, code: string) =>
|
|
78
|
+
String.fromCharCode(Number.parseInt(code, 16))
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
/** SheetJS `resolve_path` from the directory of `xl/…` (only the final segment matters here). */
|
|
82
|
+
const resolveLikeSheetJs = (target: string) => {
|
|
83
|
+
if (target.startsWith('/')) return target.slice(1)
|
|
84
|
+
|
|
85
|
+
const segments = ['xl']
|
|
86
|
+
|
|
87
|
+
for (const step of target.split('/')) {
|
|
88
|
+
if (step === '..') segments.pop()
|
|
89
|
+
else if (step !== '.') segments.push(step)
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
return segments.join('/')
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Whether a `Target` or `PartName` names a path ending in `.bin`, the only suffix SheetJS hands
|
|
97
|
+
* to a binary parser. Checked raw and SheetJS-unescaped, as written and resolved.
|
|
98
|
+
*/
|
|
99
|
+
const namesBinaryPart = (value: string) =>
|
|
100
|
+
[value, unescapeLikeSheetJs(value)].some(
|
|
101
|
+
path =>
|
|
102
|
+
path.toLowerCase().endsWith('.bin') || resolveLikeSheetJs(path).toLowerCase().endsWith('.bin')
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Relationship types whose `.bin` targets SheetJS never parses: it follows workbook
|
|
107
|
+
* relationships only as sheets (worksheet, chartsheet, dialogsheet, macrosheet, or no type) and
|
|
108
|
+
* worksheet relationships only as comments, drawings, and legacy drawings.
|
|
109
|
+
*/
|
|
110
|
+
const binaryRelationshipTypes = new Set([
|
|
111
|
+
'printerSettings',
|
|
112
|
+
'oleObject',
|
|
113
|
+
'activeXControlBinary',
|
|
114
|
+
'customProperty',
|
|
115
|
+
'attachedToolbars',
|
|
116
|
+
'image',
|
|
117
|
+
'hyperlink'
|
|
118
|
+
])
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Whether a relationships part has a `<Relationship>` whose `Target` ends in `.bin` and whose
|
|
122
|
+
* `Type` (read exactly as SheetJS reads it: case-sensitive, missing counts as a sheet) is not on
|
|
123
|
+
* the allowlist.
|
|
124
|
+
*/
|
|
125
|
+
export const relationshipsRouteToBinary = (content: Uint8Array) =>
|
|
126
|
+
someSheetJsTag(content, ({ head, attributes }) => {
|
|
127
|
+
const target = attributes.get('Target')
|
|
128
|
+
|
|
129
|
+
if (head !== '<Relationship' || target === undefined || !namesBinaryPart(target)) return false
|
|
130
|
+
|
|
131
|
+
const type = attributes.get('Type')
|
|
132
|
+
|
|
133
|
+
return type === undefined || !binaryRelationshipTypes.has(type.slice(type.lastIndexOf('/') + 1))
|
|
134
|
+
})
|
|
135
|
+
|
|
136
|
+
/** Content types of `.bin` parts SheetJS never parses, lower-cased. */
|
|
137
|
+
const binaryPartContentTypes = new Set([
|
|
138
|
+
'application/vnd.openxmlformats-officedocument.spreadsheetml.printersettings',
|
|
139
|
+
'application/vnd.ms-office.activex',
|
|
140
|
+
'application/vnd.openxmlformats-officedocument.oleobject',
|
|
141
|
+
'application/vnd.openxmlformats-officedocument.spreadsheetml.customproperty',
|
|
142
|
+
'application/vnd.ms-excel.attachedtoolbars'
|
|
143
|
+
])
|
|
144
|
+
|
|
145
|
+
/** XLSB part types: `application/vnd.ms-excel.*` without an `+xml` suffix (toolbars excepted). */
|
|
146
|
+
const isBinarySpreadsheetType = (contentType: string) =>
|
|
147
|
+
contentType.startsWith('application/vnd.ms-excel.') &&
|
|
148
|
+
!contentType.endsWith('+xml') &&
|
|
149
|
+
!binaryPartContentTypes.has(contentType)
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* Whether `[Content_Types].xml` has an `<Override>` (any prefix, as SheetJS reads it) with an
|
|
153
|
+
* XLSB content type, or a `PartName` ending in `.bin` whose content type is not on the allowlist.
|
|
154
|
+
* `<Default>` entries are ignored: SheetJS does not route by them, and its own XLSX writer emits
|
|
155
|
+
* `<Default Extension="bin">` with the XLSB workbook type.
|
|
156
|
+
*/
|
|
157
|
+
export const contentTypesRouteToBinary = (content: Uint8Array) =>
|
|
158
|
+
someSheetJsTag(content, ({ head, attributes }) => {
|
|
159
|
+
if (head.replace(/<\w*:/, '<') !== '<Override') return false
|
|
160
|
+
|
|
161
|
+
const contentType = attributes.get('ContentType')?.toLowerCase()
|
|
162
|
+
const partName = attributes.get('PartName')
|
|
163
|
+
|
|
164
|
+
if (contentType !== undefined && isBinarySpreadsheetType(contentType)) return true
|
|
165
|
+
|
|
166
|
+
return (
|
|
167
|
+
partName !== undefined &&
|
|
168
|
+
namesBinaryPart(partName) &&
|
|
169
|
+
(contentType === undefined || !binaryPartContentTypes.has(contentType))
|
|
170
|
+
)
|
|
171
|
+
})
|