@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
import { strToU8, zipSync } from 'fflate'
|
|
2
|
+
import { FileExtractionError } from '../errors.ts'
|
|
3
|
+
import {
|
|
4
|
+
coreTitle,
|
|
5
|
+
officeDocumentRelationshipsNamespace,
|
|
6
|
+
sheetJsCouldReadCdata,
|
|
7
|
+
sheetJsXmlHeader
|
|
8
|
+
} from './sheetjs-xml.ts'
|
|
9
|
+
import { readStyles, stylesXml } from './xlsx-styles.ts'
|
|
10
|
+
import { readWorkbookModel, workbookXml } from './xlsx-workbook.ts'
|
|
11
|
+
import {
|
|
12
|
+
directoryOf,
|
|
13
|
+
indexParts,
|
|
14
|
+
relationships,
|
|
15
|
+
relationshipsPathFor,
|
|
16
|
+
resolvePartPath,
|
|
17
|
+
utf8Text,
|
|
18
|
+
workbookPartPath
|
|
19
|
+
} from './xlsx-parts.ts'
|
|
20
|
+
import type { PartIndex, Relationship } from './xlsx-parts.ts'
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* The archive SheetJS reads is built here from scratch, never passed through. Every part SheetJS
|
|
24
|
+
* reads is generated or checked:
|
|
25
|
+
*
|
|
26
|
+
* - generated: `[Content_Types].xml`, `_rels/.rels`, `xl/_rels/workbook.xml.rels`,
|
|
27
|
+
* `xl/workbook.xml` (`xlsx-workbook.ts`), and `xl/styles.xml` (`xlsx-styles.ts`);
|
|
28
|
+
* - copied after checks: worksheets (stored as `xl/worksheets/sheet<n>.xml`) and
|
|
29
|
+
* `xl/sharedStrings.xml`, validated, hyperlink-stripped, and free of anything SheetJS could
|
|
30
|
+
* turn into a CDATA marker (`sheetJsCouldReadCdata`).
|
|
31
|
+
*
|
|
32
|
+
* With these entries SheetJS 0.20.3 `parse_zip` can only take its XLSX path:
|
|
33
|
+
*
|
|
34
|
+
* - no `META-INF/manifest.xml`, `objectdata.xml`, or `Index/Document.iwa`, so it never reaches
|
|
35
|
+
* `parse_ods` or `parse_numbers_iwa`; `[Content_Types].xml` exists, so `Index.zip` is never read;
|
|
36
|
+
* - every entry ends in `.xml` or `.rels`, so no binary (XLSB) parser can receive data: SheetJS
|
|
37
|
+
* dispatches on the requested path ending in `.bin`, and `safegetzipfile` only returns an entry
|
|
38
|
+
* whose name equals that path ignoring case;
|
|
39
|
+
* - the generated content types name the XML workbook, so `xlsb` stays false, and no attacker
|
|
40
|
+
* `Override`, `PartName`, relationship `Type`, or `Target` reaches SheetJS;
|
|
41
|
+
* - the generated workbook lists exactly the sheets the extractor counted, each with its own
|
|
42
|
+
* relationship to its own `xl/worksheets/sheet<n>.xml`, so every part is parsed at most once.
|
|
43
|
+
*
|
|
44
|
+
* Comments, threaded comments, VML, drawings, worksheet relationships, external links, pivot
|
|
45
|
+
* caches, calculation chains, metadata, themes, `customXml`, and `docProps/*` (the title is read
|
|
46
|
+
* by the extractor) are left out; SheetJS reads none of them to produce cell values or display
|
|
47
|
+
* text.
|
|
48
|
+
*/
|
|
49
|
+
|
|
50
|
+
const strictOfficeDocumentRelationships = 'http://purl.oclc.org/ooxml/officeDocument/relationships'
|
|
51
|
+
|
|
52
|
+
const corePropertiesTypes = new Set([
|
|
53
|
+
'http://schemas.openxmlformats.org/package/2006/relationships/metadata/core-properties',
|
|
54
|
+
'http://schemas.openxmlformats.org/officedocument/2006/relationships/metadata/core-properties'
|
|
55
|
+
])
|
|
56
|
+
|
|
57
|
+
/** A relationship of `kind` in the transitional or strict namespace. */
|
|
58
|
+
const hasKind = (relationship: Relationship, kind: string) =>
|
|
59
|
+
relationship.type === `${officeDocumentRelationshipsNamespace}/${kind}` ||
|
|
60
|
+
relationship.type === `${strictOfficeDocumentRelationships}/${kind}`
|
|
61
|
+
|
|
62
|
+
/** Canonical names SheetJS can receive besides `xl/worksheets/sheet<n>.xml`. */
|
|
63
|
+
export const sheetJsFixedParts = [
|
|
64
|
+
'[Content_Types].xml',
|
|
65
|
+
'_rels/.rels',
|
|
66
|
+
'xl/workbook.xml',
|
|
67
|
+
'xl/_rels/workbook.xml.rels',
|
|
68
|
+
'xl/sharedStrings.xml',
|
|
69
|
+
'xl/styles.xml'
|
|
70
|
+
] as const
|
|
71
|
+
|
|
72
|
+
const canonicalWorksheet = /^xl\/worksheets\/sheet[1-9]\d*\.xml$/
|
|
73
|
+
|
|
74
|
+
const fixedPartNames: ReadonlySet<string> = new Set(sheetJsFixedParts)
|
|
75
|
+
|
|
76
|
+
/** Whether `name` is one of the canonical names `buildSheetJsInput` emits. */
|
|
77
|
+
export const isSheetJsInputName = (name: string) =>
|
|
78
|
+
fixedPartNames.has(name) || canonicalWorksheet.test(name)
|
|
79
|
+
|
|
80
|
+
const contentType = {
|
|
81
|
+
workbook: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml',
|
|
82
|
+
worksheet: 'application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml',
|
|
83
|
+
sharedStrings: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sharedStrings+xml',
|
|
84
|
+
styles: 'application/vnd.openxmlformats-officedocument.spreadsheetml.styles+xml'
|
|
85
|
+
} as const
|
|
86
|
+
|
|
87
|
+
const contentTypesXml = (overrides: ReadonlyArray<readonly [string, string]>) =>
|
|
88
|
+
`${sheetJsXmlHeader}<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"><Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/><Default Extension="xml" ContentType="application/xml"/>${overrides
|
|
89
|
+
.map(([path, type]) => `<Override PartName="/${path}" ContentType="${type}"/>`)
|
|
90
|
+
.join('')}</Types>`
|
|
91
|
+
|
|
92
|
+
const relationshipsXml = (entries: ReadonlyArray<readonly [string, string, string]>) =>
|
|
93
|
+
`${sheetJsXmlHeader}<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">${entries
|
|
94
|
+
.map(([id, type, target]) => `<Relationship Id="${id}" Type="${type}" Target="${target}"/>`)
|
|
95
|
+
.join('')}</Relationships>`
|
|
96
|
+
|
|
97
|
+
/** An XML part found by name ignoring case; `.xml` only, so no binary bytes are renamed. */
|
|
98
|
+
const xmlPart = (index: PartIndex, path: string | undefined) => {
|
|
99
|
+
const part = path === undefined ? undefined : index.find(path)
|
|
100
|
+
|
|
101
|
+
return part !== undefined && part.name.toLowerCase().endsWith('.xml') ? part : undefined
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/** The part a relationship of `kind` points at, else the conventional path. */
|
|
105
|
+
const relatedXmlPart = (
|
|
106
|
+
index: PartIndex,
|
|
107
|
+
related: ReadonlyMap<string, Relationship>,
|
|
108
|
+
baseDirectory: string,
|
|
109
|
+
matches: (relationship: Relationship) => boolean,
|
|
110
|
+
conventionalPath: string
|
|
111
|
+
) => {
|
|
112
|
+
for (const relationship of related.values()) {
|
|
113
|
+
if (!relationship.external && matches(relationship))
|
|
114
|
+
return xmlPart(index, resolvePartPath(baseDirectory, relationship.target))
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
return xmlPart(index, conventionalPath)
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
const malformed = () =>
|
|
121
|
+
new FileExtractionError({ format: 'xlsx', message: 'XLSX workbook is malformed.' })
|
|
122
|
+
|
|
123
|
+
const cdataRejected = () =>
|
|
124
|
+
new FileExtractionError({
|
|
125
|
+
format: 'xlsx',
|
|
126
|
+
message:
|
|
127
|
+
'XLSX contains unsupported markup (CDATA, comments or declarations) in worksheet or shared-strings parts.'
|
|
128
|
+
})
|
|
129
|
+
|
|
130
|
+
/** A copied part, unless SheetJS could meet a CDATA marker in it (decoded or tag-removed). */
|
|
131
|
+
const withoutCdata = (bytes: Uint8Array) => {
|
|
132
|
+
if (sheetJsCouldReadCdata(bytes)) throw cdataRejected()
|
|
133
|
+
|
|
134
|
+
return bytes
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
export type SheetJsInputSheet = {
|
|
138
|
+
/** The sheet name exactly as SheetJS reads it from the generated workbook. */
|
|
139
|
+
readonly name: string
|
|
140
|
+
/** The validated source part handed over as this sheet's worksheet, if any. */
|
|
141
|
+
readonly part: string | undefined
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
export type SheetJsInput = {
|
|
145
|
+
/** The stored-entry ZIP handed to SheetJS. */
|
|
146
|
+
readonly archive: Uint8Array
|
|
147
|
+
/** Entry names in `archive`, all canonical (see `isSheetJsInputName`). */
|
|
148
|
+
readonly names: ReadonlyArray<string>
|
|
149
|
+
/** Every sheet of the generated workbook, in order. */
|
|
150
|
+
readonly sheets: ReadonlyArray<SheetJsInputSheet>
|
|
151
|
+
/** The workbook title from its core properties, read without SheetJS. */
|
|
152
|
+
readonly title: string | undefined
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* Build SheetJS's input from validated parts. Sheet `n` of the workbook gets relationship
|
|
157
|
+
* `rId<n>` to `xl/worksheets/sheet<n>.xml`. That entry holds the sheet's worksheet when its
|
|
158
|
+
* workbook relationship is an internal worksheet resolving to an `.xml` part; other sheets (chart
|
|
159
|
+
* sheets, missing parts) keep their name and order but have no entry, so SheetJS skips them.
|
|
160
|
+
*
|
|
161
|
+
* Throws `FileExtractionError` when the workbook declares more than `maxSheets` sheets, when it
|
|
162
|
+
* is malformed (see `readWorkbookModel`; two sheets on one worksheet part count too), or when a
|
|
163
|
+
* copied part holds CDATA, all before SheetJS runs.
|
|
164
|
+
*/
|
|
165
|
+
export const buildSheetJsInput = (
|
|
166
|
+
parts: Readonly<Record<string, Uint8Array>>,
|
|
167
|
+
maxSheets: number
|
|
168
|
+
): SheetJsInput => {
|
|
169
|
+
const index = indexParts(parts)
|
|
170
|
+
const workbook = index.find(workbookPartPath)
|
|
171
|
+
|
|
172
|
+
if (workbook === undefined)
|
|
173
|
+
throw new FileExtractionError({ format: 'xlsx', message: 'Invalid Office archive.' })
|
|
174
|
+
|
|
175
|
+
const workbookDirectory = directoryOf(workbookPartPath)
|
|
176
|
+
|
|
177
|
+
const workbookRelationships = relationships(
|
|
178
|
+
utf8Text(index.find(relationshipsPathFor(workbookPartPath))?.bytes)
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
const model = readWorkbookModel(workbook.bytes, maxSheets)
|
|
182
|
+
const files: Record<string, Uint8Array> = Object.create(null)
|
|
183
|
+
const overrides: Array<readonly [string, string]> = []
|
|
184
|
+
const sheetRelationships: Array<readonly [string, string, string]> = []
|
|
185
|
+
const sheets: Array<SheetJsInputSheet> = []
|
|
186
|
+
const included = new Set<string>()
|
|
187
|
+
|
|
188
|
+
for (const [position, sheet] of model.sheets.entries()) {
|
|
189
|
+
const name = `worksheets/sheet${position + 1}.xml`
|
|
190
|
+
|
|
191
|
+
sheetRelationships.push([
|
|
192
|
+
`rId${position + 1}`,
|
|
193
|
+
`${officeDocumentRelationshipsNamespace}/worksheet`,
|
|
194
|
+
name
|
|
195
|
+
])
|
|
196
|
+
|
|
197
|
+
const relationship =
|
|
198
|
+
sheet.relationshipId === undefined
|
|
199
|
+
? undefined
|
|
200
|
+
: workbookRelationships.get(sheet.relationshipId)
|
|
201
|
+
|
|
202
|
+
const part =
|
|
203
|
+
relationship === undefined || relationship.external || !hasKind(relationship, 'worksheet')
|
|
204
|
+
? undefined
|
|
205
|
+
: xmlPart(index, resolvePartPath(workbookDirectory, relationship.target))
|
|
206
|
+
|
|
207
|
+
if (part === undefined) {
|
|
208
|
+
sheets.push({ name: sheet.name, part: undefined })
|
|
209
|
+
continue
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
// One part per sheet: SheetJS would parse and keep a shared part once per declaration.
|
|
213
|
+
if (included.has(part.name)) throw malformed()
|
|
214
|
+
|
|
215
|
+
included.add(part.name)
|
|
216
|
+
files[`xl/${name}`] = withoutCdata(part.bytes)
|
|
217
|
+
overrides.push([`xl/${name}`, contentType.worksheet])
|
|
218
|
+
sheets.push({ name: sheet.name, part: part.name })
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
const sharedStrings = relatedXmlPart(
|
|
222
|
+
index,
|
|
223
|
+
workbookRelationships,
|
|
224
|
+
workbookDirectory,
|
|
225
|
+
relationship => hasKind(relationship, 'sharedStrings'),
|
|
226
|
+
'xl/sharedStrings.xml'
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
if (sharedStrings !== undefined) {
|
|
230
|
+
files['xl/sharedStrings.xml'] = withoutCdata(sharedStrings.bytes)
|
|
231
|
+
overrides.push(['xl/sharedStrings.xml', contentType.sharedStrings])
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
const styles = relatedXmlPart(
|
|
235
|
+
index,
|
|
236
|
+
workbookRelationships,
|
|
237
|
+
workbookDirectory,
|
|
238
|
+
relationship => hasKind(relationship, 'styles'),
|
|
239
|
+
'xl/styles.xml'
|
|
240
|
+
)
|
|
241
|
+
|
|
242
|
+
if (styles !== undefined) {
|
|
243
|
+
files['xl/styles.xml'] = strToU8(stylesXml(readStyles(styles.bytes)))
|
|
244
|
+
overrides.push(['xl/styles.xml', contentType.styles])
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
const core = relatedXmlPart(
|
|
248
|
+
index,
|
|
249
|
+
relationships(utf8Text(index.find('_rels/.rels')?.bytes)),
|
|
250
|
+
'',
|
|
251
|
+
relationship => relationship.type !== undefined && corePropertiesTypes.has(relationship.type),
|
|
252
|
+
'docProps/core.xml'
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
const archive = {
|
|
256
|
+
'[Content_Types].xml': strToU8(
|
|
257
|
+
contentTypesXml([[workbookPartPath, contentType.workbook], ...overrides])
|
|
258
|
+
),
|
|
259
|
+
'_rels/.rels': strToU8(
|
|
260
|
+
relationshipsXml([
|
|
261
|
+
['rId1', `${officeDocumentRelationshipsNamespace}/officeDocument`, workbookPartPath]
|
|
262
|
+
])
|
|
263
|
+
),
|
|
264
|
+
[workbookPartPath]: strToU8(workbookXml(model)),
|
|
265
|
+
'xl/_rels/workbook.xml.rels': strToU8(relationshipsXml(sheetRelationships)),
|
|
266
|
+
...files
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
return {
|
|
270
|
+
archive: zipSync(archive, { level: 0 }),
|
|
271
|
+
names: Object.keys(archive),
|
|
272
|
+
sheets,
|
|
273
|
+
title: coreTitle(core?.bytes)
|
|
274
|
+
}
|
|
275
|
+
}
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
import { Buffer } from 'node:buffer'
|
|
2
|
+
import {
|
|
3
|
+
sheetJsAttributeEscape,
|
|
4
|
+
sheetJsAttributeText,
|
|
5
|
+
sheetJsTags,
|
|
6
|
+
sheetJsXmlHeader,
|
|
7
|
+
spreadsheetMainNamespace,
|
|
8
|
+
stripSheetJsNamespace
|
|
9
|
+
} from './sheetjs-xml.ts'
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* `xl/styles.xml` is never handed to SheetJS as uploaded. With `cellText: true`, SheetJS formats
|
|
13
|
+
* every styled cell by re-parsing its number format (`SSF_format`, work proportional to the
|
|
14
|
+
* format's length, a quoted literal built one character at a time), so one huge format reused by
|
|
15
|
+
* many cells amplifies before any budget runs. The extractor writes a minimal stylesheet with
|
|
16
|
+
* only what display text needs: custom number formats (`numFmtId`, `formatCode`) of at most
|
|
17
|
+
* `maxNumberFormatCharacters`, and one `<xf numFmtId>` per source cell format, in order, so cell
|
|
18
|
+
* `s` indexes keep their meaning. Fonts, fills, borders, cell styles, and dxfs are left out:
|
|
19
|
+
* SheetJS parses fonts, fills, and borders whatever the options but uses them only with
|
|
20
|
+
* `cellStyles`, so leaving them out also removes their parsers from the input.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
/** Excel's own limit on a number format code, counted after unescaping. */
|
|
24
|
+
export const maxNumberFormatCharacters = 255
|
|
25
|
+
|
|
26
|
+
/** Custom number formats kept; later ones are dropped (their cells show General). */
|
|
27
|
+
export const maxCustomNumberFormats = 1000
|
|
28
|
+
|
|
29
|
+
/** Excel's limit on cell formats (`cellXfs`); later ones are dropped (General). */
|
|
30
|
+
export const maxCellFormats = 64_000
|
|
31
|
+
|
|
32
|
+
/** SheetJS `str_remove_ng(text, '<!--', '-->')`, including its unterminated-comment behavior. */
|
|
33
|
+
const removeComments = (text: string) => {
|
|
34
|
+
let start = text.indexOf('<!--')
|
|
35
|
+
|
|
36
|
+
if (start === -1) return text
|
|
37
|
+
|
|
38
|
+
const output: Array<string> = []
|
|
39
|
+
let last = 0
|
|
40
|
+
|
|
41
|
+
while (start > -1) {
|
|
42
|
+
output.push(text.slice(last, start))
|
|
43
|
+
|
|
44
|
+
const end = text.indexOf('-->', start + 4)
|
|
45
|
+
|
|
46
|
+
if (end === -1) break
|
|
47
|
+
|
|
48
|
+
last = end + 3
|
|
49
|
+
start = text.indexOf('<!--', last)
|
|
50
|
+
|
|
51
|
+
if (start === -1) output.push(text.slice(last))
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
return output.join('')
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/** SheetJS `remove_doctype`. */
|
|
58
|
+
const removeDoctype = (text: string) => {
|
|
59
|
+
const doctype = text.slice(0, 1024).indexOf('<!DOCTYPE')
|
|
60
|
+
|
|
61
|
+
if (doctype === -1) return text
|
|
62
|
+
|
|
63
|
+
const element = /<\w/.exec(text)
|
|
64
|
+
|
|
65
|
+
return element === null ? text : text.slice(0, doctype) + text.slice(element.index)
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** SheetJS `str_match_xml_ns`: the first `<tag>` start tag through the next `</tag>` (any prefix). */
|
|
69
|
+
const sheetJsRegion = (text: string, tag: string) => {
|
|
70
|
+
const start = new RegExp(`<(?:\\w+:)?${tag}\\b[^<>]*>`, 'g')
|
|
71
|
+
const end = new RegExp(`</(?:\\w+:)?${tag}>`, 'g')
|
|
72
|
+
const open = start.exec(text)
|
|
73
|
+
|
|
74
|
+
if (open === null) return undefined
|
|
75
|
+
|
|
76
|
+
end.lastIndex = start.lastIndex
|
|
77
|
+
|
|
78
|
+
return end.exec(text) === null ? undefined : text.slice(open.index, end.lastIndex)
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
const numberFormatId = (raw: string | undefined) => {
|
|
82
|
+
const id = Number.parseInt(raw ?? '', 10)
|
|
83
|
+
|
|
84
|
+
return Number.isSafeInteger(id) && id >= 0 ? id : undefined
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export type StylesSummary = {
|
|
88
|
+
/** Custom number formats in document order, as `[numFmtId, formatCode]`. */
|
|
89
|
+
readonly numberFormats: ReadonlyArray<readonly [number, string]>
|
|
90
|
+
/** The `numFmtId` of each cell format in order, or `undefined` without a `cellXfs` element. */
|
|
91
|
+
readonly cellFormats: ReadonlyArray<number> | undefined
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Read the number formats and cell formats SheetJS would read from a stylesheet (same comment and
|
|
96
|
+
* doctype removal, same regions, same tag grammar), bounded by the caps above. A format longer
|
|
97
|
+
* than `maxNumberFormatCharacters` after unescaping, holding CDATA, or with an invalid id is
|
|
98
|
+
* dropped; a cell format without a valid `numFmtId` becomes General (0).
|
|
99
|
+
*/
|
|
100
|
+
export const readStyles = (content: Uint8Array): StylesSummary => {
|
|
101
|
+
const text = removeDoctype(
|
|
102
|
+
removeComments(
|
|
103
|
+
Buffer.from(content.buffer, content.byteOffset, content.byteLength).toString('latin1')
|
|
104
|
+
)
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
const numberFormats: Array<readonly [number, string]> = []
|
|
108
|
+
const formatsRegion = sheetJsRegion(text, 'numFmts')
|
|
109
|
+
|
|
110
|
+
if (formatsRegion !== undefined) {
|
|
111
|
+
for (const tag of sheetJsTags(formatsRegion)) {
|
|
112
|
+
if (numberFormats.length >= maxCustomNumberFormats) break
|
|
113
|
+
|
|
114
|
+
if (stripSheetJsNamespace(tag.head) !== '<numFmt') continue
|
|
115
|
+
|
|
116
|
+
const id = numberFormatId(tag.attributes.get('numFmtId'))
|
|
117
|
+
const raw = tag.attributes.get('formatCode')
|
|
118
|
+
const code = raw === undefined ? undefined : sheetJsAttributeText(raw)
|
|
119
|
+
|
|
120
|
+
if (id !== undefined && code !== undefined && code.length <= maxNumberFormatCharacters)
|
|
121
|
+
numberFormats.push([id, code])
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const cellFormatsRegion = sheetJsRegion(text, 'cellXfs')
|
|
126
|
+
|
|
127
|
+
if (cellFormatsRegion === undefined) return { numberFormats, cellFormats: undefined }
|
|
128
|
+
|
|
129
|
+
const cellFormats: Array<number> = []
|
|
130
|
+
|
|
131
|
+
for (const tag of sheetJsTags(cellFormatsRegion)) {
|
|
132
|
+
if (cellFormats.length >= maxCellFormats) break
|
|
133
|
+
|
|
134
|
+
const head = stripSheetJsNamespace(tag.head)
|
|
135
|
+
|
|
136
|
+
if (head === '<xf' || head === '<xf/>' || head === '<xf>')
|
|
137
|
+
cellFormats.push(numberFormatId(tag.attributes.get('numFmtId')) ?? 0)
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
return { numberFormats, cellFormats }
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** The generated stylesheet SheetJS reads instead of the uploaded one. */
|
|
144
|
+
export const stylesXml = ({ numberFormats, cellFormats }: StylesSummary) =>
|
|
145
|
+
`${sheetJsXmlHeader}<styleSheet xmlns="${spreadsheetMainNamespace}">${
|
|
146
|
+
numberFormats.length === 0
|
|
147
|
+
? ''
|
|
148
|
+
: `<numFmts count="${numberFormats.length}">${numberFormats
|
|
149
|
+
.map(
|
|
150
|
+
([id, code]) =>
|
|
151
|
+
`<numFmt numFmtId="${id}" formatCode="${sheetJsAttributeEscape(code)}"/>`
|
|
152
|
+
)
|
|
153
|
+
.join('')}</numFmts>`
|
|
154
|
+
}${
|
|
155
|
+
cellFormats === undefined
|
|
156
|
+
? ''
|
|
157
|
+
: `<cellXfs count="${cellFormats.length}">${cellFormats
|
|
158
|
+
.map(id => `<xf numFmtId="${id}"/>`)
|
|
159
|
+
.join('')}</cellXfs>`
|
|
160
|
+
}</styleSheet>`
|