@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
import { Predicate } from 'effect'
|
|
2
|
+
import { FileExtractionError } from '../errors.ts'
|
|
3
|
+
import type { FileExtractorLimits } from '../limits.ts'
|
|
4
|
+
import { makeHyperlinkLookup } from './xlsx-hyperlinks.ts'
|
|
5
|
+
import type { HyperlinkLookup, XlsxHyperlinks } from './xlsx-hyperlinks.ts'
|
|
6
|
+
import type { CellRange } from './xlsx-range.ts'
|
|
7
|
+
import { cellCount, encodeCellAddress, parseCellRange } from './xlsx-range.ts'
|
|
8
|
+
|
|
9
|
+
export type XlsxTextLimits = Pick<
|
|
10
|
+
FileExtractorLimits,
|
|
11
|
+
'maxXlsxCellVisits' | 'maxXlsxSheets' | 'maxXlsxTextCharacters'
|
|
12
|
+
>
|
|
13
|
+
|
|
14
|
+
/** The parsed workbook as SheetJS returns it; sheets and cells are read as own properties. */
|
|
15
|
+
export type XlsxWorkbook = {
|
|
16
|
+
readonly SheetNames: ReadonlyArray<string>
|
|
17
|
+
readonly Sheets: object
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export type XlsxTextOptions = {
|
|
21
|
+
readonly hyperlinks?: XlsxHyperlinks
|
|
22
|
+
/** Some hyperlinks were never read because the workbook exceeded `maxXlsxHyperlinks`. */
|
|
23
|
+
readonly hyperlinksTruncated?: boolean
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
const outputLimitReason = 'output limit'
|
|
27
|
+
|
|
28
|
+
const hyperlinkLimitReason = 'hyperlink limit'
|
|
29
|
+
|
|
30
|
+
const omittedHyperlinksMarker = (reasons: ReadonlyArray<string>) =>
|
|
31
|
+
`\n\n[Some hyperlinks omitted: ${reasons.join(' and ')}]`
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Space reserved for the marker whenever a workbook has hyperlinks, so it always fits.
|
|
35
|
+
* `maxXlsxTextCharacters` must exceed it (`minimumXlsxTextCharacters`).
|
|
36
|
+
*/
|
|
37
|
+
export const omittedHyperlinksMarkerReserve = omittedHyperlinksMarker([
|
|
38
|
+
outputLimitReason,
|
|
39
|
+
hyperlinkLimitReason
|
|
40
|
+
]).length
|
|
41
|
+
|
|
42
|
+
const invalidRange = () =>
|
|
43
|
+
new FileExtractionError({ format: 'xlsx', message: 'Invalid XLSX worksheet range.' })
|
|
44
|
+
|
|
45
|
+
const tooLarge = () =>
|
|
46
|
+
new FileExtractionError({
|
|
47
|
+
format: 'xlsx',
|
|
48
|
+
message: 'XLSX exceeds the worksheet or cell-visit limit.'
|
|
49
|
+
})
|
|
50
|
+
|
|
51
|
+
const outputTooLarge = () =>
|
|
52
|
+
new FileExtractionError({
|
|
53
|
+
format: 'xlsx',
|
|
54
|
+
message: 'XLSX extracted text exceeds the output limit.'
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
/** Read an own property without walking the prototype chain (or trusting a polluted one). */
|
|
58
|
+
const ownProperty = (target: object, key: string): unknown => {
|
|
59
|
+
const descriptor = Object.getOwnPropertyDescriptor(target, key)
|
|
60
|
+
|
|
61
|
+
if (descriptor === undefined) return undefined
|
|
62
|
+
|
|
63
|
+
return 'value' in descriptor ? descriptor.value : descriptor.get?.call(target)
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const cellText = (cell: object): string => {
|
|
67
|
+
if (ownProperty(cell, 't') === 'z') return ''
|
|
68
|
+
|
|
69
|
+
const value = ownProperty(cell, 'v')
|
|
70
|
+
|
|
71
|
+
// SheetJS runs with `cellFormula: false`: only cached values exist, formula-only cells are empty.
|
|
72
|
+
if (value === undefined || value === null) return ''
|
|
73
|
+
|
|
74
|
+
// Prefer parser-provided display text. Do not run an untrusted format template here.
|
|
75
|
+
const display = ownProperty(cell, 'w')
|
|
76
|
+
|
|
77
|
+
if (Predicate.isString(display)) return display
|
|
78
|
+
|
|
79
|
+
if (Predicate.isString(value)) return value
|
|
80
|
+
|
|
81
|
+
if (Predicate.isBoolean(value)) return value ? 'TRUE' : 'FALSE'
|
|
82
|
+
|
|
83
|
+
if (value instanceof Date && Number.isFinite(value.getTime())) return value.toISOString()
|
|
84
|
+
|
|
85
|
+
if (Predicate.isNumber(value) && Number.isFinite(value)) return String(value)
|
|
86
|
+
|
|
87
|
+
throw new FileExtractionError({ format: 'xlsx', message: 'Invalid XLSX cell value.' })
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* CSV length of `text` (quotes doubled, wrapped when needed). The scan stops once the length
|
|
92
|
+
* exceeds `limit`; the returned length is then only known to be above it.
|
|
93
|
+
*/
|
|
94
|
+
const csvField = (text: string, limit = Number.POSITIVE_INFINITY) => {
|
|
95
|
+
let quoteCount = 0
|
|
96
|
+
let quote = text === 'ID'
|
|
97
|
+
|
|
98
|
+
for (let index = 0; index < text.length; index += 1) {
|
|
99
|
+
const character = text.charCodeAt(index)
|
|
100
|
+
|
|
101
|
+
// '"', ',', '\n', '\r'
|
|
102
|
+
if (character === 34) quoteCount += 1
|
|
103
|
+
|
|
104
|
+
if (character === 34 || character === 44 || character === 10 || character === 13) {
|
|
105
|
+
quote = true
|
|
106
|
+
|
|
107
|
+
if (text.length + quoteCount + 2 > limit) break
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
return { quote, length: text.length + quoteCount + (quote ? 2 : 0) }
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
type VisitedSheet = {
|
|
115
|
+
readonly name: string
|
|
116
|
+
readonly sheet: object | undefined
|
|
117
|
+
readonly range: CellRange | undefined
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** Annotated text for a cell, or `undefined` to write it plain. */
|
|
121
|
+
type CellDecorator = (input: {
|
|
122
|
+
readonly sheet: VisitedSheet
|
|
123
|
+
readonly row: number
|
|
124
|
+
readonly column: number
|
|
125
|
+
readonly text: string
|
|
126
|
+
}) => string | undefined
|
|
127
|
+
|
|
128
|
+
/** Write bounded CSV for every sheet; throws once `budget` characters would be exceeded. */
|
|
129
|
+
const renderSheets = (
|
|
130
|
+
sheets: ReadonlyArray<VisitedSheet>,
|
|
131
|
+
budget: number,
|
|
132
|
+
decorate?: CellDecorator
|
|
133
|
+
) => {
|
|
134
|
+
let characters = 0
|
|
135
|
+
const output: Array<string> = []
|
|
136
|
+
|
|
137
|
+
const append = (text: string) => {
|
|
138
|
+
if (text.length > budget - characters) throw outputTooLarge()
|
|
139
|
+
|
|
140
|
+
characters += text.length
|
|
141
|
+
output.push(text)
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
const appendCell = (text: string) => {
|
|
145
|
+
if (text.length > budget - characters) throw outputTooLarge()
|
|
146
|
+
|
|
147
|
+
const field = csvField(text, budget - characters)
|
|
148
|
+
|
|
149
|
+
if (field.length > budget - characters) throw outputTooLarge()
|
|
150
|
+
|
|
151
|
+
append(field.quote ? `"${text.replaceAll('"', '""')}"` : text)
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
for (const visited of sheets) {
|
|
155
|
+
const { name, sheet, range } = visited
|
|
156
|
+
|
|
157
|
+
if (sheet === undefined) continue
|
|
158
|
+
|
|
159
|
+
if (output.length > 0) append('\n\n')
|
|
160
|
+
|
|
161
|
+
append('# ')
|
|
162
|
+
append(name)
|
|
163
|
+
append('\n')
|
|
164
|
+
|
|
165
|
+
if (range === undefined) continue
|
|
166
|
+
|
|
167
|
+
for (let row = range.start.r; row <= range.end.r; row += 1) {
|
|
168
|
+
if (row > range.start.r) append('\n')
|
|
169
|
+
|
|
170
|
+
for (let column = range.start.c; column <= range.end.c; column += 1) {
|
|
171
|
+
if (column > range.start.c) append(',')
|
|
172
|
+
|
|
173
|
+
const cell = ownProperty(sheet, encodeCellAddress({ r: row, c: column }))
|
|
174
|
+
|
|
175
|
+
if (!Predicate.isObject(cell)) {
|
|
176
|
+
appendCell('')
|
|
177
|
+
continue
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
const text = cellText(cell)
|
|
181
|
+
|
|
182
|
+
appendCell(decorate?.({ sheet: visited, row, column, text }) ?? text)
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
return output.join('')
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* Validate every sheet before touching any cell, then generate bounded CSV incrementally, never
|
|
192
|
+
* with `sheet_to_csv` (which can walk billions of absent cells or allocate an unbounded quoted
|
|
193
|
+
* string).
|
|
194
|
+
*
|
|
195
|
+
* Existing cells inside an external hyperlink are written as `text <url>`. Plain text is
|
|
196
|
+
* rendered first, so links only use budget the plain text leaves over: in document order, an
|
|
197
|
+
* annotation that no longer fits leaves its cell plain and one marker ends the output.
|
|
198
|
+
*/
|
|
199
|
+
export const extractBoundedXlsxText = (
|
|
200
|
+
workbook: XlsxWorkbook,
|
|
201
|
+
limits: XlsxTextLimits,
|
|
202
|
+
options: XlsxTextOptions = {}
|
|
203
|
+
) => {
|
|
204
|
+
if (workbook.SheetNames.length > limits.maxXlsxSheets) throw tooLarge()
|
|
205
|
+
|
|
206
|
+
let visits = 0
|
|
207
|
+
|
|
208
|
+
const sheets = workbook.SheetNames.map((name): VisitedSheet => {
|
|
209
|
+
const candidate = ownProperty(workbook.Sheets, name)
|
|
210
|
+
const sheet = Predicate.isObject(candidate) ? candidate : undefined
|
|
211
|
+
const ref = sheet === undefined ? undefined : ownProperty(sheet, '!ref')
|
|
212
|
+
|
|
213
|
+
if (ref !== undefined && !Predicate.isString(ref)) throw invalidRange()
|
|
214
|
+
|
|
215
|
+
const range = ref === undefined ? undefined : parseCellRange(ref)
|
|
216
|
+
|
|
217
|
+
if (ref !== undefined && range === undefined) throw invalidRange()
|
|
218
|
+
|
|
219
|
+
visits += range === undefined ? 0 : cellCount(range)
|
|
220
|
+
|
|
221
|
+
if (!Number.isSafeInteger(visits) || visits > limits.maxXlsxCellVisits) throw tooLarge()
|
|
222
|
+
|
|
223
|
+
return { name, sheet, range }
|
|
224
|
+
})
|
|
225
|
+
|
|
226
|
+
const links = options.hyperlinks ?? new Map()
|
|
227
|
+
const truncated = options.hyperlinksTruncated === true
|
|
228
|
+
|
|
229
|
+
if (links.size === 0 && !truncated) return renderSheets(sheets, limits.maxXlsxTextCharacters)
|
|
230
|
+
|
|
231
|
+
// Reserve the marker's space so it always fits inside the character limit.
|
|
232
|
+
const budget = limits.maxXlsxTextCharacters - omittedHyperlinksMarkerReserve
|
|
233
|
+
let spare = budget - renderSheets(sheets, budget).length
|
|
234
|
+
let dropped = false
|
|
235
|
+
const lookups = new Map<string, HyperlinkLookup>()
|
|
236
|
+
|
|
237
|
+
const lookupFor = (sheet: VisitedSheet) => {
|
|
238
|
+
const sheetLinks = links.get(sheet.name)
|
|
239
|
+
|
|
240
|
+
if (sheetLinks === undefined || sheet.range === undefined) return undefined
|
|
241
|
+
|
|
242
|
+
const existing = lookups.get(sheet.name)
|
|
243
|
+
|
|
244
|
+
if (existing !== undefined) return existing
|
|
245
|
+
|
|
246
|
+
const lookup = makeHyperlinkLookup(sheetLinks, sheet.range)
|
|
247
|
+
lookups.set(sheet.name, lookup)
|
|
248
|
+
|
|
249
|
+
return lookup
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
const output = renderSheets(sheets, budget, ({ sheet, row, column, text }) => {
|
|
253
|
+
const link = lookupFor(sheet)?.at(row, column)
|
|
254
|
+
|
|
255
|
+
if (link === undefined || link.target === text) return undefined
|
|
256
|
+
|
|
257
|
+
const label = text.length > 0 ? text : (link.display ?? '')
|
|
258
|
+
const plainLength = csvField(text).length
|
|
259
|
+
const annotatedLength = label.length + (label.length > 0 ? 3 : 2) + link.target.length
|
|
260
|
+
|
|
261
|
+
// Constant-time bound first (CSV escaping only adds), then a scan capped at the budget.
|
|
262
|
+
if (annotatedLength - plainLength > spare) {
|
|
263
|
+
dropped = true
|
|
264
|
+
|
|
265
|
+
return undefined
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
const annotated = label.length > 0 ? `${label} <${link.target}>` : `<${link.target}>`
|
|
269
|
+
const extra = csvField(annotated, plainLength + spare).length - plainLength
|
|
270
|
+
|
|
271
|
+
if (extra > spare) {
|
|
272
|
+
dropped = true
|
|
273
|
+
|
|
274
|
+
return undefined
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
spare -= extra
|
|
278
|
+
|
|
279
|
+
return annotated
|
|
280
|
+
})
|
|
281
|
+
|
|
282
|
+
const reasons = [
|
|
283
|
+
...(dropped ? [outputLimitReason] : []),
|
|
284
|
+
...(truncated ? [hyperlinkLimitReason] : [])
|
|
285
|
+
]
|
|
286
|
+
|
|
287
|
+
return reasons.length === 0 ? output : output + omittedHyperlinksMarker(reasons)
|
|
288
|
+
}
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
import { Buffer } from 'node:buffer'
|
|
2
|
+
import { FileExtractionError } from '../errors.ts'
|
|
3
|
+
import {
|
|
4
|
+
sheetJsAttributeEscape,
|
|
5
|
+
sheetJsAttributeText,
|
|
6
|
+
sheetJsTags,
|
|
7
|
+
stripSheetJsNamespace,
|
|
8
|
+
sheetJsXmlHeader,
|
|
9
|
+
spreadsheetMainNamespace,
|
|
10
|
+
officeDocumentRelationshipsNamespace
|
|
11
|
+
} from './sheetjs-xml.ts'
|
|
12
|
+
import type { SheetJsTag } from './sheetjs-xml.ts'
|
|
13
|
+
import { decodeXmlEntities, prefixedAttribute, rawXmlAttributes } from './xml-text.ts'
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* `xl/workbook.xml` is never handed to SheetJS as uploaded. SheetJS's workbook parser slices and
|
|
17
|
+
* decodes the whole prefix of the part at every `</definedName>` (quadratic), and walks every
|
|
18
|
+
* `<sheet>` its own tag pattern finds, resolving each through its `r:id`, so one worksheet can be
|
|
19
|
+
* parsed once per declaration. The extractor reads the sheet list itself, checks that its strict
|
|
20
|
+
* scan and SheetJS's tag grammar see the same sheets, and writes a minimal workbook: the sheets in
|
|
21
|
+
* order (name, `sheetId`, hidden state, a fresh `r:id`) and the 1904 date system.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
export type WorkbookSheetDeclaration = {
|
|
25
|
+
/** The sheet name exactly as SheetJS reads it (`unescapexml(utf8read(name))`). */
|
|
26
|
+
readonly name: string
|
|
27
|
+
readonly sheetId: string
|
|
28
|
+
readonly state: 'hidden' | 'veryHidden' | undefined
|
|
29
|
+
/** The `r:id` of the sheet's relationship in the uploaded workbook, entity-decoded. */
|
|
30
|
+
readonly relationshipId: string | undefined
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export type WorkbookModel = {
|
|
34
|
+
readonly sheets: ReadonlyArray<WorkbookSheetDeclaration>
|
|
35
|
+
readonly date1904: boolean
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export const malformedWorkbookMessage = 'XLSX workbook is malformed.'
|
|
39
|
+
|
|
40
|
+
const malformed = () =>
|
|
41
|
+
new FileExtractionError({ format: 'xlsx', message: malformedWorkbookMessage })
|
|
42
|
+
|
|
43
|
+
const tooManySheets = () =>
|
|
44
|
+
new FileExtractionError({
|
|
45
|
+
format: 'xlsx',
|
|
46
|
+
message: 'XLSX exceeds the worksheet or cell-visit limit.'
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
// `[^<>]` keeps each candidate inside one tag, so the strict scan stays linear.
|
|
50
|
+
const strictSheetTag = /<(?:[\w.-]+:)?sheet(?=[\s/>])[^<>]*>/g
|
|
51
|
+
|
|
52
|
+
const plainSheetId = /^[1-9]\d{0,8}$/
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* The sheets of `xl/workbook.xml` and its date system. Throws `FileExtractionError` when the
|
|
56
|
+
* workbook declares more than `maxSheets` sheets (counted by either scan), when the strict scan
|
|
57
|
+
* and SheetJS's grammar disagree on the number or names of sheets, when a name is missing, holds
|
|
58
|
+
* CDATA, or repeats another ignoring case, or when there are no sheets.
|
|
59
|
+
*/
|
|
60
|
+
export const readWorkbookModel = (bytes: Uint8Array, maxSheets: number): WorkbookModel => {
|
|
61
|
+
// SheetJS parses the Latin-1 ("binary") view and decodes attribute text as UTF-8 afterwards.
|
|
62
|
+
const text = Buffer.from(bytes.buffer, bytes.byteOffset, bytes.byteLength).toString('latin1')
|
|
63
|
+
const sheetJsSheets: Array<SheetJsTag> = []
|
|
64
|
+
let date1904 = false
|
|
65
|
+
|
|
66
|
+
for (const tag of sheetJsTags(text)) {
|
|
67
|
+
const head = stripSheetJsNamespace(tag.head)
|
|
68
|
+
|
|
69
|
+
if (head === '<sheet') {
|
|
70
|
+
sheetJsSheets.push(tag)
|
|
71
|
+
|
|
72
|
+
if (sheetJsSheets.length > maxSheets) throw tooManySheets()
|
|
73
|
+
} else if (head === '<workbookPr' || head === '<workbookPr/>') {
|
|
74
|
+
// Each `workbookPr` that carries the attribute overrides the last (SheetJS `parsexmlbool`).
|
|
75
|
+
const value = tag.attributes.get('date1904')
|
|
76
|
+
|
|
77
|
+
if (value !== undefined) date1904 = value === '1' || value === 'true'
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
const strictSheets: Array<ReadonlyMap<string, string>> = []
|
|
82
|
+
|
|
83
|
+
for (const [tag] of text.matchAll(strictSheetTag)) {
|
|
84
|
+
strictSheets.push(rawXmlAttributes(tag))
|
|
85
|
+
|
|
86
|
+
if (strictSheets.length > maxSheets) throw tooManySheets()
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
if (sheetJsSheets.length === 0 || sheetJsSheets.length !== strictSheets.length) throw malformed()
|
|
90
|
+
|
|
91
|
+
const seen = new Set<string>()
|
|
92
|
+
|
|
93
|
+
const sheets = sheetJsSheets.map((tag, index): WorkbookSheetDeclaration => {
|
|
94
|
+
const strict = strictSheets[index]
|
|
95
|
+
const rawName = tag.attributes.get('name')
|
|
96
|
+
|
|
97
|
+
if (strict === undefined || rawName === undefined || strict.get('name') !== rawName)
|
|
98
|
+
throw malformed()
|
|
99
|
+
|
|
100
|
+
const name = sheetJsAttributeText(rawName)
|
|
101
|
+
|
|
102
|
+
// Excel compares sheet names ignoring case.
|
|
103
|
+
if (name === undefined || seen.has(name.toLowerCase())) throw malformed()
|
|
104
|
+
|
|
105
|
+
seen.add(name.toLowerCase())
|
|
106
|
+
|
|
107
|
+
const state = tag.attributes.get('state')
|
|
108
|
+
const sheetId = tag.attributes.get('sheetId')
|
|
109
|
+
const relationshipId = prefixedAttribute(strict, 'id')
|
|
110
|
+
|
|
111
|
+
return {
|
|
112
|
+
name,
|
|
113
|
+
sheetId: sheetId !== undefined && plainSheetId.test(sheetId) ? sheetId : `${index + 1}`,
|
|
114
|
+
state: state === 'hidden' || state === 'veryHidden' ? state : undefined,
|
|
115
|
+
relationshipId: relationshipId === undefined ? undefined : decodeXmlEntities(relationshipId)
|
|
116
|
+
}
|
|
117
|
+
})
|
|
118
|
+
|
|
119
|
+
return { sheets, date1904 }
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** The generated workbook: sheet `n` has `r:id="rId<n>"`. */
|
|
123
|
+
export const workbookXml = (model: WorkbookModel) =>
|
|
124
|
+
`${sheetJsXmlHeader}<workbook xmlns="${spreadsheetMainNamespace}" xmlns:r="${officeDocumentRelationshipsNamespace}">${
|
|
125
|
+
model.date1904 ? '<workbookPr date1904="1"/>' : ''
|
|
126
|
+
}<sheets>${model.sheets
|
|
127
|
+
.map(
|
|
128
|
+
(sheet, index) =>
|
|
129
|
+
`<sheet name="${sheetJsAttributeEscape(sheet.name)}" sheetId="${sheet.sheetId}"${
|
|
130
|
+
sheet.state === undefined ? '' : ` state="${sheet.state}"`
|
|
131
|
+
} r:id="rId${index + 1}"/>`
|
|
132
|
+
)
|
|
133
|
+
.join('')}</sheets></workbook>`
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
const xmlEntity = /&([^;&\s]{1,16});/g
|
|
2
|
+
|
|
3
|
+
const hexEntity = /^#x([0-9a-fA-F]+)$/
|
|
4
|
+
|
|
5
|
+
const decimalEntity = /^#(\d+)$/
|
|
6
|
+
|
|
7
|
+
const namedEntities: ReadonlyMap<string, string> = new Map([
|
|
8
|
+
['amp', '&'],
|
|
9
|
+
['lt', '<'],
|
|
10
|
+
['gt', '>'],
|
|
11
|
+
['quot', '"'],
|
|
12
|
+
['apos', "'"]
|
|
13
|
+
])
|
|
14
|
+
|
|
15
|
+
const decodeCodePoint = (raw: string, codePointText: string, radix: number) => {
|
|
16
|
+
const codePoint = Number.parseInt(codePointText, radix)
|
|
17
|
+
|
|
18
|
+
if (!Number.isInteger(codePoint) || codePoint < 0 || codePoint > 0x10ffff) return raw
|
|
19
|
+
|
|
20
|
+
return String.fromCodePoint(codePoint)
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
const decodeXmlEntity = (raw: string, entity: string) => {
|
|
24
|
+
const named = namedEntities.get(entity)
|
|
25
|
+
|
|
26
|
+
if (named !== undefined) return named
|
|
27
|
+
|
|
28
|
+
const hex = hexEntity.exec(entity)?.[1]
|
|
29
|
+
|
|
30
|
+
if (hex !== undefined) return decodeCodePoint(raw, hex, 16)
|
|
31
|
+
|
|
32
|
+
const decimal = decimalEntity.exec(entity)?.[1]
|
|
33
|
+
|
|
34
|
+
if (decimal !== undefined) return decodeCodePoint(raw, decimal, 10)
|
|
35
|
+
|
|
36
|
+
return raw
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** Decode the predefined XML entities and numeric character references; keep unknown ones. */
|
|
40
|
+
export const decodeXmlEntities = (text: string) =>
|
|
41
|
+
text.replace(xmlEntity, (raw, entity: string) => decodeXmlEntity(raw, entity))
|
|
42
|
+
|
|
43
|
+
// A name may only start after a non-name character, so a long run of name characters is scanned
|
|
44
|
+
// once rather than once per starting position.
|
|
45
|
+
const attributePattern = /(?<![\w.:-])([\w.:-]+)\s*=\s*(?:"([^"]*)"|'([^']*)')/g
|
|
46
|
+
|
|
47
|
+
/** Attributes of one start tag as written (not decoded), keyed by qualified name; first wins. */
|
|
48
|
+
export const rawXmlAttributes = (tag: string): ReadonlyMap<string, string> => {
|
|
49
|
+
const attributes = new Map<string, string>()
|
|
50
|
+
|
|
51
|
+
for (const match of tag.matchAll(attributePattern)) {
|
|
52
|
+
const name = match[1]
|
|
53
|
+
const value = match[2] ?? match[3]
|
|
54
|
+
|
|
55
|
+
if (name !== undefined && value !== undefined && !attributes.has(name))
|
|
56
|
+
attributes.set(name, value)
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
return attributes
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Attributes of one start tag, entity-decoded, keyed by their qualified name. */
|
|
63
|
+
export const xmlAttributes = (tag: string): ReadonlyMap<string, string> =>
|
|
64
|
+
new Map(
|
|
65
|
+
Array.from(rawXmlAttributes(tag), ([name, value]) => [name, decodeXmlEntities(value)] as const)
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
/** The value of a namespace-prefixed attribute such as `r:id`, whatever the prefix. */
|
|
69
|
+
export const prefixedAttribute = (attributes: ReadonlyMap<string, string>, localName: string) => {
|
|
70
|
+
for (const [name, value] of attributes) {
|
|
71
|
+
const separator = name.indexOf(':')
|
|
72
|
+
|
|
73
|
+
if (separator > 0 && name.slice(separator + 1) === localName) return value
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
return undefined
|
|
77
|
+
}
|
package/src/sanitize.ts
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
const nonPrintableCharacters = /[\u0000-\u0008\u000B\u000C\u000E-\u001F\u007F]/g
|
|
2
|
+
|
|
3
|
+
const longDotRuns = /\.{4,}/g
|
|
4
|
+
|
|
5
|
+
const horizontalWhitespaceRuns = /[\t ]{2,}/g
|
|
6
|
+
|
|
7
|
+
const blankLineRuns = /\n{3,}/g
|
|
8
|
+
|
|
9
|
+
/** Normalize line endings, drop control characters, and collapse layout noise. */
|
|
10
|
+
export const sanitizeExtractedText = (text: string) =>
|
|
11
|
+
text
|
|
12
|
+
.replaceAll('\r\n', '\n')
|
|
13
|
+
.replaceAll('\r', '\n')
|
|
14
|
+
.replace(nonPrintableCharacters, '')
|
|
15
|
+
.replace(longDotRuns, '…')
|
|
16
|
+
.replace(horizontalWhitespaceRuns, ' ')
|
|
17
|
+
.replace(blankLineRuns, '\n\n')
|
|
18
|
+
.trim()
|
package/src/service.ts
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import { Context } from 'effect'
|
|
2
|
+
import type { Effect } from 'effect'
|
|
3
|
+
import type { FileExtractorError } from './errors.ts'
|
|
4
|
+
import type { ExtractedFile, FileInput } from './format.ts'
|
|
5
|
+
|
|
6
|
+
export type FileExtractorApi = {
|
|
7
|
+
/**
|
|
8
|
+
* Extract sanitized text from one file. Fails with `UnsupportedFileFormatError` for unknown
|
|
9
|
+
* formats, `FileExtractionError` for unreadable, oversized, or empty files, and
|
|
10
|
+
* `SheetJsUnavailableError` when an XLSX file arrives but SheetJS 0.20.3+ is not installed.
|
|
11
|
+
*/
|
|
12
|
+
readonly extract: (input: FileInput) => Effect.Effect<ExtractedFile, FileExtractorError>
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* The file extractor service. The tag is runtime-portable; the Node implementation lives in
|
|
17
|
+
* `@yolk-sdk/extractors/node`.
|
|
18
|
+
*/
|
|
19
|
+
export class FileExtractor extends Context.Service<FileExtractor, FileExtractorApi>()(
|
|
20
|
+
'@yolk-sdk/extractors/FileExtractor'
|
|
21
|
+
) {}
|