@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
@@ -0,0 +1,288 @@
1
+ import { Predicate } from 'effect'
2
+ import { FileExtractionError } from '../errors.ts'
3
+ import type { FileExtractorLimits } from '../limits.ts'
4
+ import { makeHyperlinkLookup } from './xlsx-hyperlinks.ts'
5
+ import type { HyperlinkLookup, XlsxHyperlinks } from './xlsx-hyperlinks.ts'
6
+ import type { CellRange } from './xlsx-range.ts'
7
+ import { cellCount, encodeCellAddress, parseCellRange } from './xlsx-range.ts'
8
+
9
+ export type XlsxTextLimits = Pick<
10
+ FileExtractorLimits,
11
+ 'maxXlsxCellVisits' | 'maxXlsxSheets' | 'maxXlsxTextCharacters'
12
+ >
13
+
14
+ /** The parsed workbook as SheetJS returns it; sheets and cells are read as own properties. */
15
+ export type XlsxWorkbook = {
16
+ readonly SheetNames: ReadonlyArray<string>
17
+ readonly Sheets: object
18
+ }
19
+
20
+ export type XlsxTextOptions = {
21
+ readonly hyperlinks?: XlsxHyperlinks
22
+ /** Some hyperlinks were never read because the workbook exceeded `maxXlsxHyperlinks`. */
23
+ readonly hyperlinksTruncated?: boolean
24
+ }
25
+
26
+ const outputLimitReason = 'output limit'
27
+
28
+ const hyperlinkLimitReason = 'hyperlink limit'
29
+
30
+ const omittedHyperlinksMarker = (reasons: ReadonlyArray<string>) =>
31
+ `\n\n[Some hyperlinks omitted: ${reasons.join(' and ')}]`
32
+
33
+ /**
34
+ * Space reserved for the marker whenever a workbook has hyperlinks, so it always fits.
35
+ * `maxXlsxTextCharacters` must exceed it (`minimumXlsxTextCharacters`).
36
+ */
37
+ export const omittedHyperlinksMarkerReserve = omittedHyperlinksMarker([
38
+ outputLimitReason,
39
+ hyperlinkLimitReason
40
+ ]).length
41
+
42
+ const invalidRange = () =>
43
+ new FileExtractionError({ format: 'xlsx', message: 'Invalid XLSX worksheet range.' })
44
+
45
+ const tooLarge = () =>
46
+ new FileExtractionError({
47
+ format: 'xlsx',
48
+ message: 'XLSX exceeds the worksheet or cell-visit limit.'
49
+ })
50
+
51
+ const outputTooLarge = () =>
52
+ new FileExtractionError({
53
+ format: 'xlsx',
54
+ message: 'XLSX extracted text exceeds the output limit.'
55
+ })
56
+
57
+ /** Read an own property without walking the prototype chain (or trusting a polluted one). */
58
+ const ownProperty = (target: object, key: string): unknown => {
59
+ const descriptor = Object.getOwnPropertyDescriptor(target, key)
60
+
61
+ if (descriptor === undefined) return undefined
62
+
63
+ return 'value' in descriptor ? descriptor.value : descriptor.get?.call(target)
64
+ }
65
+
66
+ const cellText = (cell: object): string => {
67
+ if (ownProperty(cell, 't') === 'z') return ''
68
+
69
+ const value = ownProperty(cell, 'v')
70
+
71
+ // SheetJS runs with `cellFormula: false`: only cached values exist, formula-only cells are empty.
72
+ if (value === undefined || value === null) return ''
73
+
74
+ // Prefer parser-provided display text. Do not run an untrusted format template here.
75
+ const display = ownProperty(cell, 'w')
76
+
77
+ if (Predicate.isString(display)) return display
78
+
79
+ if (Predicate.isString(value)) return value
80
+
81
+ if (Predicate.isBoolean(value)) return value ? 'TRUE' : 'FALSE'
82
+
83
+ if (value instanceof Date && Number.isFinite(value.getTime())) return value.toISOString()
84
+
85
+ if (Predicate.isNumber(value) && Number.isFinite(value)) return String(value)
86
+
87
+ throw new FileExtractionError({ format: 'xlsx', message: 'Invalid XLSX cell value.' })
88
+ }
89
+
90
+ /**
91
+ * CSV length of `text` (quotes doubled, wrapped when needed). The scan stops once the length
92
+ * exceeds `limit`; the returned length is then only known to be above it.
93
+ */
94
+ const csvField = (text: string, limit = Number.POSITIVE_INFINITY) => {
95
+ let quoteCount = 0
96
+ let quote = text === 'ID'
97
+
98
+ for (let index = 0; index < text.length; index += 1) {
99
+ const character = text.charCodeAt(index)
100
+
101
+ // '"', ',', '\n', '\r'
102
+ if (character === 34) quoteCount += 1
103
+
104
+ if (character === 34 || character === 44 || character === 10 || character === 13) {
105
+ quote = true
106
+
107
+ if (text.length + quoteCount + 2 > limit) break
108
+ }
109
+ }
110
+
111
+ return { quote, length: text.length + quoteCount + (quote ? 2 : 0) }
112
+ }
113
+
114
+ type VisitedSheet = {
115
+ readonly name: string
116
+ readonly sheet: object | undefined
117
+ readonly range: CellRange | undefined
118
+ }
119
+
120
+ /** Annotated text for a cell, or `undefined` to write it plain. */
121
+ type CellDecorator = (input: {
122
+ readonly sheet: VisitedSheet
123
+ readonly row: number
124
+ readonly column: number
125
+ readonly text: string
126
+ }) => string | undefined
127
+
128
+ /** Write bounded CSV for every sheet; throws once `budget` characters would be exceeded. */
129
+ const renderSheets = (
130
+ sheets: ReadonlyArray<VisitedSheet>,
131
+ budget: number,
132
+ decorate?: CellDecorator
133
+ ) => {
134
+ let characters = 0
135
+ const output: Array<string> = []
136
+
137
+ const append = (text: string) => {
138
+ if (text.length > budget - characters) throw outputTooLarge()
139
+
140
+ characters += text.length
141
+ output.push(text)
142
+ }
143
+
144
+ const appendCell = (text: string) => {
145
+ if (text.length > budget - characters) throw outputTooLarge()
146
+
147
+ const field = csvField(text, budget - characters)
148
+
149
+ if (field.length > budget - characters) throw outputTooLarge()
150
+
151
+ append(field.quote ? `"${text.replaceAll('"', '""')}"` : text)
152
+ }
153
+
154
+ for (const visited of sheets) {
155
+ const { name, sheet, range } = visited
156
+
157
+ if (sheet === undefined) continue
158
+
159
+ if (output.length > 0) append('\n\n')
160
+
161
+ append('# ')
162
+ append(name)
163
+ append('\n')
164
+
165
+ if (range === undefined) continue
166
+
167
+ for (let row = range.start.r; row <= range.end.r; row += 1) {
168
+ if (row > range.start.r) append('\n')
169
+
170
+ for (let column = range.start.c; column <= range.end.c; column += 1) {
171
+ if (column > range.start.c) append(',')
172
+
173
+ const cell = ownProperty(sheet, encodeCellAddress({ r: row, c: column }))
174
+
175
+ if (!Predicate.isObject(cell)) {
176
+ appendCell('')
177
+ continue
178
+ }
179
+
180
+ const text = cellText(cell)
181
+
182
+ appendCell(decorate?.({ sheet: visited, row, column, text }) ?? text)
183
+ }
184
+ }
185
+ }
186
+
187
+ return output.join('')
188
+ }
189
+
190
+ /**
191
+ * Validate every sheet before touching any cell, then generate bounded CSV incrementally, never
192
+ * with `sheet_to_csv` (which can walk billions of absent cells or allocate an unbounded quoted
193
+ * string).
194
+ *
195
+ * Existing cells inside an external hyperlink are written as `text <url>`. Plain text is
196
+ * rendered first, so links only use budget the plain text leaves over: in document order, an
197
+ * annotation that no longer fits leaves its cell plain and one marker ends the output.
198
+ */
199
+ export const extractBoundedXlsxText = (
200
+ workbook: XlsxWorkbook,
201
+ limits: XlsxTextLimits,
202
+ options: XlsxTextOptions = {}
203
+ ) => {
204
+ if (workbook.SheetNames.length > limits.maxXlsxSheets) throw tooLarge()
205
+
206
+ let visits = 0
207
+
208
+ const sheets = workbook.SheetNames.map((name): VisitedSheet => {
209
+ const candidate = ownProperty(workbook.Sheets, name)
210
+ const sheet = Predicate.isObject(candidate) ? candidate : undefined
211
+ const ref = sheet === undefined ? undefined : ownProperty(sheet, '!ref')
212
+
213
+ if (ref !== undefined && !Predicate.isString(ref)) throw invalidRange()
214
+
215
+ const range = ref === undefined ? undefined : parseCellRange(ref)
216
+
217
+ if (ref !== undefined && range === undefined) throw invalidRange()
218
+
219
+ visits += range === undefined ? 0 : cellCount(range)
220
+
221
+ if (!Number.isSafeInteger(visits) || visits > limits.maxXlsxCellVisits) throw tooLarge()
222
+
223
+ return { name, sheet, range }
224
+ })
225
+
226
+ const links = options.hyperlinks ?? new Map()
227
+ const truncated = options.hyperlinksTruncated === true
228
+
229
+ if (links.size === 0 && !truncated) return renderSheets(sheets, limits.maxXlsxTextCharacters)
230
+
231
+ // Reserve the marker's space so it always fits inside the character limit.
232
+ const budget = limits.maxXlsxTextCharacters - omittedHyperlinksMarkerReserve
233
+ let spare = budget - renderSheets(sheets, budget).length
234
+ let dropped = false
235
+ const lookups = new Map<string, HyperlinkLookup>()
236
+
237
+ const lookupFor = (sheet: VisitedSheet) => {
238
+ const sheetLinks = links.get(sheet.name)
239
+
240
+ if (sheetLinks === undefined || sheet.range === undefined) return undefined
241
+
242
+ const existing = lookups.get(sheet.name)
243
+
244
+ if (existing !== undefined) return existing
245
+
246
+ const lookup = makeHyperlinkLookup(sheetLinks, sheet.range)
247
+ lookups.set(sheet.name, lookup)
248
+
249
+ return lookup
250
+ }
251
+
252
+ const output = renderSheets(sheets, budget, ({ sheet, row, column, text }) => {
253
+ const link = lookupFor(sheet)?.at(row, column)
254
+
255
+ if (link === undefined || link.target === text) return undefined
256
+
257
+ const label = text.length > 0 ? text : (link.display ?? '')
258
+ const plainLength = csvField(text).length
259
+ const annotatedLength = label.length + (label.length > 0 ? 3 : 2) + link.target.length
260
+
261
+ // Constant-time bound first (CSV escaping only adds), then a scan capped at the budget.
262
+ if (annotatedLength - plainLength > spare) {
263
+ dropped = true
264
+
265
+ return undefined
266
+ }
267
+
268
+ const annotated = label.length > 0 ? `${label} <${link.target}>` : `<${link.target}>`
269
+ const extra = csvField(annotated, plainLength + spare).length - plainLength
270
+
271
+ if (extra > spare) {
272
+ dropped = true
273
+
274
+ return undefined
275
+ }
276
+
277
+ spare -= extra
278
+
279
+ return annotated
280
+ })
281
+
282
+ const reasons = [
283
+ ...(dropped ? [outputLimitReason] : []),
284
+ ...(truncated ? [hyperlinkLimitReason] : [])
285
+ ]
286
+
287
+ return reasons.length === 0 ? output : output + omittedHyperlinksMarker(reasons)
288
+ }
@@ -0,0 +1,133 @@
1
+ import { Buffer } from 'node:buffer'
2
+ import { FileExtractionError } from '../errors.ts'
3
+ import {
4
+ sheetJsAttributeEscape,
5
+ sheetJsAttributeText,
6
+ sheetJsTags,
7
+ stripSheetJsNamespace,
8
+ sheetJsXmlHeader,
9
+ spreadsheetMainNamespace,
10
+ officeDocumentRelationshipsNamespace
11
+ } from './sheetjs-xml.ts'
12
+ import type { SheetJsTag } from './sheetjs-xml.ts'
13
+ import { decodeXmlEntities, prefixedAttribute, rawXmlAttributes } from './xml-text.ts'
14
+
15
+ /**
16
+ * `xl/workbook.xml` is never handed to SheetJS as uploaded. SheetJS's workbook parser slices and
17
+ * decodes the whole prefix of the part at every `</definedName>` (quadratic), and walks every
18
+ * `<sheet>` its own tag pattern finds, resolving each through its `r:id`, so one worksheet can be
19
+ * parsed once per declaration. The extractor reads the sheet list itself, checks that its strict
20
+ * scan and SheetJS's tag grammar see the same sheets, and writes a minimal workbook: the sheets in
21
+ * order (name, `sheetId`, hidden state, a fresh `r:id`) and the 1904 date system.
22
+ */
23
+
24
+ export type WorkbookSheetDeclaration = {
25
+ /** The sheet name exactly as SheetJS reads it (`unescapexml(utf8read(name))`). */
26
+ readonly name: string
27
+ readonly sheetId: string
28
+ readonly state: 'hidden' | 'veryHidden' | undefined
29
+ /** The `r:id` of the sheet's relationship in the uploaded workbook, entity-decoded. */
30
+ readonly relationshipId: string | undefined
31
+ }
32
+
33
+ export type WorkbookModel = {
34
+ readonly sheets: ReadonlyArray<WorkbookSheetDeclaration>
35
+ readonly date1904: boolean
36
+ }
37
+
38
+ export const malformedWorkbookMessage = 'XLSX workbook is malformed.'
39
+
40
+ const malformed = () =>
41
+ new FileExtractionError({ format: 'xlsx', message: malformedWorkbookMessage })
42
+
43
+ const tooManySheets = () =>
44
+ new FileExtractionError({
45
+ format: 'xlsx',
46
+ message: 'XLSX exceeds the worksheet or cell-visit limit.'
47
+ })
48
+
49
+ // `[^<>]` keeps each candidate inside one tag, so the strict scan stays linear.
50
+ const strictSheetTag = /<(?:[\w.-]+:)?sheet(?=[\s/>])[^<>]*>/g
51
+
52
+ const plainSheetId = /^[1-9]\d{0,8}$/
53
+
54
+ /**
55
+ * The sheets of `xl/workbook.xml` and its date system. Throws `FileExtractionError` when the
56
+ * workbook declares more than `maxSheets` sheets (counted by either scan), when the strict scan
57
+ * and SheetJS's grammar disagree on the number or names of sheets, when a name is missing, holds
58
+ * CDATA, or repeats another ignoring case, or when there are no sheets.
59
+ */
60
+ export const readWorkbookModel = (bytes: Uint8Array, maxSheets: number): WorkbookModel => {
61
+ // SheetJS parses the Latin-1 ("binary") view and decodes attribute text as UTF-8 afterwards.
62
+ const text = Buffer.from(bytes.buffer, bytes.byteOffset, bytes.byteLength).toString('latin1')
63
+ const sheetJsSheets: Array<SheetJsTag> = []
64
+ let date1904 = false
65
+
66
+ for (const tag of sheetJsTags(text)) {
67
+ const head = stripSheetJsNamespace(tag.head)
68
+
69
+ if (head === '<sheet') {
70
+ sheetJsSheets.push(tag)
71
+
72
+ if (sheetJsSheets.length > maxSheets) throw tooManySheets()
73
+ } else if (head === '<workbookPr' || head === '<workbookPr/>') {
74
+ // Each `workbookPr` that carries the attribute overrides the last (SheetJS `parsexmlbool`).
75
+ const value = tag.attributes.get('date1904')
76
+
77
+ if (value !== undefined) date1904 = value === '1' || value === 'true'
78
+ }
79
+ }
80
+
81
+ const strictSheets: Array<ReadonlyMap<string, string>> = []
82
+
83
+ for (const [tag] of text.matchAll(strictSheetTag)) {
84
+ strictSheets.push(rawXmlAttributes(tag))
85
+
86
+ if (strictSheets.length > maxSheets) throw tooManySheets()
87
+ }
88
+
89
+ if (sheetJsSheets.length === 0 || sheetJsSheets.length !== strictSheets.length) throw malformed()
90
+
91
+ const seen = new Set<string>()
92
+
93
+ const sheets = sheetJsSheets.map((tag, index): WorkbookSheetDeclaration => {
94
+ const strict = strictSheets[index]
95
+ const rawName = tag.attributes.get('name')
96
+
97
+ if (strict === undefined || rawName === undefined || strict.get('name') !== rawName)
98
+ throw malformed()
99
+
100
+ const name = sheetJsAttributeText(rawName)
101
+
102
+ // Excel compares sheet names ignoring case.
103
+ if (name === undefined || seen.has(name.toLowerCase())) throw malformed()
104
+
105
+ seen.add(name.toLowerCase())
106
+
107
+ const state = tag.attributes.get('state')
108
+ const sheetId = tag.attributes.get('sheetId')
109
+ const relationshipId = prefixedAttribute(strict, 'id')
110
+
111
+ return {
112
+ name,
113
+ sheetId: sheetId !== undefined && plainSheetId.test(sheetId) ? sheetId : `${index + 1}`,
114
+ state: state === 'hidden' || state === 'veryHidden' ? state : undefined,
115
+ relationshipId: relationshipId === undefined ? undefined : decodeXmlEntities(relationshipId)
116
+ }
117
+ })
118
+
119
+ return { sheets, date1904 }
120
+ }
121
+
122
+ /** The generated workbook: sheet `n` has `r:id="rId<n>"`. */
123
+ export const workbookXml = (model: WorkbookModel) =>
124
+ `${sheetJsXmlHeader}<workbook xmlns="${spreadsheetMainNamespace}" xmlns:r="${officeDocumentRelationshipsNamespace}">${
125
+ model.date1904 ? '<workbookPr date1904="1"/>' : ''
126
+ }<sheets>${model.sheets
127
+ .map(
128
+ (sheet, index) =>
129
+ `<sheet name="${sheetJsAttributeEscape(sheet.name)}" sheetId="${sheet.sheetId}"${
130
+ sheet.state === undefined ? '' : ` state="${sheet.state}"`
131
+ } r:id="rId${index + 1}"/>`
132
+ )
133
+ .join('')}</sheets></workbook>`
@@ -0,0 +1,77 @@
1
+ const xmlEntity = /&([^;&\s]{1,16});/g
2
+
3
+ const hexEntity = /^#x([0-9a-fA-F]+)$/
4
+
5
+ const decimalEntity = /^#(\d+)$/
6
+
7
+ const namedEntities: ReadonlyMap<string, string> = new Map([
8
+ ['amp', '&'],
9
+ ['lt', '<'],
10
+ ['gt', '>'],
11
+ ['quot', '"'],
12
+ ['apos', "'"]
13
+ ])
14
+
15
+ const decodeCodePoint = (raw: string, codePointText: string, radix: number) => {
16
+ const codePoint = Number.parseInt(codePointText, radix)
17
+
18
+ if (!Number.isInteger(codePoint) || codePoint < 0 || codePoint > 0x10ffff) return raw
19
+
20
+ return String.fromCodePoint(codePoint)
21
+ }
22
+
23
+ const decodeXmlEntity = (raw: string, entity: string) => {
24
+ const named = namedEntities.get(entity)
25
+
26
+ if (named !== undefined) return named
27
+
28
+ const hex = hexEntity.exec(entity)?.[1]
29
+
30
+ if (hex !== undefined) return decodeCodePoint(raw, hex, 16)
31
+
32
+ const decimal = decimalEntity.exec(entity)?.[1]
33
+
34
+ if (decimal !== undefined) return decodeCodePoint(raw, decimal, 10)
35
+
36
+ return raw
37
+ }
38
+
39
+ /** Decode the predefined XML entities and numeric character references; keep unknown ones. */
40
+ export const decodeXmlEntities = (text: string) =>
41
+ text.replace(xmlEntity, (raw, entity: string) => decodeXmlEntity(raw, entity))
42
+
43
+ // A name may only start after a non-name character, so a long run of name characters is scanned
44
+ // once rather than once per starting position.
45
+ const attributePattern = /(?<![\w.:-])([\w.:-]+)\s*=\s*(?:"([^"]*)"|'([^']*)')/g
46
+
47
+ /** Attributes of one start tag as written (not decoded), keyed by qualified name; first wins. */
48
+ export const rawXmlAttributes = (tag: string): ReadonlyMap<string, string> => {
49
+ const attributes = new Map<string, string>()
50
+
51
+ for (const match of tag.matchAll(attributePattern)) {
52
+ const name = match[1]
53
+ const value = match[2] ?? match[3]
54
+
55
+ if (name !== undefined && value !== undefined && !attributes.has(name))
56
+ attributes.set(name, value)
57
+ }
58
+
59
+ return attributes
60
+ }
61
+
62
+ /** Attributes of one start tag, entity-decoded, keyed by their qualified name. */
63
+ export const xmlAttributes = (tag: string): ReadonlyMap<string, string> =>
64
+ new Map(
65
+ Array.from(rawXmlAttributes(tag), ([name, value]) => [name, decodeXmlEntities(value)] as const)
66
+ )
67
+
68
+ /** The value of a namespace-prefixed attribute such as `r:id`, whatever the prefix. */
69
+ export const prefixedAttribute = (attributes: ReadonlyMap<string, string>, localName: string) => {
70
+ for (const [name, value] of attributes) {
71
+ const separator = name.indexOf(':')
72
+
73
+ if (separator > 0 && name.slice(separator + 1) === localName) return value
74
+ }
75
+
76
+ return undefined
77
+ }
@@ -0,0 +1,18 @@
1
+ const nonPrintableCharacters = /[\u0000-\u0008\u000B\u000C\u000E-\u001F\u007F]/g
2
+
3
+ const longDotRuns = /\.{4,}/g
4
+
5
+ const horizontalWhitespaceRuns = /[\t ]{2,}/g
6
+
7
+ const blankLineRuns = /\n{3,}/g
8
+
9
+ /** Normalize line endings, drop control characters, and collapse layout noise. */
10
+ export const sanitizeExtractedText = (text: string) =>
11
+ text
12
+ .replaceAll('\r\n', '\n')
13
+ .replaceAll('\r', '\n')
14
+ .replace(nonPrintableCharacters, '')
15
+ .replace(longDotRuns, '…')
16
+ .replace(horizontalWhitespaceRuns, ' ')
17
+ .replace(blankLineRuns, '\n\n')
18
+ .trim()
package/src/service.ts ADDED
@@ -0,0 +1,21 @@
1
+ import { Context } from 'effect'
2
+ import type { Effect } from 'effect'
3
+ import type { FileExtractorError } from './errors.ts'
4
+ import type { ExtractedFile, FileInput } from './format.ts'
5
+
6
+ export type FileExtractorApi = {
7
+ /**
8
+ * Extract sanitized text from one file. Fails with `UnsupportedFileFormatError` for unknown
9
+ * formats, `FileExtractionError` for unreadable, oversized, or empty files, and
10
+ * `SheetJsUnavailableError` when an XLSX file arrives but SheetJS 0.20.3+ is not installed.
11
+ */
12
+ readonly extract: (input: FileInput) => Effect.Effect<ExtractedFile, FileExtractorError>
13
+ }
14
+
15
+ /**
16
+ * The file extractor service. The tag is runtime-portable; the Node implementation lives in
17
+ * `@yolk-sdk/extractors/node`.
18
+ */
19
+ export class FileExtractor extends Context.Service<FileExtractor, FileExtractorApi>()(
20
+ '@yolk-sdk/extractors/FileExtractor'
21
+ ) {}