@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
@@ -0,0 +1,368 @@
1
+ import { Buffer } from 'node:buffer'
2
+ import { Readable } from 'node:stream'
3
+ import { createInflateRaw } from 'node:zlib'
4
+ import { Effect } from 'effect'
5
+ import { zipSync } from 'fflate'
6
+ import { OfficeArchiveError } from '../errors.ts'
7
+ import type { OfficeFileFormat } from '../format.ts'
8
+ import { defaultFileExtractorLimits } from '../limits.ts'
9
+ import type { FileExtractorLimits } from '../limits.ts'
10
+ import { sheetJsTextViews } from './sheetjs-xml.ts'
11
+ import {
12
+ contentTypesRouteToBinary,
13
+ isAlternateFormatEntry,
14
+ relationshipsRouteToBinary
15
+ } from './xlsx-routing.ts'
16
+
17
+ export type OfficeArchiveLimits = Pick<
18
+ FileExtractorLimits,
19
+ 'maxArchiveEntries' | 'maxExpandedBytes' | 'maxInputBytes'
20
+ >
21
+
22
+ const invalid = () => new OfficeArchiveError({ message: 'Invalid Office archive.' })
23
+
24
+ const tooLargeMessage = 'Office archive expansion exceeds its declared size or limit.'
25
+
26
+ const tooLarge = (expandedBytes?: number) =>
27
+ expandedBytes === undefined
28
+ ? new OfficeArchiveError({ message: tooLargeMessage })
29
+ : new OfficeArchiveError({ message: tooLargeMessage, expandedBytes })
30
+
31
+ const compressedChunkBytes = 1024
32
+
33
+ /** Inflated output is counted in slices of at most this many bytes. */
34
+ export const officeInflateChunkBytes = 16 * 1024
35
+
36
+ const mainParts: Readonly<Record<OfficeFileFormat, string>> = {
37
+ docx: 'word/document.xml',
38
+ pptx: 'ppt/presentation.xml',
39
+ xlsx: 'xl/workbook.xml'
40
+ }
41
+
42
+ type ArchiveEntry = {
43
+ readonly name: string
44
+ readonly start: number
45
+ readonly compressedSize: number
46
+ readonly originalSize: number
47
+ readonly method: number
48
+ }
49
+
50
+ /** Read the directory only as a bounded index. Its sizes are never trusted for allocation. */
51
+ const archiveEntries = (bytes: Buffer, limits: OfficeArchiveLimits) => {
52
+ let end = bytes.length - 22
53
+ const earliest = Math.max(0, end - 65535)
54
+
55
+ while (end >= earliest && bytes.readUInt32LE(end) !== 0x06054b50) end -= 1
56
+
57
+ if (end < earliest || end + 22 + bytes.readUInt16LE(end + 20) !== bytes.length) throw invalid()
58
+
59
+ const count = bytes.readUInt16LE(end + 10)
60
+ const directorySize = bytes.readUInt32LE(end + 12)
61
+ const directoryOffset = bytes.readUInt32LE(end + 16)
62
+
63
+ if (
64
+ bytes.readUInt32LE(end + 4) !== 0 ||
65
+ bytes.readUInt16LE(end + 8) !== count ||
66
+ count === 0 ||
67
+ count > limits.maxArchiveEntries ||
68
+ directoryOffset + directorySize !== end
69
+ )
70
+ throw invalid()
71
+
72
+ const entries: Array<ArchiveEntry> = []
73
+ // OPC part names are case-insensitive, and SheetJS looks entries up ignoring case.
74
+ const names = new Set<string>()
75
+ let offset = directoryOffset
76
+ let declaredTotal = 0
77
+
78
+ for (let index = 0; index < count; index += 1) {
79
+ if (offset + 46 > end || bytes.readUInt32LE(offset) !== 0x02014b50) throw invalid()
80
+
81
+ const flags = bytes.readUInt16LE(offset + 8)
82
+ const method = bytes.readUInt16LE(offset + 10)
83
+ const compressedSize = bytes.readUInt32LE(offset + 20)
84
+ const originalSize = bytes.readUInt32LE(offset + 24)
85
+ const nameSize = bytes.readUInt16LE(offset + 28)
86
+ const extraSize = bytes.readUInt16LE(offset + 30)
87
+ const commentSize = bytes.readUInt16LE(offset + 32)
88
+ const local = bytes.readUInt32LE(offset + 42)
89
+ const next = offset + 46 + nameSize + extraSize + commentSize
90
+
91
+ // Reject encryption, unsupported methods, split/ZIP64 archives, and ambiguous paths.
92
+ if (
93
+ next > end ||
94
+ nameSize === 0 ||
95
+ nameSize > 1024 ||
96
+ (flags & ~0x080e) !== 0 ||
97
+ (method !== 0 && method !== 8) ||
98
+ bytes.readUInt16LE(offset + 34) !== 0 ||
99
+ compressedSize === 0xffffffff ||
100
+ originalSize === 0xffffffff ||
101
+ local + 30 > directoryOffset
102
+ )
103
+ throw invalid()
104
+
105
+ const nameBytes = bytes.subarray(offset + 46, offset + 46 + nameSize)
106
+ const name = new TextDecoder('utf-8', { fatal: true }).decode(nameBytes)
107
+
108
+ // `//` is rejected (SheetJS collapses the first one), but directory entries ending in `/` stay.
109
+ if (
110
+ /[\\\u0000-\u001f]/.test(name) ||
111
+ name.startsWith('/') ||
112
+ name.includes('//') ||
113
+ name.split('/').some(part => part === '..' || part === '.') ||
114
+ names.has(name.toLowerCase()) ||
115
+ /vbaProject\.bin$/i.test(name)
116
+ )
117
+ throw invalid()
118
+
119
+ names.add(name.toLowerCase())
120
+
121
+ if (
122
+ bytes.readUInt32LE(local) !== 0x04034b50 ||
123
+ bytes.readUInt16LE(local + 6) !== flags ||
124
+ bytes.readUInt16LE(local + 8) !== method ||
125
+ bytes.readUInt16LE(local + 26) !== nameSize
126
+ )
127
+ throw invalid()
128
+
129
+ const start = local + 30 + nameSize + bytes.readUInt16LE(local + 28)
130
+
131
+ if (
132
+ start + compressedSize > directoryOffset ||
133
+ !bytes.subarray(local + 30, local + 30 + nameSize).equals(nameBytes)
134
+ )
135
+ throw invalid()
136
+
137
+ // Data descriptors may leave local sizes zero; all nonzero declarations must agree.
138
+ for (const [position, expected] of [
139
+ [18, compressedSize],
140
+ [22, originalSize]
141
+ ] as const) {
142
+ const value = bytes.readUInt32LE(local + position)
143
+
144
+ if (value !== expected && !((flags & 8) !== 0 && value === 0)) throw invalid()
145
+ }
146
+
147
+ declaredTotal += originalSize
148
+
149
+ if (declaredTotal > limits.maxExpandedBytes) throw tooLarge()
150
+
151
+ entries.push({ name, start, compressedSize, originalSize, method })
152
+ offset = next
153
+ }
154
+
155
+ if (offset !== end) throw invalid()
156
+
157
+ return entries
158
+ }
159
+
160
+ /**
161
+ * A superset of SheetJS's `hlinkregex` (`/<(?:\w+:)?hyperlink [^<>]*>/`). A candidate never spans
162
+ * a `<`, so each scan stops at the next tag and the strip stays linear in the part size.
163
+ */
164
+ const hyperlinkTag = /<\/?(?:[\w.-]+:)?hyperlink\b[^<>]*>/gi
165
+
166
+ const hyperlinkTagStart = /<\/?(?:[\w.-]+:)?hyperlink\b/i
167
+
168
+ /**
169
+ * Whether a part SheetJS would decode as BOM-marked UTF-16 contains a hyperlink tag. Covers
170
+ * SheetJS `cc2str` UTF-16 BOM decoding (little- and big-endian from byte 2, including its
171
+ * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
172
+ */
173
+ export const utf16PartHasHyperlink = (content: Uint8Array) =>
174
+ sheetJsTextViews(content)
175
+ .slice(1)
176
+ .some(text => hyperlinkTagStart.test(text))
177
+
178
+ const unsupportedXlsxParts = () =>
179
+ new OfficeArchiveError({
180
+ message: 'XLSX archive contains binary (XLSB), ODS, or Numbers parts.'
181
+ })
182
+
183
+ /** Hyperlink start tags (Latin-1 text of the original bytes) found while stripping, per part. */
184
+ export type StrippedHyperlinkTags = ReadonlyMap<string, ReadonlyArray<string>>
185
+
186
+ export type NormalizedOfficeArchive = {
187
+ /** Every validated (and, for XLSX, hyperlink-stripped) part by entry name. */
188
+ readonly parts: Readonly<Record<string, Uint8Array>>
189
+ /** XLSX only: removed hyperlink tags, at most `maxHyperlinkTags` across the workbook. */
190
+ readonly hyperlinkTags: StrippedHyperlinkTags
191
+ }
192
+
193
+ const inflateEntry = async (compressed: Buffer, record: (chunk: Buffer) => void): Promise<void> => {
194
+ function* inputChunks() {
195
+ for (let offset = 0; offset < compressed.length; offset += compressedChunkBytes) {
196
+ yield compressed.subarray(offset, offset + compressedChunkBytes)
197
+ }
198
+ }
199
+
200
+ const source = Readable.from(inputChunks(), { highWaterMark: 1 })
201
+
202
+ // Stream high-water marks are valid Transform options that `ZlibOptions` does not declare.
203
+ const inflateOptions = {
204
+ chunkSize: officeInflateChunkBytes,
205
+ readableHighWaterMark: officeInflateChunkBytes,
206
+ writableHighWaterMark: compressedChunkBytes
207
+ }
208
+
209
+ const inflater = createInflateRaw(inflateOptions)
210
+
211
+ source.pipe(inflater)
212
+
213
+ try {
214
+ // Async iteration may combine buffered chunks; bound accounting slices explicitly.
215
+ for await (const value of inflater) {
216
+ if (!Buffer.isBuffer(value)) throw invalid()
217
+
218
+ for (let offset = 0; offset < value.length; offset += officeInflateChunkBytes) {
219
+ record(value.subarray(offset, offset + officeInflateChunkBytes))
220
+ }
221
+ }
222
+
223
+ if (inflater.bytesWritten !== compressed.length) throw invalid()
224
+ } finally {
225
+ source.destroy()
226
+ inflater.destroy()
227
+ }
228
+ }
229
+
230
+ export type ReadOfficeArchiveOptions = {
231
+ /** XLSX: hyperlink start tags to capture across the workbook (default 0). */
232
+ readonly maxHyperlinkTags?: number
233
+ }
234
+
235
+ /** A fresh stored-entry ZIP of validated parts; the attacker's ZIP index is never reused. */
236
+ export const storedArchive = (parts: Readonly<Record<string, Uint8Array>>) =>
237
+ zipSync({ ...parts }, { level: 0 })
238
+
239
+ /**
240
+ * Inflate bounded input chunks, count actual output, and return the validated parts. The
241
+ * attacker's ZIP index is discarded: parsers only get archives rebuilt from these parts.
242
+ *
243
+ * For XLSX, every part loses its `<hyperlink>` tags: SheetJS expands each hyperlink range into
244
+ * per-cell objects before any budget runs, so one `ref="A1:XFD1048576"` exhausts memory. The
245
+ * removed tags are returned so links can still be shown. XLSX input that SheetJS would route to
246
+ * its binary (XLSB), ODS, or Numbers parsers is rejected early with a clear error; the guarantee
247
+ * is that SheetJS only receives `buildSheetJsInput`'s allowlisted archive.
248
+ */
249
+ export const readOfficeArchive = (
250
+ input: Uint8Array,
251
+ format: OfficeFileFormat,
252
+ limits: OfficeArchiveLimits,
253
+ options: ReadOfficeArchiveOptions = {}
254
+ ) =>
255
+ Effect.tryPromise({
256
+ try: async (): Promise<NormalizedOfficeArchive> => {
257
+ const bytes = Buffer.from(input.buffer, input.byteOffset, input.byteLength)
258
+
259
+ if (bytes.length < 22 || bytes.length > limits.maxInputBytes) throw invalid()
260
+
261
+ const entries = archiveEntries(bytes, limits)
262
+ const mainPart = mainParts[format]
263
+ const maxHyperlinkTags = options.maxHyperlinkTags ?? 0
264
+
265
+ if (format === 'xlsx' && entries.some(entry => isAlternateFormatEntry(entry.name)))
266
+ throw unsupportedXlsxParts()
267
+
268
+ if (
269
+ !entries.some(entry => entry.name === '[Content_Types].xml') ||
270
+ !entries.some(entry => entry.name === mainPart)
271
+ )
272
+ throw invalid()
273
+
274
+ const validated: Record<string, Uint8Array> = Object.create(null)
275
+ const hyperlinkTags = new Map<string, Array<string>>()
276
+ let capturedTags = 0
277
+ let expandedBytes = 0
278
+
279
+ for (const entry of entries) {
280
+ const chunks: Array<Buffer> = []
281
+ let entryBytes = 0
282
+
283
+ const record = (chunk: Buffer) => {
284
+ entryBytes += chunk.length
285
+ expandedBytes += chunk.length
286
+
287
+ if (expandedBytes > limits.maxExpandedBytes || entryBytes > entry.originalSize)
288
+ throw tooLarge(expandedBytes)
289
+
290
+ chunks.push(chunk)
291
+ }
292
+
293
+ const compressed = bytes.subarray(entry.start, entry.start + entry.compressedSize)
294
+
295
+ if (entry.method === 0) record(compressed)
296
+ else await inflateEntry(compressed, record)
297
+
298
+ if (entryBytes !== entry.originalSize) throw invalid()
299
+
300
+ const content = Buffer.concat(chunks, entryBytes)
301
+
302
+ if (format !== 'xlsx') {
303
+ validated[entry.name] = content
304
+ continue
305
+ }
306
+
307
+ // Relationship targets need not end in .xml, so scan every entry without changing other
308
+ // bytes (including UTF-8 and binary parts). A space prevents removal from joining
309
+ // attacker-controlled fragments into a new parser-accepted hyperlink tag.
310
+ const tags: Array<string> = []
311
+
312
+ const rewritten = Buffer.from(
313
+ content.toString('latin1').replace(hyperlinkTag, tag => {
314
+ if (!tag.startsWith('</') && capturedTags < maxHyperlinkTags) {
315
+ capturedTags += 1
316
+ tags.push(tag)
317
+ }
318
+
319
+ return ' '
320
+ }),
321
+ 'latin1'
322
+ )
323
+
324
+ // SheetJS also decodes BOM-marked UTF-16 parts, which the Latin-1 strip cannot see
325
+ // through, and the strip itself can shift byte alignment or create a BOM. Check the exact
326
+ // bytes SheetJS will parse; Excel never writes UTF-16 parts, so reject rather than rewrite.
327
+ if (utf16PartHasHyperlink(rewritten)) throw invalid()
328
+
329
+ // Content types and relationships decide which parser SheetJS runs on each part.
330
+ const lowerName = entry.name.toLowerCase()
331
+
332
+ if (
333
+ (lowerName === '[content_types].xml' && contentTypesRouteToBinary(rewritten)) ||
334
+ (lowerName.endsWith('.rels') && relationshipsRouteToBinary(rewritten))
335
+ )
336
+ throw unsupportedXlsxParts()
337
+
338
+ if (tags.length > 0) hyperlinkTags.set(entry.name, tags)
339
+
340
+ validated[entry.name] = rewritten
341
+ }
342
+
343
+ return { parts: validated, hyperlinkTags }
344
+ },
345
+ // Out-of-range header reads (RangeError) and inflate failures are malformed archives too.
346
+ catch: error => (error instanceof OfficeArchiveError ? error : invalid())
347
+ })
348
+
349
+ /**
350
+ * Validate a DOCX, XLSX, or PPTX archive with bounded inflation and return a rebuilt stored-entry
351
+ * ZIP of every validated part, for storage or for other parsers. XLSX parts lose their hyperlink
352
+ * tags, and XLSX input with ODS or Numbers marker entries or XLSB parts is rejected; `.bin` parts
353
+ * SheetJS never parses (printer settings, OLE objects) are kept so stored files still open.
354
+ *
355
+ * The output is not SheetJS input. Never run SheetJS on it directly: extract XLSX text through
356
+ * `FileExtractor`, which hands SheetJS only an allowlisted archive it builds itself (worksheet,
357
+ * shared-string, style, and core-property parts with generated content types and relationships).
358
+ */
359
+ export const normalizeOfficeArchive = (
360
+ bytes: Uint8Array,
361
+ format: OfficeFileFormat,
362
+ limits: Partial<OfficeArchiveLimits> = {}
363
+ ) =>
364
+ readOfficeArchive(bytes, format, {
365
+ maxArchiveEntries: limits.maxArchiveEntries ?? defaultFileExtractorLimits.maxArchiveEntries,
366
+ maxExpandedBytes: limits.maxExpandedBytes ?? defaultFileExtractorLimits.maxExpandedBytes,
367
+ maxInputBytes: limits.maxInputBytes ?? defaultFileExtractorLimits.maxInputBytes
368
+ }).pipe(Effect.map(normalized => storedArchive(normalized.parts)))
@@ -0,0 +1,125 @@
1
+ import { strFromU8 } from 'fflate'
2
+ import { decodeXmlEntities } from './xml-text.ts'
3
+
4
+ type PptxXmlFile = {
5
+ readonly fileName: string
6
+ readonly bytes: Uint8Array
7
+ readonly group: number
8
+ readonly index: number
9
+ }
10
+
11
+ const slideXmlFile = /^ppt\/slides\/slide(\d+)\.xml$/
12
+
13
+ const notesXmlFile = /^ppt\/notesSlides\/notesSlide(\d+)\.xml$/
14
+
15
+ const xmlName = '[A-Za-z_][\\w.-]*'
16
+
17
+ const optionalXmlPrefix = `(?:${xmlName}:)?`
18
+
19
+ // Every pattern stops at the next `<` (`[^<>]`), and elements are paired in one forward pass, so
20
+ // extraction stays linear in the part size even for unterminated or unbalanced markup.
21
+ type XmlElement = { readonly start: RegExp; readonly end: RegExp }
22
+
23
+ const xmlElement = (localName: string): XmlElement => ({
24
+ start: new RegExp(`<${optionalXmlPrefix}${localName}\\b[^<>]*>`, 'g'),
25
+ end: new RegExp(`</${optionalXmlPrefix}${localName}>`, 'g')
26
+ })
27
+
28
+ const paragraphXml = xmlElement('p')
29
+
30
+ const textXml = xmlElement('t')
31
+
32
+ const lineBreakXml = new RegExp(`<${optionalXmlPrefix}br\\b[^<>]*/>`, 'g')
33
+
34
+ const tabXml = new RegExp(`<${optionalXmlPrefix}tab\\b[^<>]*/>`, 'g')
35
+
36
+ type ElementMatch = {
37
+ /** Start tag through end tag. */
38
+ readonly outer: string
39
+ /** Text between the start and end tags. */
40
+ readonly inner: string
41
+ }
42
+
43
+ /** Each start tag paired with the next end tag after it (the old lazy `[\s\S]*?` match). */
44
+ const elementMatches = (xml: string, element: XmlElement): ReadonlyArray<ElementMatch> => {
45
+ const matches: Array<ElementMatch> = []
46
+ let position = 0
47
+
48
+ for (;;) {
49
+ element.start.lastIndex = position
50
+ const start = element.start.exec(xml)
51
+
52
+ if (start === null) return matches
53
+
54
+ const contentStart = start.index + start[0].length
55
+ element.end.lastIndex = contentStart
56
+ const end = element.end.exec(xml)
57
+
58
+ // No end tag after this start means none after any later start either.
59
+ if (end === null) return matches
60
+
61
+ position = end.index + end[0].length
62
+ matches.push({
63
+ outer: xml.slice(start.index, position),
64
+ inner: xml.slice(contentStart, end.index)
65
+ })
66
+ }
67
+ }
68
+
69
+ const indexedXmlFile = (
70
+ fileName: string,
71
+ bytes: Uint8Array,
72
+ pattern: RegExp,
73
+ group: number
74
+ ): PptxXmlFile | undefined => {
75
+ const indexText = pattern.exec(fileName)?.[1]
76
+
77
+ if (indexText === undefined) return undefined
78
+
79
+ const index = Number.parseInt(indexText, 10)
80
+
81
+ if (!Number.isInteger(index)) return undefined
82
+
83
+ return { fileName, bytes, group, index }
84
+ }
85
+
86
+ const pptxXmlFile = (fileName: string, bytes: Uint8Array): PptxXmlFile | undefined =>
87
+ indexedXmlFile(fileName, bytes, slideXmlFile, 0) ??
88
+ indexedXmlFile(fileName, bytes, notesXmlFile, 1)
89
+
90
+ const comparePptxXmlFiles = (left: PptxXmlFile, right: PptxXmlFile) =>
91
+ left.group - right.group ||
92
+ left.index - right.index ||
93
+ left.fileName.localeCompare(right.fileName)
94
+
95
+ const extractParagraphText = (paragraph: string) => {
96
+ const xml = paragraph.replace(lineBreakXml, '<a:t>\n</a:t>').replace(tabXml, '<a:t>\t</a:t>')
97
+
98
+ return elementMatches(xml, textXml)
99
+ .map(match => decodeXmlEntities(match.inner))
100
+ .join('')
101
+ .trim()
102
+ }
103
+
104
+ const extractXmlText = (xml: string) => {
105
+ const paragraphs = elementMatches(xml, paragraphXml).map(match => match.outer)
106
+ const textSources = paragraphs.length > 0 ? paragraphs : [xml]
107
+
108
+ return textSources
109
+ .map(extractParagraphText)
110
+ .filter(text => text.length > 0)
111
+ .join('\n')
112
+ }
113
+
114
+ /** Slide text in slide order, then speaker notes, from the parts of a validated archive. */
115
+ export const extractPptxText = (parts: Readonly<Record<string, Uint8Array>>) =>
116
+ Object.entries(parts)
117
+ .flatMap(([fileName, fileBytes]) => {
118
+ const xmlFile = pptxXmlFile(fileName, fileBytes)
119
+
120
+ return xmlFile === undefined ? [] : [xmlFile]
121
+ })
122
+ .sort(comparePptxXmlFiles)
123
+ .map(file => extractXmlText(strFromU8(file.bytes)))
124
+ .filter(text => text.length > 0)
125
+ .join('\n\n')