@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
import { Buffer } from 'node:buffer'
|
|
2
|
+
import { Readable } from 'node:stream'
|
|
3
|
+
import { createInflateRaw } from 'node:zlib'
|
|
4
|
+
import { Effect } from 'effect'
|
|
5
|
+
import { zipSync } from 'fflate'
|
|
6
|
+
import { OfficeArchiveError } from '../errors.ts'
|
|
7
|
+
import type { OfficeFileFormat } from '../format.ts'
|
|
8
|
+
import { defaultFileExtractorLimits } from '../limits.ts'
|
|
9
|
+
import type { FileExtractorLimits } from '../limits.ts'
|
|
10
|
+
import { sheetJsTextViews } from './sheetjs-xml.ts'
|
|
11
|
+
import {
|
|
12
|
+
contentTypesRouteToBinary,
|
|
13
|
+
isAlternateFormatEntry,
|
|
14
|
+
relationshipsRouteToBinary
|
|
15
|
+
} from './xlsx-routing.ts'
|
|
16
|
+
|
|
17
|
+
export type OfficeArchiveLimits = Pick<
|
|
18
|
+
FileExtractorLimits,
|
|
19
|
+
'maxArchiveEntries' | 'maxExpandedBytes' | 'maxInputBytes'
|
|
20
|
+
>
|
|
21
|
+
|
|
22
|
+
const invalid = () => new OfficeArchiveError({ message: 'Invalid Office archive.' })
|
|
23
|
+
|
|
24
|
+
const tooLargeMessage = 'Office archive expansion exceeds its declared size or limit.'
|
|
25
|
+
|
|
26
|
+
const tooLarge = (expandedBytes?: number) =>
|
|
27
|
+
expandedBytes === undefined
|
|
28
|
+
? new OfficeArchiveError({ message: tooLargeMessage })
|
|
29
|
+
: new OfficeArchiveError({ message: tooLargeMessage, expandedBytes })
|
|
30
|
+
|
|
31
|
+
const compressedChunkBytes = 1024
|
|
32
|
+
|
|
33
|
+
/** Inflated output is counted in slices of at most this many bytes. */
|
|
34
|
+
export const officeInflateChunkBytes = 16 * 1024
|
|
35
|
+
|
|
36
|
+
const mainParts: Readonly<Record<OfficeFileFormat, string>> = {
|
|
37
|
+
docx: 'word/document.xml',
|
|
38
|
+
pptx: 'ppt/presentation.xml',
|
|
39
|
+
xlsx: 'xl/workbook.xml'
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
type ArchiveEntry = {
|
|
43
|
+
readonly name: string
|
|
44
|
+
readonly start: number
|
|
45
|
+
readonly compressedSize: number
|
|
46
|
+
readonly originalSize: number
|
|
47
|
+
readonly method: number
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/** Read the directory only as a bounded index. Its sizes are never trusted for allocation. */
|
|
51
|
+
const archiveEntries = (bytes: Buffer, limits: OfficeArchiveLimits) => {
|
|
52
|
+
let end = bytes.length - 22
|
|
53
|
+
const earliest = Math.max(0, end - 65535)
|
|
54
|
+
|
|
55
|
+
while (end >= earliest && bytes.readUInt32LE(end) !== 0x06054b50) end -= 1
|
|
56
|
+
|
|
57
|
+
if (end < earliest || end + 22 + bytes.readUInt16LE(end + 20) !== bytes.length) throw invalid()
|
|
58
|
+
|
|
59
|
+
const count = bytes.readUInt16LE(end + 10)
|
|
60
|
+
const directorySize = bytes.readUInt32LE(end + 12)
|
|
61
|
+
const directoryOffset = bytes.readUInt32LE(end + 16)
|
|
62
|
+
|
|
63
|
+
if (
|
|
64
|
+
bytes.readUInt32LE(end + 4) !== 0 ||
|
|
65
|
+
bytes.readUInt16LE(end + 8) !== count ||
|
|
66
|
+
count === 0 ||
|
|
67
|
+
count > limits.maxArchiveEntries ||
|
|
68
|
+
directoryOffset + directorySize !== end
|
|
69
|
+
)
|
|
70
|
+
throw invalid()
|
|
71
|
+
|
|
72
|
+
const entries: Array<ArchiveEntry> = []
|
|
73
|
+
// OPC part names are case-insensitive, and SheetJS looks entries up ignoring case.
|
|
74
|
+
const names = new Set<string>()
|
|
75
|
+
let offset = directoryOffset
|
|
76
|
+
let declaredTotal = 0
|
|
77
|
+
|
|
78
|
+
for (let index = 0; index < count; index += 1) {
|
|
79
|
+
if (offset + 46 > end || bytes.readUInt32LE(offset) !== 0x02014b50) throw invalid()
|
|
80
|
+
|
|
81
|
+
const flags = bytes.readUInt16LE(offset + 8)
|
|
82
|
+
const method = bytes.readUInt16LE(offset + 10)
|
|
83
|
+
const compressedSize = bytes.readUInt32LE(offset + 20)
|
|
84
|
+
const originalSize = bytes.readUInt32LE(offset + 24)
|
|
85
|
+
const nameSize = bytes.readUInt16LE(offset + 28)
|
|
86
|
+
const extraSize = bytes.readUInt16LE(offset + 30)
|
|
87
|
+
const commentSize = bytes.readUInt16LE(offset + 32)
|
|
88
|
+
const local = bytes.readUInt32LE(offset + 42)
|
|
89
|
+
const next = offset + 46 + nameSize + extraSize + commentSize
|
|
90
|
+
|
|
91
|
+
// Reject encryption, unsupported methods, split/ZIP64 archives, and ambiguous paths.
|
|
92
|
+
if (
|
|
93
|
+
next > end ||
|
|
94
|
+
nameSize === 0 ||
|
|
95
|
+
nameSize > 1024 ||
|
|
96
|
+
(flags & ~0x080e) !== 0 ||
|
|
97
|
+
(method !== 0 && method !== 8) ||
|
|
98
|
+
bytes.readUInt16LE(offset + 34) !== 0 ||
|
|
99
|
+
compressedSize === 0xffffffff ||
|
|
100
|
+
originalSize === 0xffffffff ||
|
|
101
|
+
local + 30 > directoryOffset
|
|
102
|
+
)
|
|
103
|
+
throw invalid()
|
|
104
|
+
|
|
105
|
+
const nameBytes = bytes.subarray(offset + 46, offset + 46 + nameSize)
|
|
106
|
+
const name = new TextDecoder('utf-8', { fatal: true }).decode(nameBytes)
|
|
107
|
+
|
|
108
|
+
// `//` is rejected (SheetJS collapses the first one), but directory entries ending in `/` stay.
|
|
109
|
+
if (
|
|
110
|
+
/[\\\u0000-\u001f]/.test(name) ||
|
|
111
|
+
name.startsWith('/') ||
|
|
112
|
+
name.includes('//') ||
|
|
113
|
+
name.split('/').some(part => part === '..' || part === '.') ||
|
|
114
|
+
names.has(name.toLowerCase()) ||
|
|
115
|
+
/vbaProject\.bin$/i.test(name)
|
|
116
|
+
)
|
|
117
|
+
throw invalid()
|
|
118
|
+
|
|
119
|
+
names.add(name.toLowerCase())
|
|
120
|
+
|
|
121
|
+
if (
|
|
122
|
+
bytes.readUInt32LE(local) !== 0x04034b50 ||
|
|
123
|
+
bytes.readUInt16LE(local + 6) !== flags ||
|
|
124
|
+
bytes.readUInt16LE(local + 8) !== method ||
|
|
125
|
+
bytes.readUInt16LE(local + 26) !== nameSize
|
|
126
|
+
)
|
|
127
|
+
throw invalid()
|
|
128
|
+
|
|
129
|
+
const start = local + 30 + nameSize + bytes.readUInt16LE(local + 28)
|
|
130
|
+
|
|
131
|
+
if (
|
|
132
|
+
start + compressedSize > directoryOffset ||
|
|
133
|
+
!bytes.subarray(local + 30, local + 30 + nameSize).equals(nameBytes)
|
|
134
|
+
)
|
|
135
|
+
throw invalid()
|
|
136
|
+
|
|
137
|
+
// Data descriptors may leave local sizes zero; all nonzero declarations must agree.
|
|
138
|
+
for (const [position, expected] of [
|
|
139
|
+
[18, compressedSize],
|
|
140
|
+
[22, originalSize]
|
|
141
|
+
] as const) {
|
|
142
|
+
const value = bytes.readUInt32LE(local + position)
|
|
143
|
+
|
|
144
|
+
if (value !== expected && !((flags & 8) !== 0 && value === 0)) throw invalid()
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
declaredTotal += originalSize
|
|
148
|
+
|
|
149
|
+
if (declaredTotal > limits.maxExpandedBytes) throw tooLarge()
|
|
150
|
+
|
|
151
|
+
entries.push({ name, start, compressedSize, originalSize, method })
|
|
152
|
+
offset = next
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
if (offset !== end) throw invalid()
|
|
156
|
+
|
|
157
|
+
return entries
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* A superset of SheetJS's `hlinkregex` (`/<(?:\w+:)?hyperlink [^<>]*>/`). A candidate never spans
|
|
162
|
+
* a `<`, so each scan stops at the next tag and the strip stays linear in the part size.
|
|
163
|
+
*/
|
|
164
|
+
const hyperlinkTag = /<\/?(?:[\w.-]+:)?hyperlink\b[^<>]*>/gi
|
|
165
|
+
|
|
166
|
+
const hyperlinkTagStart = /<\/?(?:[\w.-]+:)?hyperlink\b/i
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Whether a part SheetJS would decode as BOM-marked UTF-16 contains a hyperlink tag. Covers
|
|
170
|
+
* SheetJS `cc2str` UTF-16 BOM decoding (little- and big-endian from byte 2, including its
|
|
171
|
+
* `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
|
|
172
|
+
*/
|
|
173
|
+
export const utf16PartHasHyperlink = (content: Uint8Array) =>
|
|
174
|
+
sheetJsTextViews(content)
|
|
175
|
+
.slice(1)
|
|
176
|
+
.some(text => hyperlinkTagStart.test(text))
|
|
177
|
+
|
|
178
|
+
const unsupportedXlsxParts = () =>
|
|
179
|
+
new OfficeArchiveError({
|
|
180
|
+
message: 'XLSX archive contains binary (XLSB), ODS, or Numbers parts.'
|
|
181
|
+
})
|
|
182
|
+
|
|
183
|
+
/** Hyperlink start tags (Latin-1 text of the original bytes) found while stripping, per part. */
|
|
184
|
+
export type StrippedHyperlinkTags = ReadonlyMap<string, ReadonlyArray<string>>
|
|
185
|
+
|
|
186
|
+
export type NormalizedOfficeArchive = {
|
|
187
|
+
/** Every validated (and, for XLSX, hyperlink-stripped) part by entry name. */
|
|
188
|
+
readonly parts: Readonly<Record<string, Uint8Array>>
|
|
189
|
+
/** XLSX only: removed hyperlink tags, at most `maxHyperlinkTags` across the workbook. */
|
|
190
|
+
readonly hyperlinkTags: StrippedHyperlinkTags
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
const inflateEntry = async (compressed: Buffer, record: (chunk: Buffer) => void): Promise<void> => {
|
|
194
|
+
function* inputChunks() {
|
|
195
|
+
for (let offset = 0; offset < compressed.length; offset += compressedChunkBytes) {
|
|
196
|
+
yield compressed.subarray(offset, offset + compressedChunkBytes)
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
const source = Readable.from(inputChunks(), { highWaterMark: 1 })
|
|
201
|
+
|
|
202
|
+
// Stream high-water marks are valid Transform options that `ZlibOptions` does not declare.
|
|
203
|
+
const inflateOptions = {
|
|
204
|
+
chunkSize: officeInflateChunkBytes,
|
|
205
|
+
readableHighWaterMark: officeInflateChunkBytes,
|
|
206
|
+
writableHighWaterMark: compressedChunkBytes
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
const inflater = createInflateRaw(inflateOptions)
|
|
210
|
+
|
|
211
|
+
source.pipe(inflater)
|
|
212
|
+
|
|
213
|
+
try {
|
|
214
|
+
// Async iteration may combine buffered chunks; bound accounting slices explicitly.
|
|
215
|
+
for await (const value of inflater) {
|
|
216
|
+
if (!Buffer.isBuffer(value)) throw invalid()
|
|
217
|
+
|
|
218
|
+
for (let offset = 0; offset < value.length; offset += officeInflateChunkBytes) {
|
|
219
|
+
record(value.subarray(offset, offset + officeInflateChunkBytes))
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
if (inflater.bytesWritten !== compressed.length) throw invalid()
|
|
224
|
+
} finally {
|
|
225
|
+
source.destroy()
|
|
226
|
+
inflater.destroy()
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
export type ReadOfficeArchiveOptions = {
|
|
231
|
+
/** XLSX: hyperlink start tags to capture across the workbook (default 0). */
|
|
232
|
+
readonly maxHyperlinkTags?: number
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
/** A fresh stored-entry ZIP of validated parts; the attacker's ZIP index is never reused. */
|
|
236
|
+
export const storedArchive = (parts: Readonly<Record<string, Uint8Array>>) =>
|
|
237
|
+
zipSync({ ...parts }, { level: 0 })
|
|
238
|
+
|
|
239
|
+
/**
|
|
240
|
+
* Inflate bounded input chunks, count actual output, and return the validated parts. The
|
|
241
|
+
* attacker's ZIP index is discarded: parsers only get archives rebuilt from these parts.
|
|
242
|
+
*
|
|
243
|
+
* For XLSX, every part loses its `<hyperlink>` tags: SheetJS expands each hyperlink range into
|
|
244
|
+
* per-cell objects before any budget runs, so one `ref="A1:XFD1048576"` exhausts memory. The
|
|
245
|
+
* removed tags are returned so links can still be shown. XLSX input that SheetJS would route to
|
|
246
|
+
* its binary (XLSB), ODS, or Numbers parsers is rejected early with a clear error; the guarantee
|
|
247
|
+
* is that SheetJS only receives `buildSheetJsInput`'s allowlisted archive.
|
|
248
|
+
*/
|
|
249
|
+
export const readOfficeArchive = (
|
|
250
|
+
input: Uint8Array,
|
|
251
|
+
format: OfficeFileFormat,
|
|
252
|
+
limits: OfficeArchiveLimits,
|
|
253
|
+
options: ReadOfficeArchiveOptions = {}
|
|
254
|
+
) =>
|
|
255
|
+
Effect.tryPromise({
|
|
256
|
+
try: async (): Promise<NormalizedOfficeArchive> => {
|
|
257
|
+
const bytes = Buffer.from(input.buffer, input.byteOffset, input.byteLength)
|
|
258
|
+
|
|
259
|
+
if (bytes.length < 22 || bytes.length > limits.maxInputBytes) throw invalid()
|
|
260
|
+
|
|
261
|
+
const entries = archiveEntries(bytes, limits)
|
|
262
|
+
const mainPart = mainParts[format]
|
|
263
|
+
const maxHyperlinkTags = options.maxHyperlinkTags ?? 0
|
|
264
|
+
|
|
265
|
+
if (format === 'xlsx' && entries.some(entry => isAlternateFormatEntry(entry.name)))
|
|
266
|
+
throw unsupportedXlsxParts()
|
|
267
|
+
|
|
268
|
+
if (
|
|
269
|
+
!entries.some(entry => entry.name === '[Content_Types].xml') ||
|
|
270
|
+
!entries.some(entry => entry.name === mainPart)
|
|
271
|
+
)
|
|
272
|
+
throw invalid()
|
|
273
|
+
|
|
274
|
+
const validated: Record<string, Uint8Array> = Object.create(null)
|
|
275
|
+
const hyperlinkTags = new Map<string, Array<string>>()
|
|
276
|
+
let capturedTags = 0
|
|
277
|
+
let expandedBytes = 0
|
|
278
|
+
|
|
279
|
+
for (const entry of entries) {
|
|
280
|
+
const chunks: Array<Buffer> = []
|
|
281
|
+
let entryBytes = 0
|
|
282
|
+
|
|
283
|
+
const record = (chunk: Buffer) => {
|
|
284
|
+
entryBytes += chunk.length
|
|
285
|
+
expandedBytes += chunk.length
|
|
286
|
+
|
|
287
|
+
if (expandedBytes > limits.maxExpandedBytes || entryBytes > entry.originalSize)
|
|
288
|
+
throw tooLarge(expandedBytes)
|
|
289
|
+
|
|
290
|
+
chunks.push(chunk)
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
const compressed = bytes.subarray(entry.start, entry.start + entry.compressedSize)
|
|
294
|
+
|
|
295
|
+
if (entry.method === 0) record(compressed)
|
|
296
|
+
else await inflateEntry(compressed, record)
|
|
297
|
+
|
|
298
|
+
if (entryBytes !== entry.originalSize) throw invalid()
|
|
299
|
+
|
|
300
|
+
const content = Buffer.concat(chunks, entryBytes)
|
|
301
|
+
|
|
302
|
+
if (format !== 'xlsx') {
|
|
303
|
+
validated[entry.name] = content
|
|
304
|
+
continue
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
// Relationship targets need not end in .xml, so scan every entry without changing other
|
|
308
|
+
// bytes (including UTF-8 and binary parts). A space prevents removal from joining
|
|
309
|
+
// attacker-controlled fragments into a new parser-accepted hyperlink tag.
|
|
310
|
+
const tags: Array<string> = []
|
|
311
|
+
|
|
312
|
+
const rewritten = Buffer.from(
|
|
313
|
+
content.toString('latin1').replace(hyperlinkTag, tag => {
|
|
314
|
+
if (!tag.startsWith('</') && capturedTags < maxHyperlinkTags) {
|
|
315
|
+
capturedTags += 1
|
|
316
|
+
tags.push(tag)
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
return ' '
|
|
320
|
+
}),
|
|
321
|
+
'latin1'
|
|
322
|
+
)
|
|
323
|
+
|
|
324
|
+
// SheetJS also decodes BOM-marked UTF-16 parts, which the Latin-1 strip cannot see
|
|
325
|
+
// through, and the strip itself can shift byte alignment or create a BOM. Check the exact
|
|
326
|
+
// bytes SheetJS will parse; Excel never writes UTF-16 parts, so reject rather than rewrite.
|
|
327
|
+
if (utf16PartHasHyperlink(rewritten)) throw invalid()
|
|
328
|
+
|
|
329
|
+
// Content types and relationships decide which parser SheetJS runs on each part.
|
|
330
|
+
const lowerName = entry.name.toLowerCase()
|
|
331
|
+
|
|
332
|
+
if (
|
|
333
|
+
(lowerName === '[content_types].xml' && contentTypesRouteToBinary(rewritten)) ||
|
|
334
|
+
(lowerName.endsWith('.rels') && relationshipsRouteToBinary(rewritten))
|
|
335
|
+
)
|
|
336
|
+
throw unsupportedXlsxParts()
|
|
337
|
+
|
|
338
|
+
if (tags.length > 0) hyperlinkTags.set(entry.name, tags)
|
|
339
|
+
|
|
340
|
+
validated[entry.name] = rewritten
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
return { parts: validated, hyperlinkTags }
|
|
344
|
+
},
|
|
345
|
+
// Out-of-range header reads (RangeError) and inflate failures are malformed archives too.
|
|
346
|
+
catch: error => (error instanceof OfficeArchiveError ? error : invalid())
|
|
347
|
+
})
|
|
348
|
+
|
|
349
|
+
/**
|
|
350
|
+
* Validate a DOCX, XLSX, or PPTX archive with bounded inflation and return a rebuilt stored-entry
|
|
351
|
+
* ZIP of every validated part, for storage or for other parsers. XLSX parts lose their hyperlink
|
|
352
|
+
* tags, and XLSX input with ODS or Numbers marker entries or XLSB parts is rejected; `.bin` parts
|
|
353
|
+
* SheetJS never parses (printer settings, OLE objects) are kept so stored files still open.
|
|
354
|
+
*
|
|
355
|
+
* The output is not SheetJS input. Never run SheetJS on it directly: extract XLSX text through
|
|
356
|
+
* `FileExtractor`, which hands SheetJS only an allowlisted archive it builds itself (worksheet,
|
|
357
|
+
* shared-string, style, and core-property parts with generated content types and relationships).
|
|
358
|
+
*/
|
|
359
|
+
export const normalizeOfficeArchive = (
|
|
360
|
+
bytes: Uint8Array,
|
|
361
|
+
format: OfficeFileFormat,
|
|
362
|
+
limits: Partial<OfficeArchiveLimits> = {}
|
|
363
|
+
) =>
|
|
364
|
+
readOfficeArchive(bytes, format, {
|
|
365
|
+
maxArchiveEntries: limits.maxArchiveEntries ?? defaultFileExtractorLimits.maxArchiveEntries,
|
|
366
|
+
maxExpandedBytes: limits.maxExpandedBytes ?? defaultFileExtractorLimits.maxExpandedBytes,
|
|
367
|
+
maxInputBytes: limits.maxInputBytes ?? defaultFileExtractorLimits.maxInputBytes
|
|
368
|
+
}).pipe(Effect.map(normalized => storedArchive(normalized.parts)))
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
import { strFromU8 } from 'fflate'
|
|
2
|
+
import { decodeXmlEntities } from './xml-text.ts'
|
|
3
|
+
|
|
4
|
+
type PptxXmlFile = {
|
|
5
|
+
readonly fileName: string
|
|
6
|
+
readonly bytes: Uint8Array
|
|
7
|
+
readonly group: number
|
|
8
|
+
readonly index: number
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
const slideXmlFile = /^ppt\/slides\/slide(\d+)\.xml$/
|
|
12
|
+
|
|
13
|
+
const notesXmlFile = /^ppt\/notesSlides\/notesSlide(\d+)\.xml$/
|
|
14
|
+
|
|
15
|
+
const xmlName = '[A-Za-z_][\\w.-]*'
|
|
16
|
+
|
|
17
|
+
const optionalXmlPrefix = `(?:${xmlName}:)?`
|
|
18
|
+
|
|
19
|
+
// Every pattern stops at the next `<` (`[^<>]`), and elements are paired in one forward pass, so
|
|
20
|
+
// extraction stays linear in the part size even for unterminated or unbalanced markup.
|
|
21
|
+
type XmlElement = { readonly start: RegExp; readonly end: RegExp }
|
|
22
|
+
|
|
23
|
+
const xmlElement = (localName: string): XmlElement => ({
|
|
24
|
+
start: new RegExp(`<${optionalXmlPrefix}${localName}\\b[^<>]*>`, 'g'),
|
|
25
|
+
end: new RegExp(`</${optionalXmlPrefix}${localName}>`, 'g')
|
|
26
|
+
})
|
|
27
|
+
|
|
28
|
+
const paragraphXml = xmlElement('p')
|
|
29
|
+
|
|
30
|
+
const textXml = xmlElement('t')
|
|
31
|
+
|
|
32
|
+
const lineBreakXml = new RegExp(`<${optionalXmlPrefix}br\\b[^<>]*/>`, 'g')
|
|
33
|
+
|
|
34
|
+
const tabXml = new RegExp(`<${optionalXmlPrefix}tab\\b[^<>]*/>`, 'g')
|
|
35
|
+
|
|
36
|
+
type ElementMatch = {
|
|
37
|
+
/** Start tag through end tag. */
|
|
38
|
+
readonly outer: string
|
|
39
|
+
/** Text between the start and end tags. */
|
|
40
|
+
readonly inner: string
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** Each start tag paired with the next end tag after it (the old lazy `[\s\S]*?` match). */
|
|
44
|
+
const elementMatches = (xml: string, element: XmlElement): ReadonlyArray<ElementMatch> => {
|
|
45
|
+
const matches: Array<ElementMatch> = []
|
|
46
|
+
let position = 0
|
|
47
|
+
|
|
48
|
+
for (;;) {
|
|
49
|
+
element.start.lastIndex = position
|
|
50
|
+
const start = element.start.exec(xml)
|
|
51
|
+
|
|
52
|
+
if (start === null) return matches
|
|
53
|
+
|
|
54
|
+
const contentStart = start.index + start[0].length
|
|
55
|
+
element.end.lastIndex = contentStart
|
|
56
|
+
const end = element.end.exec(xml)
|
|
57
|
+
|
|
58
|
+
// No end tag after this start means none after any later start either.
|
|
59
|
+
if (end === null) return matches
|
|
60
|
+
|
|
61
|
+
position = end.index + end[0].length
|
|
62
|
+
matches.push({
|
|
63
|
+
outer: xml.slice(start.index, position),
|
|
64
|
+
inner: xml.slice(contentStart, end.index)
|
|
65
|
+
})
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
const indexedXmlFile = (
|
|
70
|
+
fileName: string,
|
|
71
|
+
bytes: Uint8Array,
|
|
72
|
+
pattern: RegExp,
|
|
73
|
+
group: number
|
|
74
|
+
): PptxXmlFile | undefined => {
|
|
75
|
+
const indexText = pattern.exec(fileName)?.[1]
|
|
76
|
+
|
|
77
|
+
if (indexText === undefined) return undefined
|
|
78
|
+
|
|
79
|
+
const index = Number.parseInt(indexText, 10)
|
|
80
|
+
|
|
81
|
+
if (!Number.isInteger(index)) return undefined
|
|
82
|
+
|
|
83
|
+
return { fileName, bytes, group, index }
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
const pptxXmlFile = (fileName: string, bytes: Uint8Array): PptxXmlFile | undefined =>
|
|
87
|
+
indexedXmlFile(fileName, bytes, slideXmlFile, 0) ??
|
|
88
|
+
indexedXmlFile(fileName, bytes, notesXmlFile, 1)
|
|
89
|
+
|
|
90
|
+
const comparePptxXmlFiles = (left: PptxXmlFile, right: PptxXmlFile) =>
|
|
91
|
+
left.group - right.group ||
|
|
92
|
+
left.index - right.index ||
|
|
93
|
+
left.fileName.localeCompare(right.fileName)
|
|
94
|
+
|
|
95
|
+
const extractParagraphText = (paragraph: string) => {
|
|
96
|
+
const xml = paragraph.replace(lineBreakXml, '<a:t>\n</a:t>').replace(tabXml, '<a:t>\t</a:t>')
|
|
97
|
+
|
|
98
|
+
return elementMatches(xml, textXml)
|
|
99
|
+
.map(match => decodeXmlEntities(match.inner))
|
|
100
|
+
.join('')
|
|
101
|
+
.trim()
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
const extractXmlText = (xml: string) => {
|
|
105
|
+
const paragraphs = elementMatches(xml, paragraphXml).map(match => match.outer)
|
|
106
|
+
const textSources = paragraphs.length > 0 ? paragraphs : [xml]
|
|
107
|
+
|
|
108
|
+
return textSources
|
|
109
|
+
.map(extractParagraphText)
|
|
110
|
+
.filter(text => text.length > 0)
|
|
111
|
+
.join('\n')
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/** Slide text in slide order, then speaker notes, from the parts of a validated archive. */
|
|
115
|
+
export const extractPptxText = (parts: Readonly<Record<string, Uint8Array>>) =>
|
|
116
|
+
Object.entries(parts)
|
|
117
|
+
.flatMap(([fileName, fileBytes]) => {
|
|
118
|
+
const xmlFile = pptxXmlFile(fileName, fileBytes)
|
|
119
|
+
|
|
120
|
+
return xmlFile === undefined ? [] : [xmlFile]
|
|
121
|
+
})
|
|
122
|
+
.sort(comparePptxXmlFiles)
|
|
123
|
+
.map(file => extractXmlText(strFromU8(file.bytes)))
|
|
124
|
+
.filter(text => text.length > 0)
|
|
125
|
+
.join('\n\n')
|