@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
import { Worker } from 'node:worker_threads'
|
|
2
|
+
import { Effect, Match, Option } from 'effect'
|
|
3
|
+
import * as Schema from 'effect/Schema'
|
|
4
|
+
import { FileExtractionError } from '../errors.ts'
|
|
5
|
+
import type { FileExtractionFailureReason, FileExtractorError } from '../errors.ts'
|
|
6
|
+
import type { ExtractedFile, FileInput } from '../format.ts'
|
|
7
|
+
import type { FileExtractorLimits } from '../limits.ts'
|
|
8
|
+
import type { ParsedFileFormat } from './extract-file.ts'
|
|
9
|
+
import { decodeWorkerMessage, revivedError } from './extraction-worker-protocol.ts'
|
|
10
|
+
import type { ExtractionWorkerRequest } from './extraction-worker-protocol.ts'
|
|
11
|
+
import { makeSlotPool, processSlotPool, processWorkerLimit, withSlot } from './worker-admission.ts'
|
|
12
|
+
|
|
13
|
+
/** Limits for the worker each PDF, DOCX, XLSX, or PPTX extraction runs in. */
|
|
14
|
+
export type WorkerIsolationOptions = {
|
|
15
|
+
/** V8 old-generation heap of the worker, in MB. Default 256. */
|
|
16
|
+
readonly maxOldGenerationSizeMb?: number
|
|
17
|
+
/** V8 young-generation heap of the worker, in MB. Default 32. */
|
|
18
|
+
readonly maxYoungGenerationSizeMb?: number
|
|
19
|
+
/** Stack of the worker's main thread, in MB. Default 4 (Node's own default). */
|
|
20
|
+
readonly stackSizeMb?: number
|
|
21
|
+
/**
|
|
22
|
+
* Wall-clock time a worker may run before it is terminated, in ms. Default 30,000. At most
|
|
23
|
+
* 2³¹−1 (Node's largest timer delay).
|
|
24
|
+
*/
|
|
25
|
+
readonly timeoutMs?: number
|
|
26
|
+
/**
|
|
27
|
+
* This layer's share of the realm-wide worker pool: at most this many of its extractions run at
|
|
28
|
+
* once. Every layer in a JavaScript realm (the main thread, or each worker thread that builds
|
|
29
|
+
* the layer) shares one pool of 4 workers, however often the layer is built, so this can only
|
|
30
|
+
* lower the layer's share. 1 to 4; default 4.
|
|
31
|
+
*/
|
|
32
|
+
readonly maxConcurrentWorkers?: number
|
|
33
|
+
/**
|
|
34
|
+
* How long an extraction may wait for a worker slot, in ms, before it fails with
|
|
35
|
+
* `reason: 'busy'` without starting a worker. Slots are handed out first come, first served, and
|
|
36
|
+
* a slot freed after the deadline never admits the extraction. Default: the layer's `timeoutMs`.
|
|
37
|
+
* At most 2³¹−1 (Node's largest timer delay).
|
|
38
|
+
*/
|
|
39
|
+
readonly maxQueueWaitMs?: number
|
|
40
|
+
/**
|
|
41
|
+
* The worker entry. Default: the package's self-contained `dist/node/extraction-worker.mjs`,
|
|
42
|
+
* found from this module's own location (in `dist`, or in `src` when a workspace or a bundler
|
|
43
|
+
* that keeps module locations, such as Next.js Turbopack, runs the source). When this module has
|
|
44
|
+
* been bundled into a file of another name, there is no default and extraction fails with
|
|
45
|
+
* `reason: 'worker-unavailable'` without starting a worker; point this at a copy of
|
|
46
|
+
* `@yolk-sdk/extractors/node/extraction-worker` instead.
|
|
47
|
+
*/
|
|
48
|
+
readonly workerUrl?: string | URL
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Where parsers run. `'worker'` (the default) and an options object run each PDF, DOCX, XLSX, and
|
|
53
|
+
* PPTX extraction in a fresh `worker_threads` worker with V8 heap, stack, and time limits.
|
|
54
|
+
* `'none'` runs parsers in the calling thread: only for environments without worker threads, and
|
|
55
|
+
* unsafe for untrusted input (a crafted file can exhaust the process heap or block the event loop).
|
|
56
|
+
*/
|
|
57
|
+
export type FileExtractorIsolation = 'worker' | 'none' | WorkerIsolationOptions
|
|
58
|
+
|
|
59
|
+
const PositiveSafeInteger = Schema.Int.pipe(
|
|
60
|
+
Schema.check(Schema.isGreaterThan(0)),
|
|
61
|
+
Schema.check(Schema.isLessThanOrEqualTo(Number.MAX_SAFE_INTEGER))
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
/** Node's largest `setTimeout` delay; a larger one fires after 1 ms instead. */
|
|
65
|
+
const MaxTimerMs = PositiveSafeInteger.pipe(Schema.check(Schema.isLessThanOrEqualTo(2_147_483_647)))
|
|
66
|
+
|
|
67
|
+
/** Resolved worker settings; every value is a positive integer, and timers fit `setTimeout`. */
|
|
68
|
+
export const WorkerIsolationSettings = Schema.Struct({
|
|
69
|
+
maxOldGenerationSizeMb: PositiveSafeInteger,
|
|
70
|
+
maxYoungGenerationSizeMb: PositiveSafeInteger,
|
|
71
|
+
stackSizeMb: PositiveSafeInteger,
|
|
72
|
+
timeoutMs: MaxTimerMs,
|
|
73
|
+
maxConcurrentWorkers: PositiveSafeInteger.pipe(
|
|
74
|
+
Schema.check(Schema.isLessThanOrEqualTo(processWorkerLimit))
|
|
75
|
+
),
|
|
76
|
+
maxQueueWaitMs: MaxTimerMs
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
export type WorkerIsolationSettings = typeof WorkerIsolationSettings.Type
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Defaults. 256 MB of old generation holds SheetJS's cell objects for the default limits (100,000
|
|
83
|
+
* visited cells, 50 MiB expanded) several times over and PDF.js's working set for ordinary PDFs,
|
|
84
|
+
* and the realm-wide pool of four workers keeps their V8 heaps near 1 GB together. That bounds
|
|
85
|
+
* heap, not the process: Buffers and PDF.js's decoded data live outside it (see the README).
|
|
86
|
+
* 32 MB of young generation is twice V8's usual 64-bit default, enough for short-lived parser
|
|
87
|
+
* strings. 30 s is far above legitimate parse times at the default limits (well under a second
|
|
88
|
+
* for the research workbooks) and below common serverless request budgets; a queued extraction
|
|
89
|
+
* waits at most as long again for a slot.
|
|
90
|
+
*/
|
|
91
|
+
export const defaultWorkerIsolation: WorkerIsolationSettings = {
|
|
92
|
+
maxOldGenerationSizeMb: 256,
|
|
93
|
+
maxYoungGenerationSizeMb: 32,
|
|
94
|
+
stackSizeMb: 4,
|
|
95
|
+
timeoutMs: 30_000,
|
|
96
|
+
maxConcurrentWorkers: processWorkerLimit,
|
|
97
|
+
maxQueueWaitMs: 30_000
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
const isolationModule =
|
|
101
|
+
/\/(?:src\/node\/extraction-isolation\.ts|dist\/node\/extraction-isolation\.mjs)$/
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* The built worker for an isolation module at `moduleUrl`: `dist/node/extraction-worker.mjs` of
|
|
105
|
+
* the same package, from `dist/node/extraction-isolation.mjs` (installed) or
|
|
106
|
+
* `src/node/extraction-isolation.ts` (workspace source, which must be built first). Any other
|
|
107
|
+
* location (this module bundled into a chunk) has no default: `undefined`, and nothing is started.
|
|
108
|
+
* The URL is derived from the module URL at runtime, never written as
|
|
109
|
+
* `new URL('./…', import.meta.url)`: bundlers rewrite that pattern into a copied asset.
|
|
110
|
+
*/
|
|
111
|
+
export const extractionWorkerUrlFor = (moduleUrl: string): URL | undefined =>
|
|
112
|
+
isolationModule.test(moduleUrl)
|
|
113
|
+
? new URL(moduleUrl.replace(isolationModule, '/dist/node/extraction-worker.mjs'))
|
|
114
|
+
: undefined
|
|
115
|
+
|
|
116
|
+
/** The default worker entry for this module; see `extractionWorkerUrlFor`. */
|
|
117
|
+
export const defaultExtractionWorkerUrl = () => extractionWorkerUrlFor(import.meta.url)
|
|
118
|
+
|
|
119
|
+
const stoppedMessages: Readonly<Record<FileExtractionFailureReason, string>> = {
|
|
120
|
+
'resource-limit': 'File extraction exceeded its memory limit.',
|
|
121
|
+
timeout: 'File extraction timed out.',
|
|
122
|
+
'worker-unavailable': 'File extraction worker could not start.',
|
|
123
|
+
'worker-failed': 'File extraction worker failed.',
|
|
124
|
+
busy: 'File extraction is busy. Try again later.'
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
const stopped = (format: ParsedFileFormat, reason: FileExtractionFailureReason, cause?: unknown) =>
|
|
128
|
+
cause === undefined
|
|
129
|
+
? new FileExtractionError({ format, reason, message: stoppedMessages[reason] })
|
|
130
|
+
: new FileExtractionError({ format, reason, message: stoppedMessages[reason], cause })
|
|
131
|
+
|
|
132
|
+
const isOutOfMemory = (error: unknown) =>
|
|
133
|
+
error instanceof Error && 'code' in error && error.code === 'ERR_WORKER_OUT_OF_MEMORY'
|
|
134
|
+
|
|
135
|
+
type WorkerOutcome = Effect.Effect<ExtractedFile, FileExtractorError>
|
|
136
|
+
|
|
137
|
+
type StartedWorker = {
|
|
138
|
+
readonly worker: Worker
|
|
139
|
+
/** Settles once with the worker's result or failure; listeners are attached at construction. */
|
|
140
|
+
readonly outcome: Promise<WorkerOutcome>
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Start a worker and attach its listeners in the same tick, so no `error` event can go unheard
|
|
145
|
+
* (an unheard worker `error` would throw in the host process).
|
|
146
|
+
*/
|
|
147
|
+
const startWorker = (
|
|
148
|
+
workerUrl: string | URL,
|
|
149
|
+
options: ConstructorParameters<typeof Worker>[1],
|
|
150
|
+
format: ParsedFileFormat
|
|
151
|
+
): StartedWorker => {
|
|
152
|
+
const worker = new Worker(workerUrl, options)
|
|
153
|
+
|
|
154
|
+
const outcome = new Promise<WorkerOutcome>(resolve => {
|
|
155
|
+
let started = false
|
|
156
|
+
|
|
157
|
+
worker.on('message', (raw: unknown) =>
|
|
158
|
+
Option.match(decodeWorkerMessage(raw), {
|
|
159
|
+
onNone: () => resolve(Effect.fail(stopped(format, 'worker-failed'))),
|
|
160
|
+
onSome: message =>
|
|
161
|
+
Match.valueTags(message, {
|
|
162
|
+
WorkerStarted: () => {
|
|
163
|
+
started = true
|
|
164
|
+
},
|
|
165
|
+
WorkerSucceeded: ({ file }) => resolve(Effect.succeed(file)),
|
|
166
|
+
WorkerFailed: ({ error }) => resolve(Effect.fail(revivedError(error))),
|
|
167
|
+
WorkerDefect: ({ message: defect }) =>
|
|
168
|
+
resolve(Effect.die(new Error(`File extraction worker defect: ${defect}`)))
|
|
169
|
+
})
|
|
170
|
+
})
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
// `ERR_WORKER_OUT_OF_MEMORY` when a heap limit is hit; errors before `WorkerStarted` mean the
|
|
174
|
+
// worker file or its imports could not load.
|
|
175
|
+
worker.on('error', (error: unknown) => {
|
|
176
|
+
const reason = started ? 'worker-failed' : 'worker-unavailable'
|
|
177
|
+
|
|
178
|
+
resolve(Effect.fail(stopped(format, isOutOfMemory(error) ? 'resource-limit' : reason, error)))
|
|
179
|
+
})
|
|
180
|
+
|
|
181
|
+
worker.on('exit', (code: number) =>
|
|
182
|
+
resolve(
|
|
183
|
+
Effect.fail(
|
|
184
|
+
stopped(
|
|
185
|
+
format,
|
|
186
|
+
started ? 'worker-failed' : 'worker-unavailable',
|
|
187
|
+
new Error(`Worker exited with code ${code}`)
|
|
188
|
+
)
|
|
189
|
+
)
|
|
190
|
+
)
|
|
191
|
+
)
|
|
192
|
+
})
|
|
193
|
+
|
|
194
|
+
return { worker, outcome }
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/**
|
|
198
|
+
* Wait for the worker's outcome, at most `timeoutMs` of wall-clock time (a real timer, not the
|
|
199
|
+
* Effect `Clock`, so a test clock cannot stall it). The caller's scope terminates the worker
|
|
200
|
+
* whatever happens, including interruption.
|
|
201
|
+
*/
|
|
202
|
+
const awaitOutcome = ({ outcome }: StartedWorker, format: ParsedFileFormat, timeoutMs: number) =>
|
|
203
|
+
Effect.callback<ExtractedFile, FileExtractorError>(resume => {
|
|
204
|
+
const timer = setTimeout(() => resume(Effect.fail(stopped(format, 'timeout'))), timeoutMs)
|
|
205
|
+
|
|
206
|
+
void outcome.then(result => {
|
|
207
|
+
clearTimeout(timer)
|
|
208
|
+
resume(result)
|
|
209
|
+
})
|
|
210
|
+
|
|
211
|
+
return Effect.sync(() => clearTimeout(timer))
|
|
212
|
+
})
|
|
213
|
+
|
|
214
|
+
export type WorkerExtractor = (
|
|
215
|
+
input: FileInput,
|
|
216
|
+
format: ParsedFileFormat,
|
|
217
|
+
limits: FileExtractorLimits
|
|
218
|
+
) => Effect.Effect<ExtractedFile, FileExtractorError>
|
|
219
|
+
|
|
220
|
+
const missingWorker = new Error(
|
|
221
|
+
'No default extraction worker next to this module; set isolation.workerUrl'
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* Run each extraction in a fresh worker: the input is copied into a transferred buffer, the
|
|
226
|
+
* worker returns only text and metadata, and the worker is terminated when the result arrives,
|
|
227
|
+
* the timeout fires, or the caller is interrupted. Admission goes through this layer's share and
|
|
228
|
+
* the realm-wide pool (`worker-admission.ts`); a slot is freed only once its worker has
|
|
229
|
+
* terminated. A worker that cannot start, or a missing `workerUrl`, fails closed with
|
|
230
|
+
* `reason: 'worker-unavailable'`; there is no in-process fallback.
|
|
231
|
+
*/
|
|
232
|
+
export const makeWorkerExtractor = (
|
|
233
|
+
settings: WorkerIsolationSettings,
|
|
234
|
+
workerUrl: string | URL | undefined
|
|
235
|
+
): Effect.Effect<WorkerExtractor> =>
|
|
236
|
+
Effect.sync(() => {
|
|
237
|
+
const layerPool = makeSlotPool(settings.maxConcurrentWorkers)
|
|
238
|
+
|
|
239
|
+
return (input, format, limits) => {
|
|
240
|
+
if (workerUrl === undefined)
|
|
241
|
+
return Effect.fail(stopped(format, 'worker-unavailable', missingWorker))
|
|
242
|
+
|
|
243
|
+
const extraction = Effect.gen(function* () {
|
|
244
|
+
const bytes = new Uint8Array(input.bytes)
|
|
245
|
+
|
|
246
|
+
const request: ExtractionWorkerRequest = {
|
|
247
|
+
filename: input.filename,
|
|
248
|
+
mediaType: input.mediaType,
|
|
249
|
+
bytes,
|
|
250
|
+
format,
|
|
251
|
+
limits
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
const started = yield* Effect.acquireRelease(
|
|
255
|
+
Effect.try({
|
|
256
|
+
try: () =>
|
|
257
|
+
startWorker(
|
|
258
|
+
workerUrl,
|
|
259
|
+
{
|
|
260
|
+
workerData: request,
|
|
261
|
+
transferList: [bytes.buffer],
|
|
262
|
+
resourceLimits: {
|
|
263
|
+
maxOldGenerationSizeMb: settings.maxOldGenerationSizeMb,
|
|
264
|
+
maxYoungGenerationSizeMb: settings.maxYoungGenerationSizeMb,
|
|
265
|
+
stackSizeMb: settings.stackSizeMb
|
|
266
|
+
}
|
|
267
|
+
},
|
|
268
|
+
format
|
|
269
|
+
),
|
|
270
|
+
catch: cause => stopped(format, 'worker-unavailable', cause)
|
|
271
|
+
}),
|
|
272
|
+
({ worker }) => Effect.promise(() => worker.terminate())
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
return yield* awaitOutcome(started, format, settings.timeoutMs)
|
|
276
|
+
})
|
|
277
|
+
|
|
278
|
+
return Effect.suspend(() => {
|
|
279
|
+
// One deadline for both waits, from when the extraction asks for a worker.
|
|
280
|
+
const deadline = Date.now() + settings.maxQueueWaitMs
|
|
281
|
+
const busy = () => stopped(format, 'busy')
|
|
282
|
+
|
|
283
|
+
return Effect.scoped(extraction).pipe(
|
|
284
|
+
withSlot(processSlotPool(), deadline, busy),
|
|
285
|
+
withSlot(layerPool, deadline, busy)
|
|
286
|
+
)
|
|
287
|
+
})
|
|
288
|
+
}
|
|
289
|
+
})
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
import { Match, Option } from 'effect'
|
|
2
|
+
import * as Schema from 'effect/Schema'
|
|
3
|
+
import { FileExtractionError, OfficeArchiveError, SheetJsUnavailableError } from '../errors.ts'
|
|
4
|
+
import type { FileExtractorError } from '../errors.ts'
|
|
5
|
+
import { extractedFileFormats } from '../format.ts'
|
|
6
|
+
import { FileExtractorLimits } from '../limits.ts'
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Messages between the Node layer and `extraction-worker.ts`. Both sides encode and decode with
|
|
10
|
+
* these schemas; only plain data crosses the thread boundary (no parser objects).
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
export const ParsedFileFormat = Schema.Literals(['pdf', 'docx', 'pptx', 'xlsx'])
|
|
14
|
+
|
|
15
|
+
/** `workerData`: the input bytes arrive in a transferred buffer. */
|
|
16
|
+
export const ExtractionWorkerRequest = Schema.Struct({
|
|
17
|
+
filename: Schema.String,
|
|
18
|
+
mediaType: Schema.String,
|
|
19
|
+
bytes: Schema.Uint8Array,
|
|
20
|
+
format: ParsedFileFormat,
|
|
21
|
+
limits: FileExtractorLimits
|
|
22
|
+
})
|
|
23
|
+
|
|
24
|
+
export type ExtractionWorkerRequest = typeof ExtractionWorkerRequest.Type
|
|
25
|
+
|
|
26
|
+
const ExtractedFileMessage = Schema.Struct({
|
|
27
|
+
content: Schema.String,
|
|
28
|
+
metadata: Schema.Struct({
|
|
29
|
+
format: Schema.Literals(extractedFileFormats),
|
|
30
|
+
title: Schema.optional(Schema.String),
|
|
31
|
+
pageCount: Schema.optional(Schema.Number),
|
|
32
|
+
sheetNames: Schema.optional(Schema.Array(Schema.String))
|
|
33
|
+
})
|
|
34
|
+
})
|
|
35
|
+
|
|
36
|
+
/** Posted once the worker's modules have loaded. */
|
|
37
|
+
export class WorkerStarted extends Schema.TaggedClass<WorkerStarted>()('WorkerStarted', {}) {}
|
|
38
|
+
|
|
39
|
+
export class WorkerSucceeded extends Schema.TaggedClass<WorkerSucceeded>()('WorkerSucceeded', {
|
|
40
|
+
file: ExtractedFileMessage
|
|
41
|
+
}) {}
|
|
42
|
+
|
|
43
|
+
export class WorkerFailed extends Schema.TaggedClass<WorkerFailed>()('WorkerFailed', {
|
|
44
|
+
error: Schema.Union([FileExtractionError, SheetJsUnavailableError])
|
|
45
|
+
}) {}
|
|
46
|
+
|
|
47
|
+
/** An unexpected defect inside the worker (a bug, not a property of the file). */
|
|
48
|
+
export class WorkerDefect extends Schema.TaggedClass<WorkerDefect>()('WorkerDefect', {
|
|
49
|
+
message: Schema.String
|
|
50
|
+
}) {}
|
|
51
|
+
|
|
52
|
+
export const ExtractionWorkerMessage = Schema.Union([
|
|
53
|
+
WorkerStarted,
|
|
54
|
+
WorkerSucceeded,
|
|
55
|
+
WorkerFailed,
|
|
56
|
+
WorkerDefect
|
|
57
|
+
])
|
|
58
|
+
|
|
59
|
+
export const decodeWorkerMessage = Schema.decodeUnknownOption(ExtractionWorkerMessage)
|
|
60
|
+
|
|
61
|
+
/** A cause as it crosses the thread boundary: its name, message, and archive size, if any. */
|
|
62
|
+
const PortableCause = Schema.Struct({
|
|
63
|
+
name: Schema.String,
|
|
64
|
+
message: Schema.String,
|
|
65
|
+
expandedBytes: Schema.optional(Schema.Number)
|
|
66
|
+
})
|
|
67
|
+
|
|
68
|
+
const archiveErrorName = 'OfficeArchiveError'
|
|
69
|
+
|
|
70
|
+
const portableCause = (cause: unknown): typeof PortableCause.Type | undefined => {
|
|
71
|
+
if (cause instanceof OfficeArchiveError)
|
|
72
|
+
return { name: archiveErrorName, message: cause.message, expandedBytes: cause.expandedBytes }
|
|
73
|
+
|
|
74
|
+
return cause instanceof Error ? { name: cause.name, message: cause.message } : undefined
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** The worker's reply for a failed extraction. */
|
|
78
|
+
export const failureMessage = (error: FileExtractorError): WorkerFailed | WorkerDefect =>
|
|
79
|
+
Match.valueTags(error, {
|
|
80
|
+
FileExtractionError: failure =>
|
|
81
|
+
WorkerFailed.make({
|
|
82
|
+
error: new FileExtractionError({
|
|
83
|
+
message: failure.message,
|
|
84
|
+
format: failure.format,
|
|
85
|
+
reason: failure.reason,
|
|
86
|
+
cause: portableCause(failure.cause)
|
|
87
|
+
})
|
|
88
|
+
}),
|
|
89
|
+
SheetJsUnavailableError: failure =>
|
|
90
|
+
WorkerFailed.make({
|
|
91
|
+
error: new SheetJsUnavailableError({
|
|
92
|
+
reason: failure.reason,
|
|
93
|
+
installedVersion: failure.installedVersion,
|
|
94
|
+
cause: portableCause(failure.cause)
|
|
95
|
+
})
|
|
96
|
+
}),
|
|
97
|
+
UnsupportedFileFormatError: () =>
|
|
98
|
+
WorkerDefect.make({ message: 'The worker received an unsupported format' })
|
|
99
|
+
})
|
|
100
|
+
|
|
101
|
+
/** Turn a portable archive cause back into an `OfficeArchiveError`; other causes stay as sent. */
|
|
102
|
+
const revivedCause = (cause: unknown) =>
|
|
103
|
+
Option.match(Schema.decodeUnknownOption(PortableCause)(cause), {
|
|
104
|
+
onNone: () => cause,
|
|
105
|
+
onSome: portable =>
|
|
106
|
+
portable.name === archiveErrorName
|
|
107
|
+
? new OfficeArchiveError({
|
|
108
|
+
message: portable.message,
|
|
109
|
+
expandedBytes: portable.expandedBytes
|
|
110
|
+
})
|
|
111
|
+
: portable
|
|
112
|
+
})
|
|
113
|
+
|
|
114
|
+
/** The error a worker reported, with its archive cause revived. */
|
|
115
|
+
export const revivedError = (error: WorkerFailed['error']) =>
|
|
116
|
+
Match.valueTags(error, {
|
|
117
|
+
FileExtractionError: failure =>
|
|
118
|
+
new FileExtractionError({
|
|
119
|
+
message: failure.message,
|
|
120
|
+
format: failure.format,
|
|
121
|
+
reason: failure.reason,
|
|
122
|
+
cause: revivedCause(failure.cause)
|
|
123
|
+
}),
|
|
124
|
+
SheetJsUnavailableError: failure =>
|
|
125
|
+
new SheetJsUnavailableError({
|
|
126
|
+
reason: failure.reason,
|
|
127
|
+
installedVersion: failure.installedVersion,
|
|
128
|
+
cause: revivedCause(failure.cause)
|
|
129
|
+
})
|
|
130
|
+
})
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
// Worker entry for isolated extraction (`@yolk-sdk/extractors/node/extraction-worker`). The Node
|
|
2
|
+
// layer starts it once per extraction with V8 resource limits; it reads the request from
|
|
3
|
+
// `workerData`, runs the parsers, posts one plain-data result, and exits. Importing it outside a
|
|
4
|
+
// worker thread does nothing. The build bundles it into one self-contained
|
|
5
|
+
// `dist/node/extraction-worker.mjs` (effect, fflate, mammoth, unpdf, and this package inlined);
|
|
6
|
+
// only the optional `xlsx` peer stays a dynamic import.
|
|
7
|
+
import { parentPort, workerData } from 'node:worker_threads'
|
|
8
|
+
import { Cause, Effect, Exit, Option } from 'effect'
|
|
9
|
+
import * as Schema from 'effect/Schema'
|
|
10
|
+
import { extractParsedFile } from './extract-file.ts'
|
|
11
|
+
import {
|
|
12
|
+
ExtractionWorkerMessage,
|
|
13
|
+
ExtractionWorkerRequest,
|
|
14
|
+
failureMessage,
|
|
15
|
+
WorkerDefect,
|
|
16
|
+
WorkerStarted,
|
|
17
|
+
WorkerSucceeded
|
|
18
|
+
} from './extraction-worker-protocol.ts'
|
|
19
|
+
import { defaultSheetJsLoader } from './sheetjs.ts'
|
|
20
|
+
|
|
21
|
+
const run = (port: NonNullable<typeof parentPort>) =>
|
|
22
|
+
Effect.gen(function* () {
|
|
23
|
+
const post = (message: typeof ExtractionWorkerMessage.Type) =>
|
|
24
|
+
Schema.encodeEffect(ExtractionWorkerMessage)(message).pipe(
|
|
25
|
+
Effect.orDie,
|
|
26
|
+
Effect.map(encoded => port.postMessage(encoded))
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
yield* post(WorkerStarted.make({}))
|
|
30
|
+
|
|
31
|
+
const exit = yield* Effect.exit(
|
|
32
|
+
Schema.decodeUnknownEffect(ExtractionWorkerRequest)(workerData).pipe(
|
|
33
|
+
Effect.orDie,
|
|
34
|
+
Effect.flatMap(request =>
|
|
35
|
+
extractParsedFile(
|
|
36
|
+
{ filename: request.filename, mediaType: request.mediaType, bytes: request.bytes },
|
|
37
|
+
request.format,
|
|
38
|
+
request.limits,
|
|
39
|
+
// The worker always loads the installed SheetJS itself, with the version check.
|
|
40
|
+
defaultSheetJsLoader
|
|
41
|
+
)
|
|
42
|
+
)
|
|
43
|
+
)
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
if (Exit.isSuccess(exit)) return yield* post(WorkerSucceeded.make({ file: exit.value }))
|
|
47
|
+
|
|
48
|
+
return yield* post(
|
|
49
|
+
Option.match(Cause.findErrorOption(exit.cause), {
|
|
50
|
+
onSome: failureMessage,
|
|
51
|
+
onNone: () => WorkerDefect.make({ message: Cause.pretty(exit.cause) })
|
|
52
|
+
})
|
|
53
|
+
)
|
|
54
|
+
})
|
|
55
|
+
|
|
56
|
+
if (parentPort !== null) await Effect.runPromise(run(parentPort))
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
// Node-only implementation: PDF (unpdf), DOCX (mammoth), XLSX (SheetJS, an optional peer loaded
|
|
2
|
+
// lazily), PPTX (fflate), and bounded Office archive validation on node:zlib/node:stream. Parsers
|
|
3
|
+
// run in a `worker_threads` worker (`./extraction-worker`) unless `isolation: 'none'`.
|
|
4
|
+
|
|
5
|
+
export { FileExtractorLayer, makeFileExtractorLayer } from './live-layer.ts'
|
|
6
|
+
|
|
7
|
+
export type { FileExtractorOptions } from './live-layer.ts'
|
|
8
|
+
|
|
9
|
+
export { defaultWorkerIsolation } from './extraction-isolation.ts'
|
|
10
|
+
|
|
11
|
+
export type { FileExtractorIsolation, WorkerIsolationOptions } from './extraction-isolation.ts'
|
|
12
|
+
|
|
13
|
+
export { normalizeOfficeArchive } from './office-archive.ts'
|
|
14
|
+
|
|
15
|
+
export type { OfficeArchiveLimits } from './office-archive.ts'
|
|
16
|
+
|
|
17
|
+
export type { SheetJsLoader } from './sheetjs.ts'
|
|
18
|
+
|
|
19
|
+
export { FileExtractor } from '../service.ts'
|
|
20
|
+
|
|
21
|
+
export type { FileExtractorApi } from '../service.ts'
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
import { Effect, Layer } from 'effect'
|
|
2
|
+
import * as Schema from 'effect/Schema'
|
|
3
|
+
import { FileExtractionError, UnsupportedFileFormatError } from '../errors.ts'
|
|
4
|
+
import { fileFormatFor } from '../format.ts'
|
|
5
|
+
import { defaultFileExtractorLimits, FileExtractorLimits } from '../limits.ts'
|
|
6
|
+
import { FileExtractor } from '../service.ts'
|
|
7
|
+
import type { FileExtractorApi } from '../service.ts'
|
|
8
|
+
import { extractParsedFile, isParsedFileFormat, makeExtractedFile } from './extract-file.ts'
|
|
9
|
+
import type { ParsedFileFormat } from './extract-file.ts'
|
|
10
|
+
import {
|
|
11
|
+
defaultExtractionWorkerUrl,
|
|
12
|
+
defaultWorkerIsolation,
|
|
13
|
+
makeWorkerExtractor,
|
|
14
|
+
WorkerIsolationSettings
|
|
15
|
+
} from './extraction-isolation.ts'
|
|
16
|
+
import type { FileExtractorIsolation, WorkerExtractor } from './extraction-isolation.ts'
|
|
17
|
+
import { defaultSheetJsLoader } from './sheetjs.ts'
|
|
18
|
+
import type { SheetJsLoader } from './sheetjs.ts'
|
|
19
|
+
|
|
20
|
+
export type FileExtractorOptions = {
|
|
21
|
+
/** Override any default limit; see `defaultFileExtractorLimits`. */
|
|
22
|
+
readonly limits?: Partial<FileExtractorLimits>
|
|
23
|
+
/**
|
|
24
|
+
* Where parsers run (default `'worker'`): each PDF, DOCX, XLSX, and PPTX extraction runs in a
|
|
25
|
+
* fresh worker thread with V8 heap, stack, and time limits, admitted through a pool of 4 workers
|
|
26
|
+
* per JavaScript realm; pass an object to change them. `'none'` parses in the calling thread and is
|
|
27
|
+
* unsafe for untrusted input.
|
|
28
|
+
*/
|
|
29
|
+
readonly isolation?: FileExtractorIsolation
|
|
30
|
+
/**
|
|
31
|
+
* Load SheetJS in the calling thread. Only with `isolation: 'none'` (a worker cannot receive a
|
|
32
|
+
* function and always imports the installed `xlsx` itself); any other isolation is a defect
|
|
33
|
+
* when the layer is built. Defaults to a lazy `import('xlsx')`. Missing, non-SheetJS, or
|
|
34
|
+
* pre-0.20.3 modules fail with `SheetJsUnavailableError`.
|
|
35
|
+
*/
|
|
36
|
+
readonly loadSheetJs?: SheetJsLoader
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const decodeText = (bytes: Uint8Array) => new TextDecoder('utf-8', { fatal: false }).decode(bytes)
|
|
40
|
+
|
|
41
|
+
type ParsedFileExtractor = WorkerExtractor
|
|
42
|
+
|
|
43
|
+
/** Build the Node `FileExtractor` from validated limits and the parser runner. */
|
|
44
|
+
const makeFileExtractor = (
|
|
45
|
+
limits: FileExtractorLimits,
|
|
46
|
+
extractParsed: ParsedFileExtractor
|
|
47
|
+
): FileExtractorApi => ({
|
|
48
|
+
extract: input =>
|
|
49
|
+
Effect.gen(function* () {
|
|
50
|
+
const format = fileFormatFor(input)
|
|
51
|
+
|
|
52
|
+
if (format === undefined)
|
|
53
|
+
return yield* Effect.fail(
|
|
54
|
+
new UnsupportedFileFormatError({ filename: input.filename, mediaType: input.mediaType })
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
yield* Effect.annotateCurrentSpan({
|
|
58
|
+
'file_extractor.format': format,
|
|
59
|
+
'file_extractor.file_size': input.bytes.byteLength
|
|
60
|
+
})
|
|
61
|
+
|
|
62
|
+
if (input.bytes.byteLength > limits.maxInputBytes)
|
|
63
|
+
return yield* Effect.fail(
|
|
64
|
+
new FileExtractionError({ message: 'File exceeds the extraction size limit', format })
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
if (isParsedFileFormat(format)) return yield* extractParsed(input, format, limits)
|
|
68
|
+
|
|
69
|
+
// Text formats are only decoded and sanitized; no parser reads them.
|
|
70
|
+
return yield* makeExtractedFile(decodeText(input.bytes), { format, title: input.filename })
|
|
71
|
+
}).pipe(Effect.withSpan('FileExtractor.extract'))
|
|
72
|
+
})
|
|
73
|
+
|
|
74
|
+
const parsedFileExtractor = (
|
|
75
|
+
options: FileExtractorOptions
|
|
76
|
+
): Effect.Effect<ParsedFileExtractor, Schema.SchemaError> => {
|
|
77
|
+
const isolation = options.isolation ?? 'worker'
|
|
78
|
+
|
|
79
|
+
if (isolation === 'none') {
|
|
80
|
+
const loader = options.loadSheetJs ?? defaultSheetJsLoader
|
|
81
|
+
|
|
82
|
+
return Effect.succeed((input, format: ParsedFileFormat, limits) =>
|
|
83
|
+
extractParsedFile(input, format, limits, loader)
|
|
84
|
+
)
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
if (options.loadSheetJs !== undefined)
|
|
88
|
+
return Effect.die(
|
|
89
|
+
new Error(
|
|
90
|
+
'FileExtractorOptions.loadSheetJs runs SheetJS in the calling thread and needs isolation: "none"'
|
|
91
|
+
)
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
const overrides = isolation === 'worker' ? {} : isolation
|
|
95
|
+
const timeoutMs = overrides.timeoutMs ?? defaultWorkerIsolation.timeoutMs
|
|
96
|
+
|
|
97
|
+
return Schema.decodeUnknownEffect(WorkerIsolationSettings)({
|
|
98
|
+
maxOldGenerationSizeMb:
|
|
99
|
+
overrides.maxOldGenerationSizeMb ?? defaultWorkerIsolation.maxOldGenerationSizeMb,
|
|
100
|
+
maxYoungGenerationSizeMb:
|
|
101
|
+
overrides.maxYoungGenerationSizeMb ?? defaultWorkerIsolation.maxYoungGenerationSizeMb,
|
|
102
|
+
stackSizeMb: overrides.stackSizeMb ?? defaultWorkerIsolation.stackSizeMb,
|
|
103
|
+
timeoutMs,
|
|
104
|
+
maxConcurrentWorkers:
|
|
105
|
+
overrides.maxConcurrentWorkers ?? defaultWorkerIsolation.maxConcurrentWorkers,
|
|
106
|
+
maxQueueWaitMs: overrides.maxQueueWaitMs ?? timeoutMs
|
|
107
|
+
}).pipe(
|
|
108
|
+
Effect.flatMap(settings =>
|
|
109
|
+
makeWorkerExtractor(settings, overrides.workerUrl ?? defaultExtractionWorkerUrl())
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Node `FileExtractor` layer with custom limits, isolation, or (in-process) SheetJS loader.
|
|
116
|
+
* Invalid limits or isolation settings are defects.
|
|
117
|
+
*/
|
|
118
|
+
export const makeFileExtractorLayer = (options: FileExtractorOptions = {}) =>
|
|
119
|
+
Layer.effect(
|
|
120
|
+
FileExtractor,
|
|
121
|
+
Effect.gen(function* () {
|
|
122
|
+
const limits = yield* Schema.decodeUnknownEffect(FileExtractorLimits)({
|
|
123
|
+
...defaultFileExtractorLimits,
|
|
124
|
+
...options.limits
|
|
125
|
+
})
|
|
126
|
+
|
|
127
|
+
const extractParsed = yield* parsedFileExtractor(options)
|
|
128
|
+
|
|
129
|
+
return FileExtractor.of(makeFileExtractor(limits, extractParsed))
|
|
130
|
+
}).pipe(Effect.orDie)
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Node `FileExtractor` layer with the default limits, worker isolation, and lazy SheetJS loading.
|
|
135
|
+
*/
|
|
136
|
+
export const FileExtractorLayer = makeFileExtractorLayer()
|