@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
@@ -0,0 +1,289 @@
1
+ import { Worker } from 'node:worker_threads'
2
+ import { Effect, Match, Option } from 'effect'
3
+ import * as Schema from 'effect/Schema'
4
+ import { FileExtractionError } from '../errors.ts'
5
+ import type { FileExtractionFailureReason, FileExtractorError } from '../errors.ts'
6
+ import type { ExtractedFile, FileInput } from '../format.ts'
7
+ import type { FileExtractorLimits } from '../limits.ts'
8
+ import type { ParsedFileFormat } from './extract-file.ts'
9
+ import { decodeWorkerMessage, revivedError } from './extraction-worker-protocol.ts'
10
+ import type { ExtractionWorkerRequest } from './extraction-worker-protocol.ts'
11
+ import { makeSlotPool, processSlotPool, processWorkerLimit, withSlot } from './worker-admission.ts'
12
+
13
+ /** Limits for the worker each PDF, DOCX, XLSX, or PPTX extraction runs in. */
14
+ export type WorkerIsolationOptions = {
15
+ /** V8 old-generation heap of the worker, in MB. Default 256. */
16
+ readonly maxOldGenerationSizeMb?: number
17
+ /** V8 young-generation heap of the worker, in MB. Default 32. */
18
+ readonly maxYoungGenerationSizeMb?: number
19
+ /** Stack of the worker's main thread, in MB. Default 4 (Node's own default). */
20
+ readonly stackSizeMb?: number
21
+ /**
22
+ * Wall-clock time a worker may run before it is terminated, in ms. Default 30,000. At most
23
+ * 2³¹−1 (Node's largest timer delay).
24
+ */
25
+ readonly timeoutMs?: number
26
+ /**
27
+ * This layer's share of the realm-wide worker pool: at most this many of its extractions run at
28
+ * once. Every layer in a JavaScript realm (the main thread, or each worker thread that builds
29
+ * the layer) shares one pool of 4 workers, however often the layer is built, so this can only
30
+ * lower the layer's share. 1 to 4; default 4.
31
+ */
32
+ readonly maxConcurrentWorkers?: number
33
+ /**
34
+ * How long an extraction may wait for a worker slot, in ms, before it fails with
35
+ * `reason: 'busy'` without starting a worker. Slots are handed out first come, first served, and
36
+ * a slot freed after the deadline never admits the extraction. Default: the layer's `timeoutMs`.
37
+ * At most 2³¹−1 (Node's largest timer delay).
38
+ */
39
+ readonly maxQueueWaitMs?: number
40
+ /**
41
+ * The worker entry. Default: the package's self-contained `dist/node/extraction-worker.mjs`,
42
+ * found from this module's own location (in `dist`, or in `src` when a workspace or a bundler
43
+ * that keeps module locations, such as Next.js Turbopack, runs the source). When this module has
44
+ * been bundled into a file of another name, there is no default and extraction fails with
45
+ * `reason: 'worker-unavailable'` without starting a worker; point this at a copy of
46
+ * `@yolk-sdk/extractors/node/extraction-worker` instead.
47
+ */
48
+ readonly workerUrl?: string | URL
49
+ }
50
+
51
+ /**
52
+ * Where parsers run. `'worker'` (the default) and an options object run each PDF, DOCX, XLSX, and
53
+ * PPTX extraction in a fresh `worker_threads` worker with V8 heap, stack, and time limits.
54
+ * `'none'` runs parsers in the calling thread: only for environments without worker threads, and
55
+ * unsafe for untrusted input (a crafted file can exhaust the process heap or block the event loop).
56
+ */
57
+ export type FileExtractorIsolation = 'worker' | 'none' | WorkerIsolationOptions
58
+
59
+ const PositiveSafeInteger = Schema.Int.pipe(
60
+ Schema.check(Schema.isGreaterThan(0)),
61
+ Schema.check(Schema.isLessThanOrEqualTo(Number.MAX_SAFE_INTEGER))
62
+ )
63
+
64
+ /** Node's largest `setTimeout` delay; a larger one fires after 1 ms instead. */
65
+ const MaxTimerMs = PositiveSafeInteger.pipe(Schema.check(Schema.isLessThanOrEqualTo(2_147_483_647)))
66
+
67
+ /** Resolved worker settings; every value is a positive integer, and timers fit `setTimeout`. */
68
+ export const WorkerIsolationSettings = Schema.Struct({
69
+ maxOldGenerationSizeMb: PositiveSafeInteger,
70
+ maxYoungGenerationSizeMb: PositiveSafeInteger,
71
+ stackSizeMb: PositiveSafeInteger,
72
+ timeoutMs: MaxTimerMs,
73
+ maxConcurrentWorkers: PositiveSafeInteger.pipe(
74
+ Schema.check(Schema.isLessThanOrEqualTo(processWorkerLimit))
75
+ ),
76
+ maxQueueWaitMs: MaxTimerMs
77
+ })
78
+
79
+ export type WorkerIsolationSettings = typeof WorkerIsolationSettings.Type
80
+
81
+ /**
82
+ * Defaults. 256 MB of old generation holds SheetJS's cell objects for the default limits (100,000
83
+ * visited cells, 50 MiB expanded) several times over and PDF.js's working set for ordinary PDFs,
84
+ * and the realm-wide pool of four workers keeps their V8 heaps near 1 GB together. That bounds
85
+ * heap, not the process: Buffers and PDF.js's decoded data live outside it (see the README).
86
+ * 32 MB of young generation is twice V8's usual 64-bit default, enough for short-lived parser
87
+ * strings. 30 s is far above legitimate parse times at the default limits (well under a second
88
+ * for the research workbooks) and below common serverless request budgets; a queued extraction
89
+ * waits at most as long again for a slot.
90
+ */
91
+ export const defaultWorkerIsolation: WorkerIsolationSettings = {
92
+ maxOldGenerationSizeMb: 256,
93
+ maxYoungGenerationSizeMb: 32,
94
+ stackSizeMb: 4,
95
+ timeoutMs: 30_000,
96
+ maxConcurrentWorkers: processWorkerLimit,
97
+ maxQueueWaitMs: 30_000
98
+ }
99
+
100
+ const isolationModule =
101
+ /\/(?:src\/node\/extraction-isolation\.ts|dist\/node\/extraction-isolation\.mjs)$/
102
+
103
+ /**
104
+ * The built worker for an isolation module at `moduleUrl`: `dist/node/extraction-worker.mjs` of
105
+ * the same package, from `dist/node/extraction-isolation.mjs` (installed) or
106
+ * `src/node/extraction-isolation.ts` (workspace source, which must be built first). Any other
107
+ * location (this module bundled into a chunk) has no default: `undefined`, and nothing is started.
108
+ * The URL is derived from the module URL at runtime, never written as
109
+ * `new URL('./…', import.meta.url)`: bundlers rewrite that pattern into a copied asset.
110
+ */
111
+ export const extractionWorkerUrlFor = (moduleUrl: string): URL | undefined =>
112
+ isolationModule.test(moduleUrl)
113
+ ? new URL(moduleUrl.replace(isolationModule, '/dist/node/extraction-worker.mjs'))
114
+ : undefined
115
+
116
+ /** The default worker entry for this module; see `extractionWorkerUrlFor`. */
117
+ export const defaultExtractionWorkerUrl = () => extractionWorkerUrlFor(import.meta.url)
118
+
119
+ const stoppedMessages: Readonly<Record<FileExtractionFailureReason, string>> = {
120
+ 'resource-limit': 'File extraction exceeded its memory limit.',
121
+ timeout: 'File extraction timed out.',
122
+ 'worker-unavailable': 'File extraction worker could not start.',
123
+ 'worker-failed': 'File extraction worker failed.',
124
+ busy: 'File extraction is busy. Try again later.'
125
+ }
126
+
127
+ const stopped = (format: ParsedFileFormat, reason: FileExtractionFailureReason, cause?: unknown) =>
128
+ cause === undefined
129
+ ? new FileExtractionError({ format, reason, message: stoppedMessages[reason] })
130
+ : new FileExtractionError({ format, reason, message: stoppedMessages[reason], cause })
131
+
132
+ const isOutOfMemory = (error: unknown) =>
133
+ error instanceof Error && 'code' in error && error.code === 'ERR_WORKER_OUT_OF_MEMORY'
134
+
135
+ type WorkerOutcome = Effect.Effect<ExtractedFile, FileExtractorError>
136
+
137
+ type StartedWorker = {
138
+ readonly worker: Worker
139
+ /** Settles once with the worker's result or failure; listeners are attached at construction. */
140
+ readonly outcome: Promise<WorkerOutcome>
141
+ }
142
+
143
+ /**
144
+ * Start a worker and attach its listeners in the same tick, so no `error` event can go unheard
145
+ * (an unheard worker `error` would throw in the host process).
146
+ */
147
+ const startWorker = (
148
+ workerUrl: string | URL,
149
+ options: ConstructorParameters<typeof Worker>[1],
150
+ format: ParsedFileFormat
151
+ ): StartedWorker => {
152
+ const worker = new Worker(workerUrl, options)
153
+
154
+ const outcome = new Promise<WorkerOutcome>(resolve => {
155
+ let started = false
156
+
157
+ worker.on('message', (raw: unknown) =>
158
+ Option.match(decodeWorkerMessage(raw), {
159
+ onNone: () => resolve(Effect.fail(stopped(format, 'worker-failed'))),
160
+ onSome: message =>
161
+ Match.valueTags(message, {
162
+ WorkerStarted: () => {
163
+ started = true
164
+ },
165
+ WorkerSucceeded: ({ file }) => resolve(Effect.succeed(file)),
166
+ WorkerFailed: ({ error }) => resolve(Effect.fail(revivedError(error))),
167
+ WorkerDefect: ({ message: defect }) =>
168
+ resolve(Effect.die(new Error(`File extraction worker defect: ${defect}`)))
169
+ })
170
+ })
171
+ )
172
+
173
+ // `ERR_WORKER_OUT_OF_MEMORY` when a heap limit is hit; errors before `WorkerStarted` mean the
174
+ // worker file or its imports could not load.
175
+ worker.on('error', (error: unknown) => {
176
+ const reason = started ? 'worker-failed' : 'worker-unavailable'
177
+
178
+ resolve(Effect.fail(stopped(format, isOutOfMemory(error) ? 'resource-limit' : reason, error)))
179
+ })
180
+
181
+ worker.on('exit', (code: number) =>
182
+ resolve(
183
+ Effect.fail(
184
+ stopped(
185
+ format,
186
+ started ? 'worker-failed' : 'worker-unavailable',
187
+ new Error(`Worker exited with code ${code}`)
188
+ )
189
+ )
190
+ )
191
+ )
192
+ })
193
+
194
+ return { worker, outcome }
195
+ }
196
+
197
+ /**
198
+ * Wait for the worker's outcome, at most `timeoutMs` of wall-clock time (a real timer, not the
199
+ * Effect `Clock`, so a test clock cannot stall it). The caller's scope terminates the worker
200
+ * whatever happens, including interruption.
201
+ */
202
+ const awaitOutcome = ({ outcome }: StartedWorker, format: ParsedFileFormat, timeoutMs: number) =>
203
+ Effect.callback<ExtractedFile, FileExtractorError>(resume => {
204
+ const timer = setTimeout(() => resume(Effect.fail(stopped(format, 'timeout'))), timeoutMs)
205
+
206
+ void outcome.then(result => {
207
+ clearTimeout(timer)
208
+ resume(result)
209
+ })
210
+
211
+ return Effect.sync(() => clearTimeout(timer))
212
+ })
213
+
214
+ export type WorkerExtractor = (
215
+ input: FileInput,
216
+ format: ParsedFileFormat,
217
+ limits: FileExtractorLimits
218
+ ) => Effect.Effect<ExtractedFile, FileExtractorError>
219
+
220
+ const missingWorker = new Error(
221
+ 'No default extraction worker next to this module; set isolation.workerUrl'
222
+ )
223
+
224
+ /**
225
+ * Run each extraction in a fresh worker: the input is copied into a transferred buffer, the
226
+ * worker returns only text and metadata, and the worker is terminated when the result arrives,
227
+ * the timeout fires, or the caller is interrupted. Admission goes through this layer's share and
228
+ * the realm-wide pool (`worker-admission.ts`); a slot is freed only once its worker has
229
+ * terminated. A worker that cannot start, or a missing `workerUrl`, fails closed with
230
+ * `reason: 'worker-unavailable'`; there is no in-process fallback.
231
+ */
232
+ export const makeWorkerExtractor = (
233
+ settings: WorkerIsolationSettings,
234
+ workerUrl: string | URL | undefined
235
+ ): Effect.Effect<WorkerExtractor> =>
236
+ Effect.sync(() => {
237
+ const layerPool = makeSlotPool(settings.maxConcurrentWorkers)
238
+
239
+ return (input, format, limits) => {
240
+ if (workerUrl === undefined)
241
+ return Effect.fail(stopped(format, 'worker-unavailable', missingWorker))
242
+
243
+ const extraction = Effect.gen(function* () {
244
+ const bytes = new Uint8Array(input.bytes)
245
+
246
+ const request: ExtractionWorkerRequest = {
247
+ filename: input.filename,
248
+ mediaType: input.mediaType,
249
+ bytes,
250
+ format,
251
+ limits
252
+ }
253
+
254
+ const started = yield* Effect.acquireRelease(
255
+ Effect.try({
256
+ try: () =>
257
+ startWorker(
258
+ workerUrl,
259
+ {
260
+ workerData: request,
261
+ transferList: [bytes.buffer],
262
+ resourceLimits: {
263
+ maxOldGenerationSizeMb: settings.maxOldGenerationSizeMb,
264
+ maxYoungGenerationSizeMb: settings.maxYoungGenerationSizeMb,
265
+ stackSizeMb: settings.stackSizeMb
266
+ }
267
+ },
268
+ format
269
+ ),
270
+ catch: cause => stopped(format, 'worker-unavailable', cause)
271
+ }),
272
+ ({ worker }) => Effect.promise(() => worker.terminate())
273
+ )
274
+
275
+ return yield* awaitOutcome(started, format, settings.timeoutMs)
276
+ })
277
+
278
+ return Effect.suspend(() => {
279
+ // One deadline for both waits, from when the extraction asks for a worker.
280
+ const deadline = Date.now() + settings.maxQueueWaitMs
281
+ const busy = () => stopped(format, 'busy')
282
+
283
+ return Effect.scoped(extraction).pipe(
284
+ withSlot(processSlotPool(), deadline, busy),
285
+ withSlot(layerPool, deadline, busy)
286
+ )
287
+ })
288
+ }
289
+ })
@@ -0,0 +1,130 @@
1
+ import { Match, Option } from 'effect'
2
+ import * as Schema from 'effect/Schema'
3
+ import { FileExtractionError, OfficeArchiveError, SheetJsUnavailableError } from '../errors.ts'
4
+ import type { FileExtractorError } from '../errors.ts'
5
+ import { extractedFileFormats } from '../format.ts'
6
+ import { FileExtractorLimits } from '../limits.ts'
7
+
8
+ /**
9
+ * Messages between the Node layer and `extraction-worker.ts`. Both sides encode and decode with
10
+ * these schemas; only plain data crosses the thread boundary (no parser objects).
11
+ */
12
+
13
+ export const ParsedFileFormat = Schema.Literals(['pdf', 'docx', 'pptx', 'xlsx'])
14
+
15
+ /** `workerData`: the input bytes arrive in a transferred buffer. */
16
+ export const ExtractionWorkerRequest = Schema.Struct({
17
+ filename: Schema.String,
18
+ mediaType: Schema.String,
19
+ bytes: Schema.Uint8Array,
20
+ format: ParsedFileFormat,
21
+ limits: FileExtractorLimits
22
+ })
23
+
24
+ export type ExtractionWorkerRequest = typeof ExtractionWorkerRequest.Type
25
+
26
+ const ExtractedFileMessage = Schema.Struct({
27
+ content: Schema.String,
28
+ metadata: Schema.Struct({
29
+ format: Schema.Literals(extractedFileFormats),
30
+ title: Schema.optional(Schema.String),
31
+ pageCount: Schema.optional(Schema.Number),
32
+ sheetNames: Schema.optional(Schema.Array(Schema.String))
33
+ })
34
+ })
35
+
36
+ /** Posted once the worker's modules have loaded. */
37
+ export class WorkerStarted extends Schema.TaggedClass<WorkerStarted>()('WorkerStarted', {}) {}
38
+
39
+ export class WorkerSucceeded extends Schema.TaggedClass<WorkerSucceeded>()('WorkerSucceeded', {
40
+ file: ExtractedFileMessage
41
+ }) {}
42
+
43
+ export class WorkerFailed extends Schema.TaggedClass<WorkerFailed>()('WorkerFailed', {
44
+ error: Schema.Union([FileExtractionError, SheetJsUnavailableError])
45
+ }) {}
46
+
47
+ /** An unexpected defect inside the worker (a bug, not a property of the file). */
48
+ export class WorkerDefect extends Schema.TaggedClass<WorkerDefect>()('WorkerDefect', {
49
+ message: Schema.String
50
+ }) {}
51
+
52
+ export const ExtractionWorkerMessage = Schema.Union([
53
+ WorkerStarted,
54
+ WorkerSucceeded,
55
+ WorkerFailed,
56
+ WorkerDefect
57
+ ])
58
+
59
+ export const decodeWorkerMessage = Schema.decodeUnknownOption(ExtractionWorkerMessage)
60
+
61
+ /** A cause as it crosses the thread boundary: its name, message, and archive size, if any. */
62
+ const PortableCause = Schema.Struct({
63
+ name: Schema.String,
64
+ message: Schema.String,
65
+ expandedBytes: Schema.optional(Schema.Number)
66
+ })
67
+
68
+ const archiveErrorName = 'OfficeArchiveError'
69
+
70
+ const portableCause = (cause: unknown): typeof PortableCause.Type | undefined => {
71
+ if (cause instanceof OfficeArchiveError)
72
+ return { name: archiveErrorName, message: cause.message, expandedBytes: cause.expandedBytes }
73
+
74
+ return cause instanceof Error ? { name: cause.name, message: cause.message } : undefined
75
+ }
76
+
77
+ /** The worker's reply for a failed extraction. */
78
+ export const failureMessage = (error: FileExtractorError): WorkerFailed | WorkerDefect =>
79
+ Match.valueTags(error, {
80
+ FileExtractionError: failure =>
81
+ WorkerFailed.make({
82
+ error: new FileExtractionError({
83
+ message: failure.message,
84
+ format: failure.format,
85
+ reason: failure.reason,
86
+ cause: portableCause(failure.cause)
87
+ })
88
+ }),
89
+ SheetJsUnavailableError: failure =>
90
+ WorkerFailed.make({
91
+ error: new SheetJsUnavailableError({
92
+ reason: failure.reason,
93
+ installedVersion: failure.installedVersion,
94
+ cause: portableCause(failure.cause)
95
+ })
96
+ }),
97
+ UnsupportedFileFormatError: () =>
98
+ WorkerDefect.make({ message: 'The worker received an unsupported format' })
99
+ })
100
+
101
+ /** Turn a portable archive cause back into an `OfficeArchiveError`; other causes stay as sent. */
102
+ const revivedCause = (cause: unknown) =>
103
+ Option.match(Schema.decodeUnknownOption(PortableCause)(cause), {
104
+ onNone: () => cause,
105
+ onSome: portable =>
106
+ portable.name === archiveErrorName
107
+ ? new OfficeArchiveError({
108
+ message: portable.message,
109
+ expandedBytes: portable.expandedBytes
110
+ })
111
+ : portable
112
+ })
113
+
114
+ /** The error a worker reported, with its archive cause revived. */
115
+ export const revivedError = (error: WorkerFailed['error']) =>
116
+ Match.valueTags(error, {
117
+ FileExtractionError: failure =>
118
+ new FileExtractionError({
119
+ message: failure.message,
120
+ format: failure.format,
121
+ reason: failure.reason,
122
+ cause: revivedCause(failure.cause)
123
+ }),
124
+ SheetJsUnavailableError: failure =>
125
+ new SheetJsUnavailableError({
126
+ reason: failure.reason,
127
+ installedVersion: failure.installedVersion,
128
+ cause: revivedCause(failure.cause)
129
+ })
130
+ })
@@ -0,0 +1,56 @@
1
+ // Worker entry for isolated extraction (`@yolk-sdk/extractors/node/extraction-worker`). The Node
2
+ // layer starts it once per extraction with V8 resource limits; it reads the request from
3
+ // `workerData`, runs the parsers, posts one plain-data result, and exits. Importing it outside a
4
+ // worker thread does nothing. The build bundles it into one self-contained
5
+ // `dist/node/extraction-worker.mjs` (effect, fflate, mammoth, unpdf, and this package inlined);
6
+ // only the optional `xlsx` peer stays a dynamic import.
7
+ import { parentPort, workerData } from 'node:worker_threads'
8
+ import { Cause, Effect, Exit, Option } from 'effect'
9
+ import * as Schema from 'effect/Schema'
10
+ import { extractParsedFile } from './extract-file.ts'
11
+ import {
12
+ ExtractionWorkerMessage,
13
+ ExtractionWorkerRequest,
14
+ failureMessage,
15
+ WorkerDefect,
16
+ WorkerStarted,
17
+ WorkerSucceeded
18
+ } from './extraction-worker-protocol.ts'
19
+ import { defaultSheetJsLoader } from './sheetjs.ts'
20
+
21
+ const run = (port: NonNullable<typeof parentPort>) =>
22
+ Effect.gen(function* () {
23
+ const post = (message: typeof ExtractionWorkerMessage.Type) =>
24
+ Schema.encodeEffect(ExtractionWorkerMessage)(message).pipe(
25
+ Effect.orDie,
26
+ Effect.map(encoded => port.postMessage(encoded))
27
+ )
28
+
29
+ yield* post(WorkerStarted.make({}))
30
+
31
+ const exit = yield* Effect.exit(
32
+ Schema.decodeUnknownEffect(ExtractionWorkerRequest)(workerData).pipe(
33
+ Effect.orDie,
34
+ Effect.flatMap(request =>
35
+ extractParsedFile(
36
+ { filename: request.filename, mediaType: request.mediaType, bytes: request.bytes },
37
+ request.format,
38
+ request.limits,
39
+ // The worker always loads the installed SheetJS itself, with the version check.
40
+ defaultSheetJsLoader
41
+ )
42
+ )
43
+ )
44
+ )
45
+
46
+ if (Exit.isSuccess(exit)) return yield* post(WorkerSucceeded.make({ file: exit.value }))
47
+
48
+ return yield* post(
49
+ Option.match(Cause.findErrorOption(exit.cause), {
50
+ onSome: failureMessage,
51
+ onNone: () => WorkerDefect.make({ message: Cause.pretty(exit.cause) })
52
+ })
53
+ )
54
+ })
55
+
56
+ if (parentPort !== null) await Effect.runPromise(run(parentPort))
@@ -0,0 +1,21 @@
1
+ // Node-only implementation: PDF (unpdf), DOCX (mammoth), XLSX (SheetJS, an optional peer loaded
2
+ // lazily), PPTX (fflate), and bounded Office archive validation on node:zlib/node:stream. Parsers
3
+ // run in a `worker_threads` worker (`./extraction-worker`) unless `isolation: 'none'`.
4
+
5
+ export { FileExtractorLayer, makeFileExtractorLayer } from './live-layer.ts'
6
+
7
+ export type { FileExtractorOptions } from './live-layer.ts'
8
+
9
+ export { defaultWorkerIsolation } from './extraction-isolation.ts'
10
+
11
+ export type { FileExtractorIsolation, WorkerIsolationOptions } from './extraction-isolation.ts'
12
+
13
+ export { normalizeOfficeArchive } from './office-archive.ts'
14
+
15
+ export type { OfficeArchiveLimits } from './office-archive.ts'
16
+
17
+ export type { SheetJsLoader } from './sheetjs.ts'
18
+
19
+ export { FileExtractor } from '../service.ts'
20
+
21
+ export type { FileExtractorApi } from '../service.ts'
@@ -0,0 +1,136 @@
1
+ import { Effect, Layer } from 'effect'
2
+ import * as Schema from 'effect/Schema'
3
+ import { FileExtractionError, UnsupportedFileFormatError } from '../errors.ts'
4
+ import { fileFormatFor } from '../format.ts'
5
+ import { defaultFileExtractorLimits, FileExtractorLimits } from '../limits.ts'
6
+ import { FileExtractor } from '../service.ts'
7
+ import type { FileExtractorApi } from '../service.ts'
8
+ import { extractParsedFile, isParsedFileFormat, makeExtractedFile } from './extract-file.ts'
9
+ import type { ParsedFileFormat } from './extract-file.ts'
10
+ import {
11
+ defaultExtractionWorkerUrl,
12
+ defaultWorkerIsolation,
13
+ makeWorkerExtractor,
14
+ WorkerIsolationSettings
15
+ } from './extraction-isolation.ts'
16
+ import type { FileExtractorIsolation, WorkerExtractor } from './extraction-isolation.ts'
17
+ import { defaultSheetJsLoader } from './sheetjs.ts'
18
+ import type { SheetJsLoader } from './sheetjs.ts'
19
+
20
+ export type FileExtractorOptions = {
21
+ /** Override any default limit; see `defaultFileExtractorLimits`. */
22
+ readonly limits?: Partial<FileExtractorLimits>
23
+ /**
24
+ * Where parsers run (default `'worker'`): each PDF, DOCX, XLSX, and PPTX extraction runs in a
25
+ * fresh worker thread with V8 heap, stack, and time limits, admitted through a pool of 4 workers
26
+ * per JavaScript realm; pass an object to change them. `'none'` parses in the calling thread and is
27
+ * unsafe for untrusted input.
28
+ */
29
+ readonly isolation?: FileExtractorIsolation
30
+ /**
31
+ * Load SheetJS in the calling thread. Only with `isolation: 'none'` (a worker cannot receive a
32
+ * function and always imports the installed `xlsx` itself); any other isolation is a defect
33
+ * when the layer is built. Defaults to a lazy `import('xlsx')`. Missing, non-SheetJS, or
34
+ * pre-0.20.3 modules fail with `SheetJsUnavailableError`.
35
+ */
36
+ readonly loadSheetJs?: SheetJsLoader
37
+ }
38
+
39
+ const decodeText = (bytes: Uint8Array) => new TextDecoder('utf-8', { fatal: false }).decode(bytes)
40
+
41
+ type ParsedFileExtractor = WorkerExtractor
42
+
43
+ /** Build the Node `FileExtractor` from validated limits and the parser runner. */
44
+ const makeFileExtractor = (
45
+ limits: FileExtractorLimits,
46
+ extractParsed: ParsedFileExtractor
47
+ ): FileExtractorApi => ({
48
+ extract: input =>
49
+ Effect.gen(function* () {
50
+ const format = fileFormatFor(input)
51
+
52
+ if (format === undefined)
53
+ return yield* Effect.fail(
54
+ new UnsupportedFileFormatError({ filename: input.filename, mediaType: input.mediaType })
55
+ )
56
+
57
+ yield* Effect.annotateCurrentSpan({
58
+ 'file_extractor.format': format,
59
+ 'file_extractor.file_size': input.bytes.byteLength
60
+ })
61
+
62
+ if (input.bytes.byteLength > limits.maxInputBytes)
63
+ return yield* Effect.fail(
64
+ new FileExtractionError({ message: 'File exceeds the extraction size limit', format })
65
+ )
66
+
67
+ if (isParsedFileFormat(format)) return yield* extractParsed(input, format, limits)
68
+
69
+ // Text formats are only decoded and sanitized; no parser reads them.
70
+ return yield* makeExtractedFile(decodeText(input.bytes), { format, title: input.filename })
71
+ }).pipe(Effect.withSpan('FileExtractor.extract'))
72
+ })
73
+
74
+ const parsedFileExtractor = (
75
+ options: FileExtractorOptions
76
+ ): Effect.Effect<ParsedFileExtractor, Schema.SchemaError> => {
77
+ const isolation = options.isolation ?? 'worker'
78
+
79
+ if (isolation === 'none') {
80
+ const loader = options.loadSheetJs ?? defaultSheetJsLoader
81
+
82
+ return Effect.succeed((input, format: ParsedFileFormat, limits) =>
83
+ extractParsedFile(input, format, limits, loader)
84
+ )
85
+ }
86
+
87
+ if (options.loadSheetJs !== undefined)
88
+ return Effect.die(
89
+ new Error(
90
+ 'FileExtractorOptions.loadSheetJs runs SheetJS in the calling thread and needs isolation: "none"'
91
+ )
92
+ )
93
+
94
+ const overrides = isolation === 'worker' ? {} : isolation
95
+ const timeoutMs = overrides.timeoutMs ?? defaultWorkerIsolation.timeoutMs
96
+
97
+ return Schema.decodeUnknownEffect(WorkerIsolationSettings)({
98
+ maxOldGenerationSizeMb:
99
+ overrides.maxOldGenerationSizeMb ?? defaultWorkerIsolation.maxOldGenerationSizeMb,
100
+ maxYoungGenerationSizeMb:
101
+ overrides.maxYoungGenerationSizeMb ?? defaultWorkerIsolation.maxYoungGenerationSizeMb,
102
+ stackSizeMb: overrides.stackSizeMb ?? defaultWorkerIsolation.stackSizeMb,
103
+ timeoutMs,
104
+ maxConcurrentWorkers:
105
+ overrides.maxConcurrentWorkers ?? defaultWorkerIsolation.maxConcurrentWorkers,
106
+ maxQueueWaitMs: overrides.maxQueueWaitMs ?? timeoutMs
107
+ }).pipe(
108
+ Effect.flatMap(settings =>
109
+ makeWorkerExtractor(settings, overrides.workerUrl ?? defaultExtractionWorkerUrl())
110
+ )
111
+ )
112
+ }
113
+
114
+ /**
115
+ * Node `FileExtractor` layer with custom limits, isolation, or (in-process) SheetJS loader.
116
+ * Invalid limits or isolation settings are defects.
117
+ */
118
+ export const makeFileExtractorLayer = (options: FileExtractorOptions = {}) =>
119
+ Layer.effect(
120
+ FileExtractor,
121
+ Effect.gen(function* () {
122
+ const limits = yield* Schema.decodeUnknownEffect(FileExtractorLimits)({
123
+ ...defaultFileExtractorLimits,
124
+ ...options.limits
125
+ })
126
+
127
+ const extractParsed = yield* parsedFileExtractor(options)
128
+
129
+ return FileExtractor.of(makeFileExtractor(limits, extractParsed))
130
+ }).pipe(Effect.orDie)
131
+ )
132
+
133
+ /**
134
+ * Node `FileExtractor` layer with the default limits, worker isolation, and lazy SheetJS loading.
135
+ */
136
+ export const FileExtractorLayer = makeFileExtractorLayer()