@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
import { FileExtractionError } from "../errors.mjs";
|
|
2
|
+
import { decodeWorkerMessage, revivedError } from "./extraction-worker-protocol.mjs";
|
|
3
|
+
import { makeSlotPool, processSlotPool, withSlot } from "./worker-admission.mjs";
|
|
4
|
+
import { Effect, Match, Option } from "effect";
|
|
5
|
+
import * as Schema from "effect/Schema";
|
|
6
|
+
import { Worker } from "node:worker_threads";
|
|
7
|
+
//#region src/node/extraction-isolation.ts
|
|
8
|
+
const PositiveSafeInteger = Schema.Int.pipe(Schema.check(Schema.isGreaterThan(0)), Schema.check(Schema.isLessThanOrEqualTo(Number.MAX_SAFE_INTEGER)));
|
|
9
|
+
/** Node's largest `setTimeout` delay; a larger one fires after 1 ms instead. */
|
|
10
|
+
const MaxTimerMs = PositiveSafeInteger.pipe(Schema.check(Schema.isLessThanOrEqualTo(2147483647)));
|
|
11
|
+
/** Resolved worker settings; every value is a positive integer, and timers fit `setTimeout`. */
|
|
12
|
+
const WorkerIsolationSettings = Schema.Struct({
|
|
13
|
+
maxOldGenerationSizeMb: PositiveSafeInteger,
|
|
14
|
+
maxYoungGenerationSizeMb: PositiveSafeInteger,
|
|
15
|
+
stackSizeMb: PositiveSafeInteger,
|
|
16
|
+
timeoutMs: MaxTimerMs,
|
|
17
|
+
maxConcurrentWorkers: PositiveSafeInteger.pipe(Schema.check(Schema.isLessThanOrEqualTo(4))),
|
|
18
|
+
maxQueueWaitMs: MaxTimerMs
|
|
19
|
+
});
|
|
20
|
+
/**
|
|
21
|
+
* Defaults. 256 MB of old generation holds SheetJS's cell objects for the default limits (100,000
|
|
22
|
+
* visited cells, 50 MiB expanded) several times over and PDF.js's working set for ordinary PDFs,
|
|
23
|
+
* and the realm-wide pool of four workers keeps their V8 heaps near 1 GB together. That bounds
|
|
24
|
+
* heap, not the process: Buffers and PDF.js's decoded data live outside it (see the README).
|
|
25
|
+
* 32 MB of young generation is twice V8's usual 64-bit default, enough for short-lived parser
|
|
26
|
+
* strings. 30 s is far above legitimate parse times at the default limits (well under a second
|
|
27
|
+
* for the research workbooks) and below common serverless request budgets; a queued extraction
|
|
28
|
+
* waits at most as long again for a slot.
|
|
29
|
+
*/
|
|
30
|
+
const defaultWorkerIsolation = {
|
|
31
|
+
maxOldGenerationSizeMb: 256,
|
|
32
|
+
maxYoungGenerationSizeMb: 32,
|
|
33
|
+
stackSizeMb: 4,
|
|
34
|
+
timeoutMs: 3e4,
|
|
35
|
+
maxConcurrentWorkers: 4,
|
|
36
|
+
maxQueueWaitMs: 3e4
|
|
37
|
+
};
|
|
38
|
+
const isolationModule = /\/(?:src\/node\/extraction-isolation\.ts|dist\/node\/extraction-isolation\.mjs)$/;
|
|
39
|
+
/**
|
|
40
|
+
* The built worker for an isolation module at `moduleUrl`: `dist/node/extraction-worker.mjs` of
|
|
41
|
+
* the same package, from `dist/node/extraction-isolation.mjs` (installed) or
|
|
42
|
+
* `src/node/extraction-isolation.ts` (workspace source, which must be built first). Any other
|
|
43
|
+
* location (this module bundled into a chunk) has no default: `undefined`, and nothing is started.
|
|
44
|
+
* The URL is derived from the module URL at runtime, never written as
|
|
45
|
+
* `new URL('./…', import.meta.url)`: bundlers rewrite that pattern into a copied asset.
|
|
46
|
+
*/
|
|
47
|
+
const extractionWorkerUrlFor = (moduleUrl) => isolationModule.test(moduleUrl) ? new URL(moduleUrl.replace(isolationModule, "/dist/node/extraction-worker.mjs")) : void 0;
|
|
48
|
+
/** The default worker entry for this module; see `extractionWorkerUrlFor`. */
|
|
49
|
+
const defaultExtractionWorkerUrl = () => extractionWorkerUrlFor(import.meta.url);
|
|
50
|
+
const stoppedMessages = {
|
|
51
|
+
"resource-limit": "File extraction exceeded its memory limit.",
|
|
52
|
+
timeout: "File extraction timed out.",
|
|
53
|
+
"worker-unavailable": "File extraction worker could not start.",
|
|
54
|
+
"worker-failed": "File extraction worker failed.",
|
|
55
|
+
busy: "File extraction is busy. Try again later."
|
|
56
|
+
};
|
|
57
|
+
const stopped = (format, reason, cause) => cause === void 0 ? new FileExtractionError({
|
|
58
|
+
format,
|
|
59
|
+
reason,
|
|
60
|
+
message: stoppedMessages[reason]
|
|
61
|
+
}) : new FileExtractionError({
|
|
62
|
+
format,
|
|
63
|
+
reason,
|
|
64
|
+
message: stoppedMessages[reason],
|
|
65
|
+
cause
|
|
66
|
+
});
|
|
67
|
+
const isOutOfMemory = (error) => error instanceof Error && "code" in error && error.code === "ERR_WORKER_OUT_OF_MEMORY";
|
|
68
|
+
/**
|
|
69
|
+
* Start a worker and attach its listeners in the same tick, so no `error` event can go unheard
|
|
70
|
+
* (an unheard worker `error` would throw in the host process).
|
|
71
|
+
*/
|
|
72
|
+
const startWorker = (workerUrl, options, format) => {
|
|
73
|
+
const worker = new Worker(workerUrl, options);
|
|
74
|
+
return {
|
|
75
|
+
worker,
|
|
76
|
+
outcome: new Promise((resolve) => {
|
|
77
|
+
let started = false;
|
|
78
|
+
worker.on("message", (raw) => Option.match(decodeWorkerMessage(raw), {
|
|
79
|
+
onNone: () => resolve(Effect.fail(stopped(format, "worker-failed"))),
|
|
80
|
+
onSome: (message) => Match.valueTags(message, {
|
|
81
|
+
WorkerStarted: () => {
|
|
82
|
+
started = true;
|
|
83
|
+
},
|
|
84
|
+
WorkerSucceeded: ({ file }) => resolve(Effect.succeed(file)),
|
|
85
|
+
WorkerFailed: ({ error }) => resolve(Effect.fail(revivedError(error))),
|
|
86
|
+
WorkerDefect: ({ message: defect }) => resolve(Effect.die(/* @__PURE__ */ new Error(`File extraction worker defect: ${defect}`)))
|
|
87
|
+
})
|
|
88
|
+
}));
|
|
89
|
+
worker.on("error", (error) => {
|
|
90
|
+
const reason = started ? "worker-failed" : "worker-unavailable";
|
|
91
|
+
resolve(Effect.fail(stopped(format, isOutOfMemory(error) ? "resource-limit" : reason, error)));
|
|
92
|
+
});
|
|
93
|
+
worker.on("exit", (code) => resolve(Effect.fail(stopped(format, started ? "worker-failed" : "worker-unavailable", /* @__PURE__ */ new Error(`Worker exited with code ${code}`)))));
|
|
94
|
+
})
|
|
95
|
+
};
|
|
96
|
+
};
|
|
97
|
+
/**
|
|
98
|
+
* Wait for the worker's outcome, at most `timeoutMs` of wall-clock time (a real timer, not the
|
|
99
|
+
* Effect `Clock`, so a test clock cannot stall it). The caller's scope terminates the worker
|
|
100
|
+
* whatever happens, including interruption.
|
|
101
|
+
*/
|
|
102
|
+
const awaitOutcome = ({ outcome }, format, timeoutMs) => Effect.callback((resume) => {
|
|
103
|
+
const timer = setTimeout(() => resume(Effect.fail(stopped(format, "timeout"))), timeoutMs);
|
|
104
|
+
outcome.then((result) => {
|
|
105
|
+
clearTimeout(timer);
|
|
106
|
+
resume(result);
|
|
107
|
+
});
|
|
108
|
+
return Effect.sync(() => clearTimeout(timer));
|
|
109
|
+
});
|
|
110
|
+
const missingWorker = /* @__PURE__ */ new Error("No default extraction worker next to this module; set isolation.workerUrl");
|
|
111
|
+
/**
|
|
112
|
+
* Run each extraction in a fresh worker: the input is copied into a transferred buffer, the
|
|
113
|
+
* worker returns only text and metadata, and the worker is terminated when the result arrives,
|
|
114
|
+
* the timeout fires, or the caller is interrupted. Admission goes through this layer's share and
|
|
115
|
+
* the realm-wide pool (`worker-admission.ts`); a slot is freed only once its worker has
|
|
116
|
+
* terminated. A worker that cannot start, or a missing `workerUrl`, fails closed with
|
|
117
|
+
* `reason: 'worker-unavailable'`; there is no in-process fallback.
|
|
118
|
+
*/
|
|
119
|
+
const makeWorkerExtractor = (settings, workerUrl) => Effect.sync(() => {
|
|
120
|
+
const layerPool = makeSlotPool(settings.maxConcurrentWorkers);
|
|
121
|
+
return (input, format, limits) => {
|
|
122
|
+
if (workerUrl === void 0) return Effect.fail(stopped(format, "worker-unavailable", missingWorker));
|
|
123
|
+
const extraction = Effect.gen(function* () {
|
|
124
|
+
const bytes = new Uint8Array(input.bytes);
|
|
125
|
+
const request = {
|
|
126
|
+
filename: input.filename,
|
|
127
|
+
mediaType: input.mediaType,
|
|
128
|
+
bytes,
|
|
129
|
+
format,
|
|
130
|
+
limits
|
|
131
|
+
};
|
|
132
|
+
return yield* awaitOutcome(yield* Effect.acquireRelease(Effect.try({
|
|
133
|
+
try: () => startWorker(workerUrl, {
|
|
134
|
+
workerData: request,
|
|
135
|
+
transferList: [bytes.buffer],
|
|
136
|
+
resourceLimits: {
|
|
137
|
+
maxOldGenerationSizeMb: settings.maxOldGenerationSizeMb,
|
|
138
|
+
maxYoungGenerationSizeMb: settings.maxYoungGenerationSizeMb,
|
|
139
|
+
stackSizeMb: settings.stackSizeMb
|
|
140
|
+
}
|
|
141
|
+
}, format),
|
|
142
|
+
catch: (cause) => stopped(format, "worker-unavailable", cause)
|
|
143
|
+
}), ({ worker }) => Effect.promise(() => worker.terminate())), format, settings.timeoutMs);
|
|
144
|
+
});
|
|
145
|
+
return Effect.suspend(() => {
|
|
146
|
+
const deadline = Date.now() + settings.maxQueueWaitMs;
|
|
147
|
+
const busy = () => stopped(format, "busy");
|
|
148
|
+
return Effect.scoped(extraction).pipe(withSlot(processSlotPool(), deadline, busy), withSlot(layerPool, deadline, busy));
|
|
149
|
+
});
|
|
150
|
+
};
|
|
151
|
+
});
|
|
152
|
+
//#endregion
|
|
153
|
+
export { WorkerIsolationSettings, defaultExtractionWorkerUrl, defaultWorkerIsolation, extractionWorkerUrlFor, makeWorkerExtractor };
|
|
154
|
+
|
|
155
|
+
//# sourceMappingURL=extraction-isolation.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extraction-isolation.mjs","names":[],"sources":["../../src/node/extraction-isolation.ts"],"sourcesContent":["import { Worker } from 'node:worker_threads'\nimport { Effect, Match, Option } from 'effect'\nimport * as Schema from 'effect/Schema'\nimport { FileExtractionError } from '../errors.ts'\nimport type { FileExtractionFailureReason, FileExtractorError } from '../errors.ts'\nimport type { ExtractedFile, FileInput } from '../format.ts'\nimport type { FileExtractorLimits } from '../limits.ts'\nimport type { ParsedFileFormat } from './extract-file.ts'\nimport { decodeWorkerMessage, revivedError } from './extraction-worker-protocol.ts'\nimport type { ExtractionWorkerRequest } from './extraction-worker-protocol.ts'\nimport { makeSlotPool, processSlotPool, processWorkerLimit, withSlot } from './worker-admission.ts'\n\n/** Limits for the worker each PDF, DOCX, XLSX, or PPTX extraction runs in. */\nexport type WorkerIsolationOptions = {\n /** V8 old-generation heap of the worker, in MB. Default 256. */\n readonly maxOldGenerationSizeMb?: number\n /** V8 young-generation heap of the worker, in MB. Default 32. */\n readonly maxYoungGenerationSizeMb?: number\n /** Stack of the worker's main thread, in MB. Default 4 (Node's own default). */\n readonly stackSizeMb?: number\n /**\n * Wall-clock time a worker may run before it is terminated, in ms. Default 30,000. At most\n * 2³¹−1 (Node's largest timer delay).\n */\n readonly timeoutMs?: number\n /**\n * This layer's share of the realm-wide worker pool: at most this many of its extractions run at\n * once. Every layer in a JavaScript realm (the main thread, or each worker thread that builds\n * the layer) shares one pool of 4 workers, however often the layer is built, so this can only\n * lower the layer's share. 1 to 4; default 4.\n */\n readonly maxConcurrentWorkers?: number\n /**\n * How long an extraction may wait for a worker slot, in ms, before it fails with\n * `reason: 'busy'` without starting a worker. Slots are handed out first come, first served, and\n * a slot freed after the deadline never admits the extraction. Default: the layer's `timeoutMs`.\n * At most 2³¹−1 (Node's largest timer delay).\n */\n readonly maxQueueWaitMs?: number\n /**\n * The worker entry. Default: the package's self-contained `dist/node/extraction-worker.mjs`,\n * found from this module's own location (in `dist`, or in `src` when a workspace or a bundler\n * that keeps module locations, such as Next.js Turbopack, runs the source). When this module has\n * been bundled into a file of another name, there is no default and extraction fails with\n * `reason: 'worker-unavailable'` without starting a worker; point this at a copy of\n * `@yolk-sdk/extractors/node/extraction-worker` instead.\n */\n readonly workerUrl?: string | URL\n}\n\n/**\n * Where parsers run. `'worker'` (the default) and an options object run each PDF, DOCX, XLSX, and\n * PPTX extraction in a fresh `worker_threads` worker with V8 heap, stack, and time limits.\n * `'none'` runs parsers in the calling thread: only for environments without worker threads, and\n * unsafe for untrusted input (a crafted file can exhaust the process heap or block the event loop).\n */\nexport type FileExtractorIsolation = 'worker' | 'none' | WorkerIsolationOptions\n\nconst PositiveSafeInteger = Schema.Int.pipe(\n Schema.check(Schema.isGreaterThan(0)),\n Schema.check(Schema.isLessThanOrEqualTo(Number.MAX_SAFE_INTEGER))\n)\n\n/** Node's largest `setTimeout` delay; a larger one fires after 1 ms instead. */\nconst MaxTimerMs = PositiveSafeInteger.pipe(Schema.check(Schema.isLessThanOrEqualTo(2_147_483_647)))\n\n/** Resolved worker settings; every value is a positive integer, and timers fit `setTimeout`. */\nexport const WorkerIsolationSettings = Schema.Struct({\n maxOldGenerationSizeMb: PositiveSafeInteger,\n maxYoungGenerationSizeMb: PositiveSafeInteger,\n stackSizeMb: PositiveSafeInteger,\n timeoutMs: MaxTimerMs,\n maxConcurrentWorkers: PositiveSafeInteger.pipe(\n Schema.check(Schema.isLessThanOrEqualTo(processWorkerLimit))\n ),\n maxQueueWaitMs: MaxTimerMs\n})\n\nexport type WorkerIsolationSettings = typeof WorkerIsolationSettings.Type\n\n/**\n * Defaults. 256 MB of old generation holds SheetJS's cell objects for the default limits (100,000\n * visited cells, 50 MiB expanded) several times over and PDF.js's working set for ordinary PDFs,\n * and the realm-wide pool of four workers keeps their V8 heaps near 1 GB together. That bounds\n * heap, not the process: Buffers and PDF.js's decoded data live outside it (see the README).\n * 32 MB of young generation is twice V8's usual 64-bit default, enough for short-lived parser\n * strings. 30 s is far above legitimate parse times at the default limits (well under a second\n * for the research workbooks) and below common serverless request budgets; a queued extraction\n * waits at most as long again for a slot.\n */\nexport const defaultWorkerIsolation: WorkerIsolationSettings = {\n maxOldGenerationSizeMb: 256,\n maxYoungGenerationSizeMb: 32,\n stackSizeMb: 4,\n timeoutMs: 30_000,\n maxConcurrentWorkers: processWorkerLimit,\n maxQueueWaitMs: 30_000\n}\n\nconst isolationModule =\n /\\/(?:src\\/node\\/extraction-isolation\\.ts|dist\\/node\\/extraction-isolation\\.mjs)$/\n\n/**\n * The built worker for an isolation module at `moduleUrl`: `dist/node/extraction-worker.mjs` of\n * the same package, from `dist/node/extraction-isolation.mjs` (installed) or\n * `src/node/extraction-isolation.ts` (workspace source, which must be built first). Any other\n * location (this module bundled into a chunk) has no default: `undefined`, and nothing is started.\n * The URL is derived from the module URL at runtime, never written as\n * `new URL('./…', import.meta.url)`: bundlers rewrite that pattern into a copied asset.\n */\nexport const extractionWorkerUrlFor = (moduleUrl: string): URL | undefined =>\n isolationModule.test(moduleUrl)\n ? new URL(moduleUrl.replace(isolationModule, '/dist/node/extraction-worker.mjs'))\n : undefined\n\n/** The default worker entry for this module; see `extractionWorkerUrlFor`. */\nexport const defaultExtractionWorkerUrl = () => extractionWorkerUrlFor(import.meta.url)\n\nconst stoppedMessages: Readonly<Record<FileExtractionFailureReason, string>> = {\n 'resource-limit': 'File extraction exceeded its memory limit.',\n timeout: 'File extraction timed out.',\n 'worker-unavailable': 'File extraction worker could not start.',\n 'worker-failed': 'File extraction worker failed.',\n busy: 'File extraction is busy. Try again later.'\n}\n\nconst stopped = (format: ParsedFileFormat, reason: FileExtractionFailureReason, cause?: unknown) =>\n cause === undefined\n ? new FileExtractionError({ format, reason, message: stoppedMessages[reason] })\n : new FileExtractionError({ format, reason, message: stoppedMessages[reason], cause })\n\nconst isOutOfMemory = (error: unknown) =>\n error instanceof Error && 'code' in error && error.code === 'ERR_WORKER_OUT_OF_MEMORY'\n\ntype WorkerOutcome = Effect.Effect<ExtractedFile, FileExtractorError>\n\ntype StartedWorker = {\n readonly worker: Worker\n /** Settles once with the worker's result or failure; listeners are attached at construction. */\n readonly outcome: Promise<WorkerOutcome>\n}\n\n/**\n * Start a worker and attach its listeners in the same tick, so no `error` event can go unheard\n * (an unheard worker `error` would throw in the host process).\n */\nconst startWorker = (\n workerUrl: string | URL,\n options: ConstructorParameters<typeof Worker>[1],\n format: ParsedFileFormat\n): StartedWorker => {\n const worker = new Worker(workerUrl, options)\n\n const outcome = new Promise<WorkerOutcome>(resolve => {\n let started = false\n\n worker.on('message', (raw: unknown) =>\n Option.match(decodeWorkerMessage(raw), {\n onNone: () => resolve(Effect.fail(stopped(format, 'worker-failed'))),\n onSome: message =>\n Match.valueTags(message, {\n WorkerStarted: () => {\n started = true\n },\n WorkerSucceeded: ({ file }) => resolve(Effect.succeed(file)),\n WorkerFailed: ({ error }) => resolve(Effect.fail(revivedError(error))),\n WorkerDefect: ({ message: defect }) =>\n resolve(Effect.die(new Error(`File extraction worker defect: ${defect}`)))\n })\n })\n )\n\n // `ERR_WORKER_OUT_OF_MEMORY` when a heap limit is hit; errors before `WorkerStarted` mean the\n // worker file or its imports could not load.\n worker.on('error', (error: unknown) => {\n const reason = started ? 'worker-failed' : 'worker-unavailable'\n\n resolve(Effect.fail(stopped(format, isOutOfMemory(error) ? 'resource-limit' : reason, error)))\n })\n\n worker.on('exit', (code: number) =>\n resolve(\n Effect.fail(\n stopped(\n format,\n started ? 'worker-failed' : 'worker-unavailable',\n new Error(`Worker exited with code ${code}`)\n )\n )\n )\n )\n })\n\n return { worker, outcome }\n}\n\n/**\n * Wait for the worker's outcome, at most `timeoutMs` of wall-clock time (a real timer, not the\n * Effect `Clock`, so a test clock cannot stall it). The caller's scope terminates the worker\n * whatever happens, including interruption.\n */\nconst awaitOutcome = ({ outcome }: StartedWorker, format: ParsedFileFormat, timeoutMs: number) =>\n Effect.callback<ExtractedFile, FileExtractorError>(resume => {\n const timer = setTimeout(() => resume(Effect.fail(stopped(format, 'timeout'))), timeoutMs)\n\n void outcome.then(result => {\n clearTimeout(timer)\n resume(result)\n })\n\n return Effect.sync(() => clearTimeout(timer))\n })\n\nexport type WorkerExtractor = (\n input: FileInput,\n format: ParsedFileFormat,\n limits: FileExtractorLimits\n) => Effect.Effect<ExtractedFile, FileExtractorError>\n\nconst missingWorker = new Error(\n 'No default extraction worker next to this module; set isolation.workerUrl'\n)\n\n/**\n * Run each extraction in a fresh worker: the input is copied into a transferred buffer, the\n * worker returns only text and metadata, and the worker is terminated when the result arrives,\n * the timeout fires, or the caller is interrupted. Admission goes through this layer's share and\n * the realm-wide pool (`worker-admission.ts`); a slot is freed only once its worker has\n * terminated. A worker that cannot start, or a missing `workerUrl`, fails closed with\n * `reason: 'worker-unavailable'`; there is no in-process fallback.\n */\nexport const makeWorkerExtractor = (\n settings: WorkerIsolationSettings,\n workerUrl: string | URL | undefined\n): Effect.Effect<WorkerExtractor> =>\n Effect.sync(() => {\n const layerPool = makeSlotPool(settings.maxConcurrentWorkers)\n\n return (input, format, limits) => {\n if (workerUrl === undefined)\n return Effect.fail(stopped(format, 'worker-unavailable', missingWorker))\n\n const extraction = Effect.gen(function* () {\n const bytes = new Uint8Array(input.bytes)\n\n const request: ExtractionWorkerRequest = {\n filename: input.filename,\n mediaType: input.mediaType,\n bytes,\n format,\n limits\n }\n\n const started = yield* Effect.acquireRelease(\n Effect.try({\n try: () =>\n startWorker(\n workerUrl,\n {\n workerData: request,\n transferList: [bytes.buffer],\n resourceLimits: {\n maxOldGenerationSizeMb: settings.maxOldGenerationSizeMb,\n maxYoungGenerationSizeMb: settings.maxYoungGenerationSizeMb,\n stackSizeMb: settings.stackSizeMb\n }\n },\n format\n ),\n catch: cause => stopped(format, 'worker-unavailable', cause)\n }),\n ({ worker }) => Effect.promise(() => worker.terminate())\n )\n\n return yield* awaitOutcome(started, format, settings.timeoutMs)\n })\n\n return Effect.suspend(() => {\n // One deadline for both waits, from when the extraction asks for a worker.\n const deadline = Date.now() + settings.maxQueueWaitMs\n const busy = () => stopped(format, 'busy')\n\n return Effect.scoped(extraction).pipe(\n withSlot(processSlotPool(), deadline, busy),\n withSlot(layerPool, deadline, busy)\n )\n })\n }\n })\n"],"mappings":";;;;;;;AA0DA,MAAM,sBAAsB,OAAO,IAAI,KACrC,OAAO,MAAM,OAAO,cAAc,CAAC,CAAC,GACpC,OAAO,MAAM,OAAO,oBAAoB,OAAO,gBAAgB,CAAC,CAClE;;AAGA,MAAM,aAAa,oBAAoB,KAAK,OAAO,MAAM,OAAO,oBAAoB,UAAa,CAAC,CAAC;;AAGnG,MAAa,0BAA0B,OAAO,OAAO;CACnD,wBAAwB;CACxB,0BAA0B;CAC1B,aAAa;CACb,WAAW;CACX,sBAAsB,oBAAoB,KACxC,OAAO,MAAM,OAAO,oBAAA,CAAsC,CAAC,CAC7D;CACA,gBAAgB;AAClB,CAAC;;;;;;;;;;;AAcD,MAAa,yBAAkD;CAC7D,wBAAwB;CACxB,0BAA0B;CAC1B,aAAa;CACb,WAAW;CACX,sBAAA;CACA,gBAAgB;AAClB;AAEA,MAAM,kBACJ;;;;;;;;;AAUF,MAAa,0BAA0B,cACrC,gBAAgB,KAAK,SAAS,IAC1B,IAAI,IAAI,UAAU,QAAQ,iBAAiB,kCAAkC,CAAC,IAC9E,KAAA;;AAGN,MAAa,mCAAmC,uBAAuB,OAAO,KAAK,GAAG;AAEtF,MAAM,kBAAyE;CAC7E,kBAAkB;CAClB,SAAS;CACT,sBAAsB;CACtB,iBAAiB;CACjB,MAAM;AACR;AAEA,MAAM,WAAW,QAA0B,QAAqC,UAC9E,UAAU,KAAA,IACN,IAAI,oBAAoB;CAAE;CAAQ;CAAQ,SAAS,gBAAgB;AAAQ,CAAC,IAC5E,IAAI,oBAAoB;CAAE;CAAQ;CAAQ,SAAS,gBAAgB;CAAS;AAAM,CAAC;AAEzF,MAAM,iBAAiB,UACrB,iBAAiB,SAAS,UAAU,SAAS,MAAM,SAAS;;;;;AAc9D,MAAM,eACJ,WACA,SACA,WACkB;CAClB,MAAM,SAAS,IAAI,OAAO,WAAW,OAAO;CA0C5C,OAAO;EAAE;EAAQ,SAAA,IAxCG,SAAuB,YAAW;GACpD,IAAI,UAAU;GAEd,OAAO,GAAG,YAAY,QACpB,OAAO,MAAM,oBAAoB,GAAG,GAAG;IACrC,cAAc,QAAQ,OAAO,KAAK,QAAQ,QAAQ,eAAe,CAAC,CAAC;IACnE,SAAQ,YACN,MAAM,UAAU,SAAS;KACvB,qBAAqB;MACnB,UAAU;KACZ;KACA,kBAAkB,EAAE,WAAW,QAAQ,OAAO,QAAQ,IAAI,CAAC;KAC3D,eAAe,EAAE,YAAY,QAAQ,OAAO,KAAK,aAAa,KAAK,CAAC,CAAC;KACrE,eAAe,EAAE,SAAS,aACxB,QAAQ,OAAO,oBAAI,IAAI,MAAM,kCAAkC,QAAQ,CAAC,CAAC;IAC7E,CAAC;GACL,CAAC,CACH;GAIA,OAAO,GAAG,UAAU,UAAmB;IACrC,MAAM,SAAS,UAAU,kBAAkB;IAE3C,QAAQ,OAAO,KAAK,QAAQ,QAAQ,cAAc,KAAK,IAAI,mBAAmB,QAAQ,KAAK,CAAC,CAAC;GAC/F,CAAC;GAED,OAAO,GAAG,SAAS,SACjB,QACE,OAAO,KACL,QACE,QACA,UAAU,kBAAkB,sCAC5B,IAAI,MAAM,2BAA2B,MAAM,CAC7C,CACF,CACF,CACF;EACF,CAEuB;CAAE;AAC3B;;;;;;AAOA,MAAM,gBAAgB,EAAE,WAA0B,QAA0B,cAC1E,OAAO,UAA4C,WAAU;CAC3D,MAAM,QAAQ,iBAAiB,OAAO,OAAO,KAAK,QAAQ,QAAQ,SAAS,CAAC,CAAC,GAAG,SAAS;CAEzF,QAAa,MAAK,WAAU;EAC1B,aAAa,KAAK;EAClB,OAAO,MAAM;CACf,CAAC;CAED,OAAO,OAAO,WAAW,aAAa,KAAK,CAAC;AAC9C,CAAC;AAQH,MAAM,gCAAgB,IAAI,MACxB,2EACF;;;;;;;;;AAUA,MAAa,uBACX,UACA,cAEA,OAAO,WAAW;CAChB,MAAM,YAAY,aAAa,SAAS,oBAAoB;CAE5D,QAAQ,OAAO,QAAQ,WAAW;EAChC,IAAI,cAAc,KAAA,GAChB,OAAO,OAAO,KAAK,QAAQ,QAAQ,sBAAsB,aAAa,CAAC;EAEzE,MAAM,aAAa,OAAO,IAAI,aAAa;GACzC,MAAM,QAAQ,IAAI,WAAW,MAAM,KAAK;GAExC,MAAM,UAAmC;IACvC,UAAU,MAAM;IAChB,WAAW,MAAM;IACjB;IACA;IACA;GACF;GAuBA,OAAO,OAAO,aAAa,OArBJ,OAAO,eAC5B,OAAO,IAAI;IACT,WACE,YACE,WACA;KACE,YAAY;KACZ,cAAc,CAAC,MAAM,MAAM;KAC3B,gBAAgB;MACd,wBAAwB,SAAS;MACjC,0BAA0B,SAAS;MACnC,aAAa,SAAS;KACxB;IACF,GACA,MACF;IACF,QAAO,UAAS,QAAQ,QAAQ,sBAAsB,KAAK;GAC7D,CAAC,IACA,EAAE,aAAa,OAAO,cAAc,OAAO,UAAU,CAAC,CACzD,GAEoC,QAAQ,SAAS,SAAS;EAChE,CAAC;EAED,OAAO,OAAO,cAAc;GAE1B,MAAM,WAAW,KAAK,IAAI,IAAI,SAAS;GACvC,MAAM,aAAa,QAAQ,QAAQ,MAAM;GAEzC,OAAO,OAAO,OAAO,UAAU,EAAE,KAC/B,SAAS,gBAAgB,GAAG,UAAU,IAAI,GAC1C,SAAS,WAAW,UAAU,IAAI,CACpC;EACF,CAAC;CACH;AACF,CAAC"}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { FileExtractionError, FileExtractorError, SheetJsUnavailableError } from "../errors.mjs";
|
|
2
|
+
import { Option } from "effect";
|
|
3
|
+
import * as Schema from "effect/Schema";
|
|
4
|
+
|
|
5
|
+
//#region src/node/extraction-worker-protocol.d.ts
|
|
6
|
+
/**
|
|
7
|
+
* Messages between the Node layer and `extraction-worker.ts`. Both sides encode and decode with
|
|
8
|
+
* these schemas; only plain data crosses the thread boundary (no parser objects).
|
|
9
|
+
*/
|
|
10
|
+
declare const ParsedFileFormat: Schema.Literals<readonly ["pdf", "docx", "pptx", "xlsx"]>;
|
|
11
|
+
/** `workerData`: the input bytes arrive in a transferred buffer. */
|
|
12
|
+
declare const ExtractionWorkerRequest: Schema.Struct<{
|
|
13
|
+
readonly filename: Schema.String;
|
|
14
|
+
readonly mediaType: Schema.String;
|
|
15
|
+
readonly bytes: Schema.Uint8Array;
|
|
16
|
+
readonly format: Schema.Literals<readonly ["pdf", "docx", "pptx", "xlsx"]>;
|
|
17
|
+
readonly limits: Schema.Struct<{
|
|
18
|
+
readonly maxInputBytes: Schema.Int;
|
|
19
|
+
readonly maxArchiveEntries: Schema.Int;
|
|
20
|
+
readonly maxExpandedBytes: Schema.Int;
|
|
21
|
+
readonly maxXlsxSheets: Schema.Int;
|
|
22
|
+
readonly maxXlsxCellVisits: Schema.Int;
|
|
23
|
+
readonly maxXlsxTextCharacters: Schema.Int;
|
|
24
|
+
readonly maxXlsxHyperlinks: Schema.Int;
|
|
25
|
+
}>;
|
|
26
|
+
}>;
|
|
27
|
+
type ExtractionWorkerRequest = typeof ExtractionWorkerRequest.Type;
|
|
28
|
+
declare const WorkerStarted_base: Schema.Class<WorkerStarted, Schema.TaggedStruct<"WorkerStarted", {}>, {}>;
|
|
29
|
+
/** Posted once the worker's modules have loaded. */
|
|
30
|
+
declare class WorkerStarted extends WorkerStarted_base {}
|
|
31
|
+
declare const WorkerSucceeded_base: Schema.Class<WorkerSucceeded, Schema.TaggedStruct<"WorkerSucceeded", {
|
|
32
|
+
readonly file: Schema.Struct<{
|
|
33
|
+
readonly content: Schema.String;
|
|
34
|
+
readonly metadata: Schema.Struct<{
|
|
35
|
+
readonly format: Schema.Literals<readonly ["csv", "docx", "json", "markdown", "pdf", "pptx", "text", "xlsx"]>;
|
|
36
|
+
readonly title: Schema.optional<Schema.String>;
|
|
37
|
+
readonly pageCount: Schema.optional<Schema.Number>;
|
|
38
|
+
readonly sheetNames: Schema.optional<Schema.$Array<Schema.String>>;
|
|
39
|
+
}>;
|
|
40
|
+
}>;
|
|
41
|
+
}>, {}>;
|
|
42
|
+
declare class WorkerSucceeded extends WorkerSucceeded_base {}
|
|
43
|
+
declare const WorkerFailed_base: Schema.Class<WorkerFailed, Schema.TaggedStruct<"WorkerFailed", {
|
|
44
|
+
readonly error: Schema.Union<readonly [typeof FileExtractionError, typeof SheetJsUnavailableError]>;
|
|
45
|
+
}>, {}>;
|
|
46
|
+
declare class WorkerFailed extends WorkerFailed_base {}
|
|
47
|
+
declare const WorkerDefect_base: Schema.Class<WorkerDefect, Schema.TaggedStruct<"WorkerDefect", {
|
|
48
|
+
readonly message: Schema.String;
|
|
49
|
+
}>, {}>;
|
|
50
|
+
/** An unexpected defect inside the worker (a bug, not a property of the file). */
|
|
51
|
+
declare class WorkerDefect extends WorkerDefect_base {}
|
|
52
|
+
declare const ExtractionWorkerMessage: Schema.Union<readonly [typeof WorkerStarted, typeof WorkerSucceeded, typeof WorkerFailed, typeof WorkerDefect]>;
|
|
53
|
+
declare const decodeWorkerMessage: (input: unknown, options?: import("effect/SchemaAST").ParseOptions) => Option.Option<WorkerStarted | WorkerSucceeded | WorkerFailed | WorkerDefect>;
|
|
54
|
+
/** The worker's reply for a failed extraction. */
|
|
55
|
+
declare const failureMessage: (error: FileExtractorError) => WorkerFailed | WorkerDefect;
|
|
56
|
+
/** The error a worker reported, with its archive cause revived. */
|
|
57
|
+
declare const revivedError: (error: WorkerFailed["error"]) => FileExtractionError | SheetJsUnavailableError;
|
|
58
|
+
//#endregion
|
|
59
|
+
export { ExtractionWorkerMessage, ExtractionWorkerRequest, ParsedFileFormat, WorkerDefect, WorkerFailed, WorkerStarted, WorkerSucceeded, decodeWorkerMessage, failureMessage, revivedError };
|
|
60
|
+
//# sourceMappingURL=extraction-worker-protocol.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extraction-worker-protocol.d.mts","names":[],"sources":["../../src/node/extraction-worker-protocol.ts"],"mappings":";;;;;;AAYA;;;cAAa,gBAAA,EAAgB,MAAA,CAAA,QAAA;AAAA;AAAA,cAGhB,uBAAA,EAAuB,MAAA,CAAA,MAAA;EAAA;;;;;;;;;;;;;;KAQxB,uBAAA,UAAiC,uBAAA,CAAwB,IAAI;AAAA,cAAA,kBAAA;;cAa5D,aAAA,SAAsB,kBAAwD;AAAA,cAAG,oBAAA;;;;;;;;;;;cAEjF,eAAA,SAAwB,oBAEnC;AAAA,cAAG,iBAAA;;;cAEQ,YAAA,SAAqB,iBAEhC;AAAA,cAAG,iBAAA;;;;cAGQ,YAAA,SAAqB,iBAEhC;AAAA,cAEW,uBAAA,EAAuB,MAAA,CAAA,KAAA,kBAAA,aAAA,SAAA,eAAA,SAAA,YAAA,SAAA,YAAA;AAAA,cAOvB,mBAAA,GAAmB,KAAA,WAAA,OAAA,8BAAA,YAAA,KAAA,MAAA,CAAA,MAAA,CAAA,aAAA,GAAA,eAAA,GAAA,YAAA,GAAA,YAAA;;cAmBnB,cAAA,GAAkB,KAAA,EAAO,kBAAA,KAAqB,YAAA,GAAe,YAAA;;cAqC7D,YAAA,GAAgB,KAAA,EAAO,YAAA,cAAqB,mBAAA,GAAA,uBAAA"}
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
import { FileExtractionError, OfficeArchiveError, SheetJsUnavailableError } from "../errors.mjs";
|
|
2
|
+
import { extractedFileFormats } from "../format.mjs";
|
|
3
|
+
import { FileExtractorLimits } from "../limits.mjs";
|
|
4
|
+
import { Match, Option } from "effect";
|
|
5
|
+
import * as Schema from "effect/Schema";
|
|
6
|
+
//#region src/node/extraction-worker-protocol.ts
|
|
7
|
+
/**
|
|
8
|
+
* Messages between the Node layer and `extraction-worker.ts`. Both sides encode and decode with
|
|
9
|
+
* these schemas; only plain data crosses the thread boundary (no parser objects).
|
|
10
|
+
*/
|
|
11
|
+
const ParsedFileFormat = Schema.Literals([
|
|
12
|
+
"pdf",
|
|
13
|
+
"docx",
|
|
14
|
+
"pptx",
|
|
15
|
+
"xlsx"
|
|
16
|
+
]);
|
|
17
|
+
/** `workerData`: the input bytes arrive in a transferred buffer. */
|
|
18
|
+
const ExtractionWorkerRequest = Schema.Struct({
|
|
19
|
+
filename: Schema.String,
|
|
20
|
+
mediaType: Schema.String,
|
|
21
|
+
bytes: Schema.Uint8Array,
|
|
22
|
+
format: ParsedFileFormat,
|
|
23
|
+
limits: FileExtractorLimits
|
|
24
|
+
});
|
|
25
|
+
const ExtractedFileMessage = Schema.Struct({
|
|
26
|
+
content: Schema.String,
|
|
27
|
+
metadata: Schema.Struct({
|
|
28
|
+
format: Schema.Literals(extractedFileFormats),
|
|
29
|
+
title: Schema.optional(Schema.String),
|
|
30
|
+
pageCount: Schema.optional(Schema.Number),
|
|
31
|
+
sheetNames: Schema.optional(Schema.Array(Schema.String))
|
|
32
|
+
})
|
|
33
|
+
});
|
|
34
|
+
/** Posted once the worker's modules have loaded. */
|
|
35
|
+
var WorkerStarted = class extends Schema.TaggedClass()("WorkerStarted", {}) {};
|
|
36
|
+
var WorkerSucceeded = class extends Schema.TaggedClass()("WorkerSucceeded", { file: ExtractedFileMessage }) {};
|
|
37
|
+
var WorkerFailed = class extends Schema.TaggedClass()("WorkerFailed", { error: Schema.Union([FileExtractionError, SheetJsUnavailableError]) }) {};
|
|
38
|
+
/** An unexpected defect inside the worker (a bug, not a property of the file). */
|
|
39
|
+
var WorkerDefect = class extends Schema.TaggedClass()("WorkerDefect", { message: Schema.String }) {};
|
|
40
|
+
const ExtractionWorkerMessage = Schema.Union([
|
|
41
|
+
WorkerStarted,
|
|
42
|
+
WorkerSucceeded,
|
|
43
|
+
WorkerFailed,
|
|
44
|
+
WorkerDefect
|
|
45
|
+
]);
|
|
46
|
+
const decodeWorkerMessage = Schema.decodeUnknownOption(ExtractionWorkerMessage);
|
|
47
|
+
/** A cause as it crosses the thread boundary: its name, message, and archive size, if any. */
|
|
48
|
+
const PortableCause = Schema.Struct({
|
|
49
|
+
name: Schema.String,
|
|
50
|
+
message: Schema.String,
|
|
51
|
+
expandedBytes: Schema.optional(Schema.Number)
|
|
52
|
+
});
|
|
53
|
+
const archiveErrorName = "OfficeArchiveError";
|
|
54
|
+
const portableCause = (cause) => {
|
|
55
|
+
if (cause instanceof OfficeArchiveError) return {
|
|
56
|
+
name: archiveErrorName,
|
|
57
|
+
message: cause.message,
|
|
58
|
+
expandedBytes: cause.expandedBytes
|
|
59
|
+
};
|
|
60
|
+
return cause instanceof Error ? {
|
|
61
|
+
name: cause.name,
|
|
62
|
+
message: cause.message
|
|
63
|
+
} : void 0;
|
|
64
|
+
};
|
|
65
|
+
/** The worker's reply for a failed extraction. */
|
|
66
|
+
const failureMessage = (error) => Match.valueTags(error, {
|
|
67
|
+
FileExtractionError: (failure) => WorkerFailed.make({ error: new FileExtractionError({
|
|
68
|
+
message: failure.message,
|
|
69
|
+
format: failure.format,
|
|
70
|
+
reason: failure.reason,
|
|
71
|
+
cause: portableCause(failure.cause)
|
|
72
|
+
}) }),
|
|
73
|
+
SheetJsUnavailableError: (failure) => WorkerFailed.make({ error: new SheetJsUnavailableError({
|
|
74
|
+
reason: failure.reason,
|
|
75
|
+
installedVersion: failure.installedVersion,
|
|
76
|
+
cause: portableCause(failure.cause)
|
|
77
|
+
}) }),
|
|
78
|
+
UnsupportedFileFormatError: () => WorkerDefect.make({ message: "The worker received an unsupported format" })
|
|
79
|
+
});
|
|
80
|
+
/** Turn a portable archive cause back into an `OfficeArchiveError`; other causes stay as sent. */
|
|
81
|
+
const revivedCause = (cause) => Option.match(Schema.decodeUnknownOption(PortableCause)(cause), {
|
|
82
|
+
onNone: () => cause,
|
|
83
|
+
onSome: (portable) => portable.name === archiveErrorName ? new OfficeArchiveError({
|
|
84
|
+
message: portable.message,
|
|
85
|
+
expandedBytes: portable.expandedBytes
|
|
86
|
+
}) : portable
|
|
87
|
+
});
|
|
88
|
+
/** The error a worker reported, with its archive cause revived. */
|
|
89
|
+
const revivedError = (error) => Match.valueTags(error, {
|
|
90
|
+
FileExtractionError: (failure) => new FileExtractionError({
|
|
91
|
+
message: failure.message,
|
|
92
|
+
format: failure.format,
|
|
93
|
+
reason: failure.reason,
|
|
94
|
+
cause: revivedCause(failure.cause)
|
|
95
|
+
}),
|
|
96
|
+
SheetJsUnavailableError: (failure) => new SheetJsUnavailableError({
|
|
97
|
+
reason: failure.reason,
|
|
98
|
+
installedVersion: failure.installedVersion,
|
|
99
|
+
cause: revivedCause(failure.cause)
|
|
100
|
+
})
|
|
101
|
+
});
|
|
102
|
+
//#endregion
|
|
103
|
+
export { ExtractionWorkerMessage, ExtractionWorkerRequest, ParsedFileFormat, WorkerDefect, WorkerFailed, WorkerStarted, WorkerSucceeded, decodeWorkerMessage, failureMessage, revivedError };
|
|
104
|
+
|
|
105
|
+
//# sourceMappingURL=extraction-worker-protocol.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extraction-worker-protocol.mjs","names":[],"sources":["../../src/node/extraction-worker-protocol.ts"],"sourcesContent":["import { Match, Option } from 'effect'\nimport * as Schema from 'effect/Schema'\nimport { FileExtractionError, OfficeArchiveError, SheetJsUnavailableError } from '../errors.ts'\nimport type { FileExtractorError } from '../errors.ts'\nimport { extractedFileFormats } from '../format.ts'\nimport { FileExtractorLimits } from '../limits.ts'\n\n/**\n * Messages between the Node layer and `extraction-worker.ts`. Both sides encode and decode with\n * these schemas; only plain data crosses the thread boundary (no parser objects).\n */\n\nexport const ParsedFileFormat = Schema.Literals(['pdf', 'docx', 'pptx', 'xlsx'])\n\n/** `workerData`: the input bytes arrive in a transferred buffer. */\nexport const ExtractionWorkerRequest = Schema.Struct({\n filename: Schema.String,\n mediaType: Schema.String,\n bytes: Schema.Uint8Array,\n format: ParsedFileFormat,\n limits: FileExtractorLimits\n})\n\nexport type ExtractionWorkerRequest = typeof ExtractionWorkerRequest.Type\n\nconst ExtractedFileMessage = Schema.Struct({\n content: Schema.String,\n metadata: Schema.Struct({\n format: Schema.Literals(extractedFileFormats),\n title: Schema.optional(Schema.String),\n pageCount: Schema.optional(Schema.Number),\n sheetNames: Schema.optional(Schema.Array(Schema.String))\n })\n})\n\n/** Posted once the worker's modules have loaded. */\nexport class WorkerStarted extends Schema.TaggedClass<WorkerStarted>()('WorkerStarted', {}) {}\n\nexport class WorkerSucceeded extends Schema.TaggedClass<WorkerSucceeded>()('WorkerSucceeded', {\n file: ExtractedFileMessage\n}) {}\n\nexport class WorkerFailed extends Schema.TaggedClass<WorkerFailed>()('WorkerFailed', {\n error: Schema.Union([FileExtractionError, SheetJsUnavailableError])\n}) {}\n\n/** An unexpected defect inside the worker (a bug, not a property of the file). */\nexport class WorkerDefect extends Schema.TaggedClass<WorkerDefect>()('WorkerDefect', {\n message: Schema.String\n}) {}\n\nexport const ExtractionWorkerMessage = Schema.Union([\n WorkerStarted,\n WorkerSucceeded,\n WorkerFailed,\n WorkerDefect\n])\n\nexport const decodeWorkerMessage = Schema.decodeUnknownOption(ExtractionWorkerMessage)\n\n/** A cause as it crosses the thread boundary: its name, message, and archive size, if any. */\nconst PortableCause = Schema.Struct({\n name: Schema.String,\n message: Schema.String,\n expandedBytes: Schema.optional(Schema.Number)\n})\n\nconst archiveErrorName = 'OfficeArchiveError'\n\nconst portableCause = (cause: unknown): typeof PortableCause.Type | undefined => {\n if (cause instanceof OfficeArchiveError)\n return { name: archiveErrorName, message: cause.message, expandedBytes: cause.expandedBytes }\n\n return cause instanceof Error ? { name: cause.name, message: cause.message } : undefined\n}\n\n/** The worker's reply for a failed extraction. */\nexport const failureMessage = (error: FileExtractorError): WorkerFailed | WorkerDefect =>\n Match.valueTags(error, {\n FileExtractionError: failure =>\n WorkerFailed.make({\n error: new FileExtractionError({\n message: failure.message,\n format: failure.format,\n reason: failure.reason,\n cause: portableCause(failure.cause)\n })\n }),\n SheetJsUnavailableError: failure =>\n WorkerFailed.make({\n error: new SheetJsUnavailableError({\n reason: failure.reason,\n installedVersion: failure.installedVersion,\n cause: portableCause(failure.cause)\n })\n }),\n UnsupportedFileFormatError: () =>\n WorkerDefect.make({ message: 'The worker received an unsupported format' })\n })\n\n/** Turn a portable archive cause back into an `OfficeArchiveError`; other causes stay as sent. */\nconst revivedCause = (cause: unknown) =>\n Option.match(Schema.decodeUnknownOption(PortableCause)(cause), {\n onNone: () => cause,\n onSome: portable =>\n portable.name === archiveErrorName\n ? new OfficeArchiveError({\n message: portable.message,\n expandedBytes: portable.expandedBytes\n })\n : portable\n })\n\n/** The error a worker reported, with its archive cause revived. */\nexport const revivedError = (error: WorkerFailed['error']) =>\n Match.valueTags(error, {\n FileExtractionError: failure =>\n new FileExtractionError({\n message: failure.message,\n format: failure.format,\n reason: failure.reason,\n cause: revivedCause(failure.cause)\n }),\n SheetJsUnavailableError: failure =>\n new SheetJsUnavailableError({\n reason: failure.reason,\n installedVersion: failure.installedVersion,\n cause: revivedCause(failure.cause)\n })\n })\n"],"mappings":";;;;;;;;;;AAYA,MAAa,mBAAmB,OAAO,SAAS;CAAC;CAAO;CAAQ;CAAQ;AAAM,CAAC;;AAG/E,MAAa,0BAA0B,OAAO,OAAO;CACnD,UAAU,OAAO;CACjB,WAAW,OAAO;CAClB,OAAO,OAAO;CACd,QAAQ;CACR,QAAQ;AACV,CAAC;AAID,MAAM,uBAAuB,OAAO,OAAO;CACzC,SAAS,OAAO;CAChB,UAAU,OAAO,OAAO;EACtB,QAAQ,OAAO,SAAS,oBAAoB;EAC5C,OAAO,OAAO,SAAS,OAAO,MAAM;EACpC,WAAW,OAAO,SAAS,OAAO,MAAM;EACxC,YAAY,OAAO,SAAS,OAAO,MAAM,OAAO,MAAM,CAAC;CACzD,CAAC;AACH,CAAC;;AAGD,IAAa,gBAAb,cAAmC,OAAO,YAA2B,EAAE,iBAAiB,CAAC,CAAC,EAAE,CAAC;AAE7F,IAAa,kBAAb,cAAqC,OAAO,YAA6B,EAAE,mBAAmB,EAC5F,MAAM,qBACR,CAAC,EAAE,CAAC;AAEJ,IAAa,eAAb,cAAkC,OAAO,YAA0B,EAAE,gBAAgB,EACnF,OAAO,OAAO,MAAM,CAAC,qBAAqB,uBAAuB,CAAC,EACpE,CAAC,EAAE,CAAC;;AAGJ,IAAa,eAAb,cAAkC,OAAO,YAA0B,EAAE,gBAAgB,EACnF,SAAS,OAAO,OAClB,CAAC,EAAE,CAAC;AAEJ,MAAa,0BAA0B,OAAO,MAAM;CAClD;CACA;CACA;CACA;AACF,CAAC;AAED,MAAa,sBAAsB,OAAO,oBAAoB,uBAAuB;;AAGrF,MAAM,gBAAgB,OAAO,OAAO;CAClC,MAAM,OAAO;CACb,SAAS,OAAO;CAChB,eAAe,OAAO,SAAS,OAAO,MAAM;AAC9C,CAAC;AAED,MAAM,mBAAmB;AAEzB,MAAM,iBAAiB,UAA0D;CAC/E,IAAI,iBAAiB,oBACnB,OAAO;EAAE,MAAM;EAAkB,SAAS,MAAM;EAAS,eAAe,MAAM;CAAc;CAE9F,OAAO,iBAAiB,QAAQ;EAAE,MAAM,MAAM;EAAM,SAAS,MAAM;CAAQ,IAAI,KAAA;AACjF;;AAGA,MAAa,kBAAkB,UAC7B,MAAM,UAAU,OAAO;CACrB,sBAAqB,YACnB,aAAa,KAAK,EAChB,OAAO,IAAI,oBAAoB;EAC7B,SAAS,QAAQ;EACjB,QAAQ,QAAQ;EAChB,QAAQ,QAAQ;EAChB,OAAO,cAAc,QAAQ,KAAK;CACpC,CAAC,EACH,CAAC;CACH,0BAAyB,YACvB,aAAa,KAAK,EAChB,OAAO,IAAI,wBAAwB;EACjC,QAAQ,QAAQ;EAChB,kBAAkB,QAAQ;EAC1B,OAAO,cAAc,QAAQ,KAAK;CACpC,CAAC,EACH,CAAC;CACH,kCACE,aAAa,KAAK,EAAE,SAAS,4CAA4C,CAAC;AAC9E,CAAC;;AAGH,MAAM,gBAAgB,UACpB,OAAO,MAAM,OAAO,oBAAoB,aAAa,EAAE,KAAK,GAAG;CAC7D,cAAc;CACd,SAAQ,aACN,SAAS,SAAS,mBACd,IAAI,mBAAmB;EACrB,SAAS,SAAS;EAClB,eAAe,SAAS;CAC1B,CAAC,IACD;AACR,CAAC;;AAGH,MAAa,gBAAgB,UAC3B,MAAM,UAAU,OAAO;CACrB,sBAAqB,YACnB,IAAI,oBAAoB;EACtB,SAAS,QAAQ;EACjB,QAAQ,QAAQ;EAChB,QAAQ,QAAQ;EAChB,OAAO,aAAa,QAAQ,KAAK;CACnC,CAAC;CACH,0BAAyB,YACvB,IAAI,wBAAwB;EAC1B,QAAQ,QAAQ;EAChB,kBAAkB,QAAQ;EAC1B,OAAO,aAAa,QAAQ,KAAK;CACnC,CAAC;AACL,CAAC"}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export { };
|