@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
import { FileExtractor } from "./service.mjs";
|
|
2
|
+
import { Effect, Layer, Match, Predicate } from "effect";
|
|
3
|
+
import { KnowledgeExtractionError } from "@yolk-sdk/knowledge/errors";
|
|
4
|
+
import { KnowledgeExtractor } from "@yolk-sdk/knowledge/extraction";
|
|
5
|
+
//#region src/knowledge.ts
|
|
6
|
+
/** Percent-decode a path segment; malformed escapes (`%zz`, invalid UTF-8) stay encoded. */
|
|
7
|
+
const decodeSegment = (segment) => {
|
|
8
|
+
try {
|
|
9
|
+
return decodeURIComponent(segment);
|
|
10
|
+
} catch {
|
|
11
|
+
return segment;
|
|
12
|
+
}
|
|
13
|
+
};
|
|
14
|
+
const urlFilename = (url) => {
|
|
15
|
+
if (!URL.canParse(url)) return url;
|
|
16
|
+
const last = new URL(url).pathname.split("/").filter((segment) => segment.length > 0).at(-1);
|
|
17
|
+
return last === void 0 ? url : decodeSegment(last);
|
|
18
|
+
};
|
|
19
|
+
/** The filename used for format detection: file name or ref, URL path, or text label. */
|
|
20
|
+
const filenameFor = (source) => Match.value(source).pipe(Match.tagsExhaustive({
|
|
21
|
+
File: (file) => file.name ?? file.ref,
|
|
22
|
+
Url: (url) => urlFilename(url.url),
|
|
23
|
+
Text: (text) => text.label ?? "text"
|
|
24
|
+
}));
|
|
25
|
+
/** Loaded media type first, then the file source's, then `text/plain` for text sources. */
|
|
26
|
+
const mediaTypeFor = (loaded) => {
|
|
27
|
+
if (loaded.mediaType !== void 0) return loaded.mediaType;
|
|
28
|
+
if (Predicate.isTagged(loaded.source, "File") && loaded.source.mediaType !== void 0) return loaded.source.mediaType;
|
|
29
|
+
return Predicate.isTagged(loaded.source, "Text") ? "text/plain" : "";
|
|
30
|
+
};
|
|
31
|
+
const documentFrom = (loaded, extracted) => {
|
|
32
|
+
const { title, ...fileMetadata } = extracted.metadata;
|
|
33
|
+
const metadata = {
|
|
34
|
+
...fileMetadata,
|
|
35
|
+
...loaded.metadata
|
|
36
|
+
};
|
|
37
|
+
return title === void 0 || title.trim().length === 0 ? {
|
|
38
|
+
content: extracted.content,
|
|
39
|
+
metadata
|
|
40
|
+
} : {
|
|
41
|
+
content: extracted.content,
|
|
42
|
+
title: title.trim(),
|
|
43
|
+
metadata
|
|
44
|
+
};
|
|
45
|
+
};
|
|
46
|
+
/**
|
|
47
|
+
* A `KnowledgeExtractor` backed by a `FileExtractor`. String content is already text and passes
|
|
48
|
+
* through unchanged (it must not be blank); bytes are extracted with the format chosen from the
|
|
49
|
+
* source name and media type. File metadata (`format`, `pageCount`, `sheetNames`) is merged
|
|
50
|
+
* under the loaded source's own metadata, and the extracted title becomes the document title.
|
|
51
|
+
*/
|
|
52
|
+
const makeFileKnowledgeExtractor = (extractor) => ({ extract: (loaded) => {
|
|
53
|
+
if (Predicate.isString(loaded.content)) {
|
|
54
|
+
const content = loaded.content;
|
|
55
|
+
return content.trim().length === 0 ? Effect.fail(new KnowledgeExtractionError({ message: "Knowledge source text is empty" })) : Effect.succeed(loaded.metadata === void 0 ? { content } : {
|
|
56
|
+
content,
|
|
57
|
+
metadata: loaded.metadata
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
return extractor.extract({
|
|
61
|
+
filename: filenameFor(loaded.source),
|
|
62
|
+
mediaType: mediaTypeFor(loaded),
|
|
63
|
+
bytes: loaded.content
|
|
64
|
+
}).pipe(Effect.map((extracted) => documentFrom(loaded, extracted)), Effect.mapError((error) => new KnowledgeExtractionError({
|
|
65
|
+
message: error.message,
|
|
66
|
+
cause: error
|
|
67
|
+
})));
|
|
68
|
+
} });
|
|
69
|
+
/** Provide `KnowledgeExtractor` from the `FileExtractor` in context (for example the Node layer). */
|
|
70
|
+
const FileKnowledgeExtractorLayer = Layer.effect(KnowledgeExtractor, Effect.gen(function* () {
|
|
71
|
+
const extractor = yield* FileExtractor;
|
|
72
|
+
return KnowledgeExtractor.of(makeFileKnowledgeExtractor(extractor));
|
|
73
|
+
}));
|
|
74
|
+
//#endregion
|
|
75
|
+
export { FileKnowledgeExtractorLayer, makeFileKnowledgeExtractor };
|
|
76
|
+
|
|
77
|
+
//# sourceMappingURL=knowledge.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"knowledge.mjs","names":[],"sources":["../src/knowledge.ts"],"sourcesContent":["import { Effect, Layer, Match, Predicate } from 'effect'\nimport type { ExtractedKnowledgeDocument, KnowledgeSource } from '@yolk-sdk/knowledge/documents'\nimport { KnowledgeExtractionError } from '@yolk-sdk/knowledge/errors'\nimport { KnowledgeExtractor } from '@yolk-sdk/knowledge/extraction'\nimport type { KnowledgeExtractorApi, LoadedKnowledgeSource } from '@yolk-sdk/knowledge/extraction'\nimport type { ExtractedFile } from './format.ts'\nimport { FileExtractor } from './service.ts'\nimport type { FileExtractorApi } from './service.ts'\n\n/** Percent-decode a path segment; malformed escapes (`%zz`, invalid UTF-8) stay encoded. */\nconst decodeSegment = (segment: string) => {\n try {\n return decodeURIComponent(segment)\n } catch {\n return segment\n }\n}\n\nconst urlFilename = (url: string) => {\n if (!URL.canParse(url)) return url\n\n const segments = new URL(url).pathname.split('/').filter(segment => segment.length > 0)\n const last = segments.at(-1)\n\n return last === undefined ? url : decodeSegment(last)\n}\n\n/** The filename used for format detection: file name or ref, URL path, or text label. */\nconst filenameFor = (source: KnowledgeSource) =>\n Match.value(source).pipe(\n Match.tagsExhaustive({\n File: file => file.name ?? file.ref,\n Url: url => urlFilename(url.url),\n Text: text => text.label ?? 'text'\n })\n )\n\n/** Loaded media type first, then the file source's, then `text/plain` for text sources. */\nconst mediaTypeFor = (loaded: LoadedKnowledgeSource) => {\n if (loaded.mediaType !== undefined) return loaded.mediaType\n\n if (Predicate.isTagged(loaded.source, 'File') && loaded.source.mediaType !== undefined)\n return loaded.source.mediaType\n\n return Predicate.isTagged(loaded.source, 'Text') ? 'text/plain' : ''\n}\n\nconst documentFrom = (\n loaded: LoadedKnowledgeSource,\n extracted: ExtractedFile\n): ExtractedKnowledgeDocument => {\n const { title, ...fileMetadata } = extracted.metadata\n const metadata = { ...fileMetadata, ...loaded.metadata }\n\n return title === undefined || title.trim().length === 0\n ? { content: extracted.content, metadata }\n : { content: extracted.content, title: title.trim(), metadata }\n}\n\n/**\n * A `KnowledgeExtractor` backed by a `FileExtractor`. String content is already text and passes\n * through unchanged (it must not be blank); bytes are extracted with the format chosen from the\n * source name and media type. File metadata (`format`, `pageCount`, `sheetNames`) is merged\n * under the loaded source's own metadata, and the extracted title becomes the document title.\n */\nexport const makeFileKnowledgeExtractor = (extractor: FileExtractorApi): KnowledgeExtractorApi => ({\n extract: loaded => {\n if (Predicate.isString(loaded.content)) {\n const content = loaded.content\n\n return content.trim().length === 0\n ? Effect.fail(new KnowledgeExtractionError({ message: 'Knowledge source text is empty' }))\n : Effect.succeed(\n loaded.metadata === undefined ? { content } : { content, metadata: loaded.metadata }\n )\n }\n\n return extractor\n .extract({\n filename: filenameFor(loaded.source),\n mediaType: mediaTypeFor(loaded),\n bytes: loaded.content\n })\n .pipe(\n Effect.map(extracted => documentFrom(loaded, extracted)),\n Effect.mapError(\n error => new KnowledgeExtractionError({ message: error.message, cause: error })\n )\n )\n }\n})\n\n/** Provide `KnowledgeExtractor` from the `FileExtractor` in context (for example the Node layer). */\nexport const FileKnowledgeExtractorLayer = Layer.effect(\n KnowledgeExtractor,\n Effect.gen(function* () {\n const extractor = yield* FileExtractor\n\n return KnowledgeExtractor.of(makeFileKnowledgeExtractor(extractor))\n })\n)\n"],"mappings":";;;;;;AAUA,MAAM,iBAAiB,YAAoB;CACzC,IAAI;EACF,OAAO,mBAAmB,OAAO;CACnC,QAAQ;EACN,OAAO;CACT;AACF;AAEA,MAAM,eAAe,QAAgB;CACnC,IAAI,CAAC,IAAI,SAAS,GAAG,GAAG,OAAO;CAG/B,MAAM,OADW,IAAI,IAAI,GAAG,EAAE,SAAS,MAAM,GAAG,EAAE,QAAO,YAAW,QAAQ,SAAS,CACjE,EAAE,GAAG,EAAE;CAE3B,OAAO,SAAS,KAAA,IAAY,MAAM,cAAc,IAAI;AACtD;;AAGA,MAAM,eAAe,WACnB,MAAM,MAAM,MAAM,EAAE,KAClB,MAAM,eAAe;CACnB,OAAM,SAAQ,KAAK,QAAQ,KAAK;CAChC,MAAK,QAAO,YAAY,IAAI,GAAG;CAC/B,OAAM,SAAQ,KAAK,SAAS;AAC9B,CAAC,CACH;;AAGF,MAAM,gBAAgB,WAAkC;CACtD,IAAI,OAAO,cAAc,KAAA,GAAW,OAAO,OAAO;CAElD,IAAI,UAAU,SAAS,OAAO,QAAQ,MAAM,KAAK,OAAO,OAAO,cAAc,KAAA,GAC3E,OAAO,OAAO,OAAO;CAEvB,OAAO,UAAU,SAAS,OAAO,QAAQ,MAAM,IAAI,eAAe;AACpE;AAEA,MAAM,gBACJ,QACA,cAC+B;CAC/B,MAAM,EAAE,OAAO,GAAG,iBAAiB,UAAU;CAC7C,MAAM,WAAW;EAAE,GAAG;EAAc,GAAG,OAAO;CAAS;CAEvD,OAAO,UAAU,KAAA,KAAa,MAAM,KAAK,EAAE,WAAW,IAClD;EAAE,SAAS,UAAU;EAAS;CAAS,IACvC;EAAE,SAAS,UAAU;EAAS,OAAO,MAAM,KAAK;EAAG;CAAS;AAClE;;;;;;;AAQA,MAAa,8BAA8B,eAAwD,EACjG,UAAS,WAAU;CACjB,IAAI,UAAU,SAAS,OAAO,OAAO,GAAG;EACtC,MAAM,UAAU,OAAO;EAEvB,OAAO,QAAQ,KAAK,EAAE,WAAW,IAC7B,OAAO,KAAK,IAAI,yBAAyB,EAAE,SAAS,iCAAiC,CAAC,CAAC,IACvF,OAAO,QACL,OAAO,aAAa,KAAA,IAAY,EAAE,QAAQ,IAAI;GAAE;GAAS,UAAU,OAAO;EAAS,CACrF;CACN;CAEA,OAAO,UACJ,QAAQ;EACP,UAAU,YAAY,OAAO,MAAM;EACnC,WAAW,aAAa,MAAM;EAC9B,OAAO,OAAO;CAChB,CAAC,EACA,KACC,OAAO,KAAI,cAAa,aAAa,QAAQ,SAAS,CAAC,GACvD,OAAO,UACL,UAAS,IAAI,yBAAyB;EAAE,SAAS,MAAM;EAAS,OAAO;CAAM,CAAC,CAChF,CACF;AACJ,EACF;;AAGA,MAAa,8BAA8B,MAAM,OAC/C,oBACA,OAAO,IAAI,aAAa;CACtB,MAAM,YAAY,OAAO;CAEzB,OAAO,mBAAmB,GAAG,2BAA2B,SAAS,CAAC;AACpE,CAAC,CACH"}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import * as Schema from "effect/Schema";
|
|
2
|
+
|
|
3
|
+
//#region src/limits.d.ts
|
|
4
|
+
/**
|
|
5
|
+
* Smallest `maxXlsxTextCharacters`: one more than the space always reserved for the
|
|
6
|
+
* `[Some hyperlinks omitted: output limit and hyperlink limit]` marker (61 characters with its
|
|
7
|
+
* leading blank line), so a workbook with links can still produce text.
|
|
8
|
+
*/
|
|
9
|
+
declare const minimumXlsxTextCharacters = 62;
|
|
10
|
+
/**
|
|
11
|
+
* Work and output bounds for one `extract` call. Every value is a positive integer, and
|
|
12
|
+
* `maxXlsxTextCharacters` is at least `minimumXlsxTextCharacters`.
|
|
13
|
+
*/
|
|
14
|
+
declare const FileExtractorLimits: Schema.Struct<{
|
|
15
|
+
/** Input bytes accepted for any format. */readonly maxInputBytes: Schema.Int; /** Entries in a DOCX, XLSX, or PPTX ZIP archive. */
|
|
16
|
+
readonly maxArchiveEntries: Schema.Int; /** Total inflated bytes of a DOCX, XLSX, or PPTX archive, counted while inflating. */
|
|
17
|
+
readonly maxExpandedBytes: Schema.Int; /** Worksheets in a workbook. */
|
|
18
|
+
readonly maxXlsxSheets: Schema.Int; /** Cells visited across all worksheet ranges (absent cells count too). */
|
|
19
|
+
readonly maxXlsxCellVisits: Schema.Int; /** Characters of XLSX text, including hyperlink annotations and the omitted-links marker. */
|
|
20
|
+
readonly maxXlsxTextCharacters: Schema.Int; /** Hyperlinks read per workbook; later ones are ignored (still removed before SheetJS). */
|
|
21
|
+
readonly maxXlsxHyperlinks: Schema.Int;
|
|
22
|
+
}>;
|
|
23
|
+
type FileExtractorLimits = typeof FileExtractorLimits.Type;
|
|
24
|
+
/** Default limits; override any of them through `makeFileExtractorLayer({ limits })`. */
|
|
25
|
+
declare const defaultFileExtractorLimits: FileExtractorLimits;
|
|
26
|
+
//#endregion
|
|
27
|
+
export { FileExtractorLimits, defaultFileExtractorLimits, minimumXlsxTextCharacters };
|
|
28
|
+
//# sourceMappingURL=limits.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"limits.d.mts","names":[],"sources":["../src/limits.ts"],"mappings":";;;;;AAYA;;;cAAa,yBAAA;AAAyB;AAMtC;;;AANsC,cAMzB,mBAAA,EAAmB,MAAA,CAAA,MAAA;;;yCAAA;EAAA,oCAAA;EAAA,wCAAA;EAAA;;;KAmBpB,mBAAA,UAA6B,mBAAA,CAAoB,IAAI;;cAGpD,0BAAA,EAA4B,mBAQxC"}
|
package/dist/limits.mjs
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
import * as Schema from "effect/Schema";
|
|
2
|
+
//#region src/limits.ts
|
|
3
|
+
const PositiveSafeInteger = Schema.Int.pipe(Schema.check(Schema.isGreaterThan(0)), Schema.check(Schema.isLessThanOrEqualTo(Number.MAX_SAFE_INTEGER)));
|
|
4
|
+
/**
|
|
5
|
+
* Smallest `maxXlsxTextCharacters`: one more than the space always reserved for the
|
|
6
|
+
* `[Some hyperlinks omitted: output limit and hyperlink limit]` marker (61 characters with its
|
|
7
|
+
* leading blank line), so a workbook with links can still produce text.
|
|
8
|
+
*/
|
|
9
|
+
const minimumXlsxTextCharacters = 62;
|
|
10
|
+
/**
|
|
11
|
+
* Work and output bounds for one `extract` call. Every value is a positive integer, and
|
|
12
|
+
* `maxXlsxTextCharacters` is at least `minimumXlsxTextCharacters`.
|
|
13
|
+
*/
|
|
14
|
+
const FileExtractorLimits = Schema.Struct({
|
|
15
|
+
/** Input bytes accepted for any format. */
|
|
16
|
+
maxInputBytes: PositiveSafeInteger,
|
|
17
|
+
/** Entries in a DOCX, XLSX, or PPTX ZIP archive. */
|
|
18
|
+
maxArchiveEntries: PositiveSafeInteger,
|
|
19
|
+
/** Total inflated bytes of a DOCX, XLSX, or PPTX archive, counted while inflating. */
|
|
20
|
+
maxExpandedBytes: PositiveSafeInteger,
|
|
21
|
+
/** Worksheets in a workbook. */
|
|
22
|
+
maxXlsxSheets: PositiveSafeInteger,
|
|
23
|
+
/** Cells visited across all worksheet ranges (absent cells count too). */
|
|
24
|
+
maxXlsxCellVisits: PositiveSafeInteger,
|
|
25
|
+
/** Characters of XLSX text, including hyperlink annotations and the omitted-links marker. */
|
|
26
|
+
maxXlsxTextCharacters: PositiveSafeInteger.pipe(Schema.check(Schema.isGreaterThanOrEqualTo(62))),
|
|
27
|
+
/** Hyperlinks read per workbook; later ones are ignored (still removed before SheetJS). */
|
|
28
|
+
maxXlsxHyperlinks: PositiveSafeInteger
|
|
29
|
+
});
|
|
30
|
+
/** Default limits; override any of them through `makeFileExtractorLayer({ limits })`. */
|
|
31
|
+
const defaultFileExtractorLimits = {
|
|
32
|
+
maxInputBytes: 50 * 1024 * 1024,
|
|
33
|
+
maxArchiveEntries: 1e4,
|
|
34
|
+
maxExpandedBytes: 50 * 1024 * 1024,
|
|
35
|
+
maxXlsxSheets: 100,
|
|
36
|
+
maxXlsxCellVisits: 1e5,
|
|
37
|
+
maxXlsxTextCharacters: 512 * 1024,
|
|
38
|
+
maxXlsxHyperlinks: 1e4
|
|
39
|
+
};
|
|
40
|
+
//#endregion
|
|
41
|
+
export { FileExtractorLimits, defaultFileExtractorLimits, minimumXlsxTextCharacters };
|
|
42
|
+
|
|
43
|
+
//# sourceMappingURL=limits.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"limits.mjs","names":[],"sources":["../src/limits.ts"],"sourcesContent":["import * as Schema from 'effect/Schema'\n\nconst PositiveSafeInteger = Schema.Int.pipe(\n Schema.check(Schema.isGreaterThan(0)),\n Schema.check(Schema.isLessThanOrEqualTo(Number.MAX_SAFE_INTEGER))\n)\n\n/**\n * Smallest `maxXlsxTextCharacters`: one more than the space always reserved for the\n * `[Some hyperlinks omitted: output limit and hyperlink limit]` marker (61 characters with its\n * leading blank line), so a workbook with links can still produce text.\n */\nexport const minimumXlsxTextCharacters = 62\n\n/**\n * Work and output bounds for one `extract` call. Every value is a positive integer, and\n * `maxXlsxTextCharacters` is at least `minimumXlsxTextCharacters`.\n */\nexport const FileExtractorLimits = Schema.Struct({\n /** Input bytes accepted for any format. */\n maxInputBytes: PositiveSafeInteger,\n /** Entries in a DOCX, XLSX, or PPTX ZIP archive. */\n maxArchiveEntries: PositiveSafeInteger,\n /** Total inflated bytes of a DOCX, XLSX, or PPTX archive, counted while inflating. */\n maxExpandedBytes: PositiveSafeInteger,\n /** Worksheets in a workbook. */\n maxXlsxSheets: PositiveSafeInteger,\n /** Cells visited across all worksheet ranges (absent cells count too). */\n maxXlsxCellVisits: PositiveSafeInteger,\n /** Characters of XLSX text, including hyperlink annotations and the omitted-links marker. */\n maxXlsxTextCharacters: PositiveSafeInteger.pipe(\n Schema.check(Schema.isGreaterThanOrEqualTo(minimumXlsxTextCharacters))\n ),\n /** Hyperlinks read per workbook; later ones are ignored (still removed before SheetJS). */\n maxXlsxHyperlinks: PositiveSafeInteger\n})\n\nexport type FileExtractorLimits = typeof FileExtractorLimits.Type\n\n/** Default limits; override any of them through `makeFileExtractorLayer({ limits })`. */\nexport const defaultFileExtractorLimits: FileExtractorLimits = {\n maxInputBytes: 50 * 1024 * 1024,\n maxArchiveEntries: 10_000,\n maxExpandedBytes: 50 * 1024 * 1024,\n maxXlsxSheets: 100,\n maxXlsxCellVisits: 100_000,\n maxXlsxTextCharacters: 512 * 1024,\n maxXlsxHyperlinks: 10_000\n}\n"],"mappings":";;AAEA,MAAM,sBAAsB,OAAO,IAAI,KACrC,OAAO,MAAM,OAAO,cAAc,CAAC,CAAC,GACpC,OAAO,MAAM,OAAO,oBAAoB,OAAO,gBAAgB,CAAC,CAClE;;;;;;AAOA,MAAa,4BAA4B;;;;;AAMzC,MAAa,sBAAsB,OAAO,OAAO;;CAE/C,eAAe;;CAEf,mBAAmB;;CAEnB,kBAAkB;;CAElB,eAAe;;CAEf,mBAAmB;;CAEnB,uBAAuB,oBAAoB,KACzC,OAAO,MAAM,OAAO,uBAAA,EAAgD,CAAC,CACvE;;CAEA,mBAAmB;AACrB,CAAC;;AAKD,MAAa,6BAAkD;CAC7D,eAAe,KAAK,OAAO;CAC3B,mBAAmB;CACnB,kBAAkB,KAAK,OAAO;CAC9B,eAAe;CACf,mBAAmB;CACnB,uBAAuB,MAAM;CAC7B,mBAAmB;AACrB"}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import { FileExtractionError, FileExtractorError } from "../errors.mjs";
|
|
2
|
+
import { ExtractedFile, ExtractedFileFormat, ExtractedFileMetadata, FileInput, OfficeFileFormat } from "../format.mjs";
|
|
3
|
+
import { FileExtractorLimits } from "../limits.mjs";
|
|
4
|
+
import { SheetJsLoader } from "./sheetjs.mjs";
|
|
5
|
+
import { Effect } from "effect";
|
|
6
|
+
|
|
7
|
+
//#region src/node/extract-file.d.ts
|
|
8
|
+
/** Formats whose bytes go through a parser (and, by default, an isolated worker). */
|
|
9
|
+
type ParsedFileFormat = 'pdf' | OfficeFileFormat;
|
|
10
|
+
declare const isParsedFileFormat: (format: ExtractedFileFormat) => format is ParsedFileFormat;
|
|
11
|
+
/** Sanitize extracted text; empty results fail. */
|
|
12
|
+
declare const makeExtractedFile: (content: string, metadata: ExtractedFileMetadata) => Effect.Effect<never, FileExtractionError, never> | Effect.Effect<ExtractedFile, never, never>;
|
|
13
|
+
/**
|
|
14
|
+
* PDF.js 6 releases a document through its loading task (`PDFDocumentProxy` no longer has
|
|
15
|
+
* `destroy()`); unpdf releases the documents it opens the same way.
|
|
16
|
+
*/
|
|
17
|
+
type PdfDocumentResource = {
|
|
18
|
+
readonly loadingTask: {
|
|
19
|
+
readonly destroy: () => Promise<void>;
|
|
20
|
+
};
|
|
21
|
+
};
|
|
22
|
+
/** Run `use` with an opened PDF document and always release the parser afterwards. */
|
|
23
|
+
declare const withAcquiredPdfDocument: <D extends PdfDocumentResource, A, E, R>(open: Effect.Effect<D, E, R>, use: (document: D) => Effect.Effect<A, E, R>, format: ExtractedFileFormat) => Effect.Effect<A, E, Exclude<R, import("effect/Scope").Scope>>;
|
|
24
|
+
/**
|
|
25
|
+
* Extract a PDF, DOCX, XLSX, or PPTX file in the current thread. The Node layer runs this inside
|
|
26
|
+
* an isolated worker unless `isolation: 'none'` is configured.
|
|
27
|
+
*/
|
|
28
|
+
declare const extractParsedFile: (input: FileInput, format: ParsedFileFormat, limits: FileExtractorLimits, loader: SheetJsLoader) => Effect.Effect<ExtractedFile, FileExtractorError>;
|
|
29
|
+
//#endregion
|
|
30
|
+
export { ParsedFileFormat, extractParsedFile, isParsedFileFormat, makeExtractedFile, withAcquiredPdfDocument };
|
|
31
|
+
//# sourceMappingURL=extract-file.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-file.d.mts","names":[],"sources":["../../src/node/extract-file.ts"],"mappings":";;;;;;;;KAuBY,gBAAA,WAA2B,gBAAgB;AAAA,cAE1C,kBAAA,GAAsB,MAAA,EAAQ,mBAAA,KAAsB,MAAA,IAAU,gBACM;;cAGpE,iBAAA,GAAqB,OAAA,UAAiB,QAAA,EAAU,qBAAA,KAAqB,MAAA,CAAA,MAAA,QAAA,mBAAA,WAAA,MAAA,CAAA,MAAA,CAAA,aAAA;AAN3B;AAEvD;;;AAFuD,KAwBlD,mBAAA;EAAA,SACM,WAAA;IAAA,SAAwB,OAAA,QAAe,OAAO;EAAA;AAAA;;cAI5C,uBAAA,aAAqC,mBAAA,WAChD,IAAA,EAAM,MAAA,CAAO,MAAA,CAAO,CAAA,EAAG,CAAA,EAAG,CAAA,GAC1B,GAAA,GAAM,QAAA,EAAU,CAAA,KAAM,MAAA,CAAO,MAAA,CAAO,CAAA,EAAG,CAAA,EAAG,CAAA,GAC1C,MAAA,EAAQ,mBAAA,KAAmB,MAAA,CAAA,MAAA,CAAA,CAAA,EAAA,CAAA,EAAA,OAAA,CAAA,CAAA,yBAAA,KAAA;AA1B7B;;;;AAAA,cAkOa,iBAAA,GACX,KAAA,EAAO,SAAA,EACP,MAAA,EAAQ,gBAAA,EACR,MAAA,EAAQ,mBAAA,EACR,MAAA,EAAQ,aAAA,KACP,MAAA,CAAO,MAAA,CAAO,aAAA,EAAe,kBAAA"}
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
import { FileExtractionError } from "../errors.mjs";
|
|
2
|
+
import { sanitizeExtractedText } from "../sanitize.mjs";
|
|
3
|
+
import { readOfficeArchive, storedArchive } from "./office-archive.mjs";
|
|
4
|
+
import { extractPptxText } from "./pptx-text.mjs";
|
|
5
|
+
import { asXlsxWorkbook, loadSheetJs } from "./sheetjs.mjs";
|
|
6
|
+
import { resolveXlsxHyperlinks } from "./xlsx-hyperlinks.mjs";
|
|
7
|
+
import { buildSheetJsInput } from "./xlsx-sheetjs-input.mjs";
|
|
8
|
+
import { extractBoundedXlsxText } from "./xlsx-text.mjs";
|
|
9
|
+
import { Effect, Option, Predicate } from "effect";
|
|
10
|
+
import { Buffer } from "node:buffer";
|
|
11
|
+
//#region src/node/extract-file.ts
|
|
12
|
+
const isParsedFileFormat = (format) => format === "pdf" || format === "docx" || format === "pptx" || format === "xlsx";
|
|
13
|
+
/** Sanitize extracted text; empty results fail. */
|
|
14
|
+
const makeExtractedFile = (content, metadata) => {
|
|
15
|
+
const sanitized = sanitizeExtractedText(content);
|
|
16
|
+
if (sanitized.length === 0) return Effect.fail(new FileExtractionError({
|
|
17
|
+
message: "Extracted file content is empty",
|
|
18
|
+
format: metadata.format
|
|
19
|
+
}));
|
|
20
|
+
return Effect.succeed({
|
|
21
|
+
content: sanitized,
|
|
22
|
+
metadata
|
|
23
|
+
});
|
|
24
|
+
};
|
|
25
|
+
/** Run `use` with an opened PDF document and always release the parser afterwards. */
|
|
26
|
+
const withAcquiredPdfDocument = (open, use, format) => Effect.scoped(Effect.gen(function* () {
|
|
27
|
+
return yield* use(yield* Effect.acquireRelease(open, (document) => Effect.tryPromise({
|
|
28
|
+
try: () => document.loadingTask.destroy(),
|
|
29
|
+
catch: () => new FileExtractionError({
|
|
30
|
+
message: "Could not release PDF parser",
|
|
31
|
+
format
|
|
32
|
+
})
|
|
33
|
+
}).pipe(Effect.ignore)));
|
|
34
|
+
}));
|
|
35
|
+
const extractPdf = (input) => Effect.gen(function* () {
|
|
36
|
+
const { extractText, getDocumentProxy, getMeta } = yield* Effect.tryPromise({
|
|
37
|
+
try: () => import("unpdf"),
|
|
38
|
+
catch: (cause) => new FileExtractionError({
|
|
39
|
+
message: "Could not read PDF",
|
|
40
|
+
format: "pdf",
|
|
41
|
+
cause
|
|
42
|
+
})
|
|
43
|
+
});
|
|
44
|
+
return yield* withAcquiredPdfDocument(Effect.tryPromise({
|
|
45
|
+
try: () => getDocumentProxy(new Uint8Array(input.bytes)),
|
|
46
|
+
catch: (cause) => new FileExtractionError({
|
|
47
|
+
message: "Could not read PDF",
|
|
48
|
+
format: "pdf",
|
|
49
|
+
cause
|
|
50
|
+
})
|
|
51
|
+
}), (document) => Effect.gen(function* () {
|
|
52
|
+
const extracted = yield* Effect.tryPromise({
|
|
53
|
+
try: () => extractText(document, { mergePages: true }),
|
|
54
|
+
catch: (cause) => new FileExtractionError({
|
|
55
|
+
message: "Could not extract PDF text",
|
|
56
|
+
format: "pdf",
|
|
57
|
+
cause
|
|
58
|
+
})
|
|
59
|
+
});
|
|
60
|
+
const meta = yield* Effect.tryPromise({
|
|
61
|
+
try: () => getMeta(document),
|
|
62
|
+
catch: (cause) => new FileExtractionError({
|
|
63
|
+
message: "Could not read PDF metadata",
|
|
64
|
+
format: "pdf",
|
|
65
|
+
cause
|
|
66
|
+
})
|
|
67
|
+
}).pipe(Effect.option);
|
|
68
|
+
const rawTitle = Option.isSome(meta) ? meta.value.info.Title : void 0;
|
|
69
|
+
const title = Predicate.isString(rawTitle) && rawTitle.length > 0 ? rawTitle : void 0;
|
|
70
|
+
const metadata = title === void 0 ? {
|
|
71
|
+
format: "pdf",
|
|
72
|
+
pageCount: extracted.totalPages
|
|
73
|
+
} : {
|
|
74
|
+
format: "pdf",
|
|
75
|
+
title,
|
|
76
|
+
pageCount: extracted.totalPages
|
|
77
|
+
};
|
|
78
|
+
return yield* makeExtractedFile(extracted.text, metadata);
|
|
79
|
+
}), "pdf");
|
|
80
|
+
});
|
|
81
|
+
const validatedArchive = (input, format, limits) => readOfficeArchive(input.bytes, format, limits, format === "xlsx" ? { maxHyperlinkTags: limits.maxXlsxHyperlinks + 1 } : {}).pipe(Effect.mapError((error) => new FileExtractionError({
|
|
82
|
+
message: error.message,
|
|
83
|
+
format,
|
|
84
|
+
cause: error
|
|
85
|
+
})));
|
|
86
|
+
const extractDocx = (input, limits) => Effect.gen(function* () {
|
|
87
|
+
const { parts } = yield* validatedArchive(input, "docx", limits);
|
|
88
|
+
return yield* makeExtractedFile((yield* Effect.tryPromise({
|
|
89
|
+
try: async () => {
|
|
90
|
+
const { default: mammoth } = await import("mammoth");
|
|
91
|
+
return await mammoth.extractRawText({ buffer: Buffer.from(storedArchive(parts)) });
|
|
92
|
+
},
|
|
93
|
+
catch: (cause) => new FileExtractionError({
|
|
94
|
+
message: "Could not extract DOCX text",
|
|
95
|
+
format: "docx",
|
|
96
|
+
cause
|
|
97
|
+
})
|
|
98
|
+
})).value, {
|
|
99
|
+
format: "docx",
|
|
100
|
+
title: input.filename
|
|
101
|
+
});
|
|
102
|
+
});
|
|
103
|
+
/** Keep the first `max` tags in archive order. */
|
|
104
|
+
const capHyperlinkTags = (tagsByPart, max) => {
|
|
105
|
+
const capped = /* @__PURE__ */ new Map();
|
|
106
|
+
let remaining = max;
|
|
107
|
+
for (const [part, tags] of tagsByPart) {
|
|
108
|
+
if (remaining <= 0) break;
|
|
109
|
+
const kept = tags.slice(0, remaining);
|
|
110
|
+
remaining -= kept.length;
|
|
111
|
+
capped.set(part, kept);
|
|
112
|
+
}
|
|
113
|
+
return capped;
|
|
114
|
+
};
|
|
115
|
+
const extractXlsx = (input, limits, loader) => Effect.gen(function* () {
|
|
116
|
+
const normalized = yield* validatedArchive(input, "xlsx", limits);
|
|
117
|
+
const sheetJsInput = yield* Effect.try({
|
|
118
|
+
try: () => buildSheetJsInput(normalized.parts, limits.maxXlsxSheets),
|
|
119
|
+
catch: (cause) => cause instanceof FileExtractionError ? cause : new FileExtractionError({
|
|
120
|
+
message: "Could not read XLSX",
|
|
121
|
+
format: "xlsx",
|
|
122
|
+
cause
|
|
123
|
+
})
|
|
124
|
+
});
|
|
125
|
+
const sheetJs = yield* loadSheetJs(loader);
|
|
126
|
+
const workbook = asXlsxWorkbook(yield* Effect.try({
|
|
127
|
+
try: () => sheetJs.read(sheetJsInput.archive),
|
|
128
|
+
catch: (cause) => new FileExtractionError({
|
|
129
|
+
message: "Could not read XLSX",
|
|
130
|
+
format: "xlsx",
|
|
131
|
+
cause
|
|
132
|
+
})
|
|
133
|
+
}));
|
|
134
|
+
if (workbook === void 0) return yield* Effect.fail(new FileExtractionError({
|
|
135
|
+
message: "Could not read XLSX",
|
|
136
|
+
format: "xlsx"
|
|
137
|
+
}));
|
|
138
|
+
const hyperlinksTruncated = [...normalized.hyperlinkTags.values()].reduce((total, tags) => total + tags.length, 0) > limits.maxXlsxHyperlinks;
|
|
139
|
+
const hyperlinks = resolveXlsxHyperlinks(normalized.parts, hyperlinksTruncated ? capHyperlinkTags(normalized.hyperlinkTags, limits.maxXlsxHyperlinks) : normalized.hyperlinkTags, sheetJsInput.sheets);
|
|
140
|
+
return yield* makeExtractedFile(yield* Effect.try({
|
|
141
|
+
try: () => extractBoundedXlsxText(workbook, limits, {
|
|
142
|
+
hyperlinks,
|
|
143
|
+
hyperlinksTruncated
|
|
144
|
+
}),
|
|
145
|
+
catch: (cause) => cause instanceof FileExtractionError ? cause : new FileExtractionError({
|
|
146
|
+
message: "Could not extract XLSX text",
|
|
147
|
+
format: "xlsx",
|
|
148
|
+
cause
|
|
149
|
+
})
|
|
150
|
+
}), {
|
|
151
|
+
format: "xlsx",
|
|
152
|
+
title: sheetJsInput.title ?? input.filename,
|
|
153
|
+
sheetNames: workbook.SheetNames
|
|
154
|
+
});
|
|
155
|
+
});
|
|
156
|
+
const extractPptx = (input, limits) => Effect.gen(function* () {
|
|
157
|
+
const { parts } = yield* validatedArchive(input, "pptx", limits);
|
|
158
|
+
return yield* makeExtractedFile(yield* Effect.try({
|
|
159
|
+
try: () => extractPptxText(parts),
|
|
160
|
+
catch: (cause) => new FileExtractionError({
|
|
161
|
+
message: "Could not extract PPTX text",
|
|
162
|
+
format: "pptx",
|
|
163
|
+
cause
|
|
164
|
+
})
|
|
165
|
+
}), {
|
|
166
|
+
format: "pptx",
|
|
167
|
+
title: input.filename
|
|
168
|
+
});
|
|
169
|
+
});
|
|
170
|
+
/**
|
|
171
|
+
* Extract a PDF, DOCX, XLSX, or PPTX file in the current thread. The Node layer runs this inside
|
|
172
|
+
* an isolated worker unless `isolation: 'none'` is configured.
|
|
173
|
+
*/
|
|
174
|
+
const extractParsedFile = (input, format, limits, loader) => {
|
|
175
|
+
if (format === "pdf") return extractPdf(input);
|
|
176
|
+
if (format === "docx") return extractDocx(input, limits);
|
|
177
|
+
if (format === "xlsx") return extractXlsx(input, limits, loader);
|
|
178
|
+
return extractPptx(input, limits);
|
|
179
|
+
};
|
|
180
|
+
//#endregion
|
|
181
|
+
export { extractParsedFile, isParsedFileFormat, makeExtractedFile, withAcquiredPdfDocument };
|
|
182
|
+
|
|
183
|
+
//# sourceMappingURL=extract-file.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-file.mjs","names":[],"sources":["../../src/node/extract-file.ts"],"sourcesContent":["import { Buffer } from 'node:buffer'\nimport { Effect, Option, Predicate } from 'effect'\nimport { FileExtractionError } from '../errors.ts'\nimport type { FileExtractorError } from '../errors.ts'\nimport type {\n ExtractedFile,\n ExtractedFileFormat,\n ExtractedFileMetadata,\n FileInput,\n OfficeFileFormat\n} from '../format.ts'\nimport type { FileExtractorLimits } from '../limits.ts'\nimport { sanitizeExtractedText } from '../sanitize.ts'\nimport { readOfficeArchive, storedArchive } from './office-archive.ts'\nimport type { NormalizedOfficeArchive } from './office-archive.ts'\nimport { extractPptxText } from './pptx-text.ts'\nimport { asXlsxWorkbook, loadSheetJs } from './sheetjs.ts'\nimport type { SheetJsLoader } from './sheetjs.ts'\nimport { resolveXlsxHyperlinks } from './xlsx-hyperlinks.ts'\nimport { buildSheetJsInput } from './xlsx-sheetjs-input.ts'\nimport { extractBoundedXlsxText } from './xlsx-text.ts'\n\n/** Formats whose bytes go through a parser (and, by default, an isolated worker). */\nexport type ParsedFileFormat = 'pdf' | OfficeFileFormat\n\nexport const isParsedFileFormat = (format: ExtractedFileFormat): format is ParsedFileFormat =>\n format === 'pdf' || format === 'docx' || format === 'pptx' || format === 'xlsx'\n\n/** Sanitize extracted text; empty results fail. */\nexport const makeExtractedFile = (content: string, metadata: ExtractedFileMetadata) => {\n const sanitized = sanitizeExtractedText(content)\n\n if (sanitized.length === 0)\n return Effect.fail(\n new FileExtractionError({\n message: 'Extracted file content is empty',\n format: metadata.format\n })\n )\n\n return Effect.succeed<ExtractedFile>({ content: sanitized, metadata })\n}\n\n/**\n * PDF.js 6 releases a document through its loading task (`PDFDocumentProxy` no longer has\n * `destroy()`); unpdf releases the documents it opens the same way.\n */\ntype PdfDocumentResource = {\n readonly loadingTask: { readonly destroy: () => Promise<void> }\n}\n\n/** Run `use` with an opened PDF document and always release the parser afterwards. */\nexport const withAcquiredPdfDocument = <D extends PdfDocumentResource, A, E, R>(\n open: Effect.Effect<D, E, R>,\n use: (document: D) => Effect.Effect<A, E, R>,\n format: ExtractedFileFormat\n) =>\n Effect.scoped(\n Effect.gen(function* () {\n const document = yield* Effect.acquireRelease(open, document =>\n Effect.tryPromise({\n try: () => document.loadingTask.destroy(),\n catch: () => new FileExtractionError({ message: 'Could not release PDF parser', format })\n }).pipe(Effect.ignore)\n )\n\n return yield* use(document)\n })\n )\n\nconst extractPdf = (input: FileInput) =>\n Effect.gen(function* () {\n // Loaded on first use, so a worker extracting another format never loads PDF.js.\n const { extractText, getDocumentProxy, getMeta } = yield* Effect.tryPromise({\n try: () => import('unpdf'),\n catch: cause =>\n new FileExtractionError({ message: 'Could not read PDF', format: 'pdf', cause })\n })\n\n return yield* withAcquiredPdfDocument(\n Effect.tryPromise({\n // PDF.js may detach the buffer it is given; never hand it the caller's bytes.\n try: () => getDocumentProxy(new Uint8Array(input.bytes)),\n catch: cause =>\n new FileExtractionError({ message: 'Could not read PDF', format: 'pdf', cause })\n }),\n document =>\n Effect.gen(function* () {\n const extracted = yield* Effect.tryPromise({\n try: () => extractText(document, { mergePages: true }),\n catch: cause =>\n new FileExtractionError({\n message: 'Could not extract PDF text',\n format: 'pdf',\n cause\n })\n })\n\n const meta = yield* Effect.tryPromise({\n try: () => getMeta(document),\n catch: cause =>\n new FileExtractionError({\n message: 'Could not read PDF metadata',\n format: 'pdf',\n cause\n })\n }).pipe(Effect.option)\n\n const rawTitle = Option.isSome(meta) ? meta.value.info.Title : undefined\n const title = Predicate.isString(rawTitle) && rawTitle.length > 0 ? rawTitle : undefined\n\n const metadata: ExtractedFileMetadata =\n title === undefined\n ? { format: 'pdf', pageCount: extracted.totalPages }\n : { format: 'pdf', title, pageCount: extracted.totalPages }\n\n return yield* makeExtractedFile(extracted.text, metadata)\n }),\n 'pdf'\n )\n })\n\nconst validatedArchive = (\n input: FileInput,\n format: OfficeFileFormat,\n limits: FileExtractorLimits\n): Effect.Effect<NormalizedOfficeArchive, FileExtractionError> =>\n readOfficeArchive(\n input.bytes,\n format,\n limits,\n // One extra tag detects a workbook over the cap.\n format === 'xlsx' ? { maxHyperlinkTags: limits.maxXlsxHyperlinks + 1 } : {}\n ).pipe(\n Effect.mapError(\n error => new FileExtractionError({ message: error.message, format, cause: error })\n )\n )\n\nconst extractDocx = (input: FileInput, limits: FileExtractorLimits) =>\n Effect.gen(function* () {\n const { parts } = yield* validatedArchive(input, 'docx', limits)\n\n const result = yield* Effect.tryPromise({\n try: async () => {\n const { default: mammoth } = await import('mammoth')\n\n return await mammoth.extractRawText({ buffer: Buffer.from(storedArchive(parts)) })\n },\n catch: cause =>\n new FileExtractionError({ message: 'Could not extract DOCX text', format: 'docx', cause })\n })\n\n return yield* makeExtractedFile(result.value, { format: 'docx', title: input.filename })\n })\n\n/** Keep the first `max` tags in archive order. */\nconst capHyperlinkTags = (\n tagsByPart: ReadonlyMap<string, ReadonlyArray<string>>,\n max: number\n): ReadonlyMap<string, ReadonlyArray<string>> => {\n const capped = new Map<string, ReadonlyArray<string>>()\n let remaining = max\n\n for (const [part, tags] of tagsByPart) {\n if (remaining <= 0) break\n\n const kept = tags.slice(0, remaining)\n remaining -= kept.length\n capped.set(part, kept)\n }\n\n return capped\n}\n\nconst extractXlsx = (input: FileInput, limits: FileExtractorLimits, loader: SheetJsLoader) =>\n Effect.gen(function* () {\n const normalized = yield* validatedArchive(input, 'xlsx', limits)\n\n // SheetJS never sees the uploaded archive: only this allowlisted rebuild of its parts.\n const sheetJsInput = yield* Effect.try({\n try: () => buildSheetJsInput(normalized.parts, limits.maxXlsxSheets),\n catch: cause =>\n cause instanceof FileExtractionError\n ? cause\n : new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx', cause })\n })\n\n const sheetJs = yield* loadSheetJs(loader)\n\n const parsed = yield* Effect.try({\n try: () => sheetJs.read(sheetJsInput.archive),\n catch: cause =>\n new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx', cause })\n })\n\n const workbook = asXlsxWorkbook(parsed)\n\n if (workbook === undefined)\n return yield* Effect.fail(\n new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx' })\n )\n\n // One extra tag was captured to detect that the workbook exceeds the hyperlink cap.\n const capturedTags = [...normalized.hyperlinkTags.values()].reduce(\n (total, tags) => total + tags.length,\n 0\n )\n\n const hyperlinksTruncated = capturedTags > limits.maxXlsxHyperlinks\n\n const hyperlinks = resolveXlsxHyperlinks(\n normalized.parts,\n hyperlinksTruncated\n ? capHyperlinkTags(normalized.hyperlinkTags, limits.maxXlsxHyperlinks)\n : normalized.hyperlinkTags,\n sheetJsInput.sheets\n )\n\n const content = yield* Effect.try({\n try: () => extractBoundedXlsxText(workbook, limits, { hyperlinks, hyperlinksTruncated }),\n catch: cause =>\n cause instanceof FileExtractionError\n ? cause\n : new FileExtractionError({\n message: 'Could not extract XLSX text',\n format: 'xlsx',\n cause\n })\n })\n\n return yield* makeExtractedFile(content, {\n format: 'xlsx',\n title: sheetJsInput.title ?? input.filename,\n sheetNames: workbook.SheetNames\n })\n })\n\nconst extractPptx = (input: FileInput, limits: FileExtractorLimits) =>\n Effect.gen(function* () {\n const { parts } = yield* validatedArchive(input, 'pptx', limits)\n\n const text = yield* Effect.try({\n try: () => extractPptxText(parts),\n catch: cause =>\n new FileExtractionError({ message: 'Could not extract PPTX text', format: 'pptx', cause })\n })\n\n return yield* makeExtractedFile(text, { format: 'pptx', title: input.filename })\n })\n\n/**\n * Extract a PDF, DOCX, XLSX, or PPTX file in the current thread. The Node layer runs this inside\n * an isolated worker unless `isolation: 'none'` is configured.\n */\nexport const extractParsedFile = (\n input: FileInput,\n format: ParsedFileFormat,\n limits: FileExtractorLimits,\n loader: SheetJsLoader\n): Effect.Effect<ExtractedFile, FileExtractorError> => {\n if (format === 'pdf') return extractPdf(input)\n\n if (format === 'docx') return extractDocx(input, limits)\n\n if (format === 'xlsx') return extractXlsx(input, limits, loader)\n\n return extractPptx(input, limits)\n}\n"],"mappings":";;;;;;;;;;;AAyBA,MAAa,sBAAsB,WACjC,WAAW,SAAS,WAAW,UAAU,WAAW,UAAU,WAAW;;AAG3E,MAAa,qBAAqB,SAAiB,aAAoC;CACrF,MAAM,YAAY,sBAAsB,OAAO;CAE/C,IAAI,UAAU,WAAW,GACvB,OAAO,OAAO,KACZ,IAAI,oBAAoB;EACtB,SAAS;EACT,QAAQ,SAAS;CACnB,CAAC,CACH;CAEF,OAAO,OAAO,QAAuB;EAAE,SAAS;EAAW;CAAS,CAAC;AACvE;;AAWA,MAAa,2BACX,MACA,KACA,WAEA,OAAO,OACL,OAAO,IAAI,aAAa;CAQtB,OAAO,OAAO,IAAI,OAPM,OAAO,eAAe,OAAM,aAClD,OAAO,WAAW;EAChB,WAAW,SAAS,YAAY,QAAQ;EACxC,aAAa,IAAI,oBAAoB;GAAE,SAAS;GAAgC;EAAO,CAAC;CAC1F,CAAC,EAAE,KAAK,OAAO,MAAM,CACvB,CAE0B;AAC5B,CAAC,CACH;AAEF,MAAM,cAAc,UAClB,OAAO,IAAI,aAAa;CAEtB,MAAM,EAAE,aAAa,kBAAkB,YAAY,OAAO,OAAO,WAAW;EAC1E,WAAW,OAAO;EAClB,QAAO,UACL,IAAI,oBAAoB;GAAE,SAAS;GAAsB,QAAQ;GAAO;EAAM,CAAC;CACnF,CAAC;CAED,OAAO,OAAO,wBACZ,OAAO,WAAW;EAEhB,WAAW,iBAAiB,IAAI,WAAW,MAAM,KAAK,CAAC;EACvD,QAAO,UACL,IAAI,oBAAoB;GAAE,SAAS;GAAsB,QAAQ;GAAO;EAAM,CAAC;CACnF,CAAC,IACD,aACE,OAAO,IAAI,aAAa;EACtB,MAAM,YAAY,OAAO,OAAO,WAAW;GACzC,WAAW,YAAY,UAAU,EAAE,YAAY,KAAK,CAAC;GACrD,QAAO,UACL,IAAI,oBAAoB;IACtB,SAAS;IACT,QAAQ;IACR;GACF,CAAC;EACL,CAAC;EAED,MAAM,OAAO,OAAO,OAAO,WAAW;GACpC,WAAW,QAAQ,QAAQ;GAC3B,QAAO,UACL,IAAI,oBAAoB;IACtB,SAAS;IACT,QAAQ;IACR;GACF,CAAC;EACL,CAAC,EAAE,KAAK,OAAO,MAAM;EAErB,MAAM,WAAW,OAAO,OAAO,IAAI,IAAI,KAAK,MAAM,KAAK,QAAQ,KAAA;EAC/D,MAAM,QAAQ,UAAU,SAAS,QAAQ,KAAK,SAAS,SAAS,IAAI,WAAW,KAAA;EAE/E,MAAM,WACJ,UAAU,KAAA,IACN;GAAE,QAAQ;GAAO,WAAW,UAAU;EAAW,IACjD;GAAE,QAAQ;GAAO;GAAO,WAAW,UAAU;EAAW;EAE9D,OAAO,OAAO,kBAAkB,UAAU,MAAM,QAAQ;CAC1D,CAAC,GACH,KACF;AACF,CAAC;AAEH,MAAM,oBACJ,OACA,QACA,WAEA,kBACE,MAAM,OACN,QACA,QAEA,WAAW,SAAS,EAAE,kBAAkB,OAAO,oBAAoB,EAAE,IAAI,CAAC,CAC5E,EAAE,KACA,OAAO,UACL,UAAS,IAAI,oBAAoB;CAAE,SAAS,MAAM;CAAS;CAAQ,OAAO;AAAM,CAAC,CACnF,CACF;AAEF,MAAM,eAAe,OAAkB,WACrC,OAAO,IAAI,aAAa;CACtB,MAAM,EAAE,UAAU,OAAO,iBAAiB,OAAO,QAAQ,MAAM;CAY/D,OAAO,OAAO,mBAAkB,OAVV,OAAO,WAAW;EACtC,KAAK,YAAY;GACf,MAAM,EAAE,SAAS,YAAY,MAAM,OAAO;GAE1C,OAAO,MAAM,QAAQ,eAAe,EAAE,QAAQ,OAAO,KAAK,cAAc,KAAK,CAAC,EAAE,CAAC;EACnF;EACA,QAAO,UACL,IAAI,oBAAoB;GAAE,SAAS;GAA+B,QAAQ;GAAQ;EAAM,CAAC;CAC7F,CAAC,GAEsC,OAAO;EAAE,QAAQ;EAAQ,OAAO,MAAM;CAAS,CAAC;AACzF,CAAC;;AAGH,MAAM,oBACJ,YACA,QAC+C;CAC/C,MAAM,yBAAS,IAAI,IAAmC;CACtD,IAAI,YAAY;CAEhB,KAAK,MAAM,CAAC,MAAM,SAAS,YAAY;EACrC,IAAI,aAAa,GAAG;EAEpB,MAAM,OAAO,KAAK,MAAM,GAAG,SAAS;EACpC,aAAa,KAAK;EAClB,OAAO,IAAI,MAAM,IAAI;CACvB;CAEA,OAAO;AACT;AAEA,MAAM,eAAe,OAAkB,QAA6B,WAClE,OAAO,IAAI,aAAa;CACtB,MAAM,aAAa,OAAO,iBAAiB,OAAO,QAAQ,MAAM;CAGhE,MAAM,eAAe,OAAO,OAAO,IAAI;EACrC,WAAW,kBAAkB,WAAW,OAAO,OAAO,aAAa;EACnE,QAAO,UACL,iBAAiB,sBACb,QACA,IAAI,oBAAoB;GAAE,SAAS;GAAuB,QAAQ;GAAQ;EAAM,CAAC;CACzF,CAAC;CAED,MAAM,UAAU,OAAO,YAAY,MAAM;CAQzC,MAAM,WAAW,eAAe,OANV,OAAO,IAAI;EAC/B,WAAW,QAAQ,KAAK,aAAa,OAAO;EAC5C,QAAO,UACL,IAAI,oBAAoB;GAAE,SAAS;GAAuB,QAAQ;GAAQ;EAAM,CAAC;CACrF,CAAC,CAEqC;CAEtC,IAAI,aAAa,KAAA,GACf,OAAO,OAAO,OAAO,KACnB,IAAI,oBAAoB;EAAE,SAAS;EAAuB,QAAQ;CAAO,CAAC,CAC5E;CAQF,MAAM,sBALe,CAAC,GAAG,WAAW,cAAc,OAAO,CAAC,EAAE,QACzD,OAAO,SAAS,QAAQ,KAAK,QAC9B,CAGqC,IAAI,OAAO;CAElD,MAAM,aAAa,sBACjB,WAAW,OACX,sBACI,iBAAiB,WAAW,eAAe,OAAO,iBAAiB,IACnE,WAAW,eACf,aAAa,MACf;CAcA,OAAO,OAAO,kBAAkB,OAZT,OAAO,IAAI;EAChC,WAAW,uBAAuB,UAAU,QAAQ;GAAE;GAAY;EAAoB,CAAC;EACvF,QAAO,UACL,iBAAiB,sBACb,QACA,IAAI,oBAAoB;GACtB,SAAS;GACT,QAAQ;GACR;EACF,CAAC;CACT,CAAC,GAEwC;EACvC,QAAQ;EACR,OAAO,aAAa,SAAS,MAAM;EACnC,YAAY,SAAS;CACvB,CAAC;AACH,CAAC;AAEH,MAAM,eAAe,OAAkB,WACrC,OAAO,IAAI,aAAa;CACtB,MAAM,EAAE,UAAU,OAAO,iBAAiB,OAAO,QAAQ,MAAM;CAQ/D,OAAO,OAAO,kBAAkB,OANZ,OAAO,IAAI;EAC7B,WAAW,gBAAgB,KAAK;EAChC,QAAO,UACL,IAAI,oBAAoB;GAAE,SAAS;GAA+B,QAAQ;GAAQ;EAAM,CAAC;CAC7F,CAAC,GAEqC;EAAE,QAAQ;EAAQ,OAAO,MAAM;CAAS,CAAC;AACjF,CAAC;;;;;AAMH,MAAa,qBACX,OACA,QACA,QACA,WACqD;CACrD,IAAI,WAAW,OAAO,OAAO,WAAW,KAAK;CAE7C,IAAI,WAAW,QAAQ,OAAO,YAAY,OAAO,MAAM;CAEvD,IAAI,WAAW,QAAQ,OAAO,YAAY,OAAO,QAAQ,MAAM;CAE/D,OAAO,YAAY,OAAO,MAAM;AAClC"}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import { FileExtractorError } from "../errors.mjs";
|
|
2
|
+
import { ExtractedFile, FileInput } from "../format.mjs";
|
|
3
|
+
import { FileExtractorLimits } from "../limits.mjs";
|
|
4
|
+
import { ParsedFileFormat } from "./extract-file.mjs";
|
|
5
|
+
import { Effect } from "effect";
|
|
6
|
+
import * as Schema from "effect/Schema";
|
|
7
|
+
|
|
8
|
+
//#region src/node/extraction-isolation.d.ts
|
|
9
|
+
/** Limits for the worker each PDF, DOCX, XLSX, or PPTX extraction runs in. */
|
|
10
|
+
type WorkerIsolationOptions = {
|
|
11
|
+
/** V8 old-generation heap of the worker, in MB. Default 256. */readonly maxOldGenerationSizeMb?: number; /** V8 young-generation heap of the worker, in MB. Default 32. */
|
|
12
|
+
readonly maxYoungGenerationSizeMb?: number; /** Stack of the worker's main thread, in MB. Default 4 (Node's own default). */
|
|
13
|
+
readonly stackSizeMb?: number;
|
|
14
|
+
/**
|
|
15
|
+
* Wall-clock time a worker may run before it is terminated, in ms. Default 30,000. At most
|
|
16
|
+
* 2³¹−1 (Node's largest timer delay).
|
|
17
|
+
*/
|
|
18
|
+
readonly timeoutMs?: number;
|
|
19
|
+
/**
|
|
20
|
+
* This layer's share of the realm-wide worker pool: at most this many of its extractions run at
|
|
21
|
+
* once. Every layer in a JavaScript realm (the main thread, or each worker thread that builds
|
|
22
|
+
* the layer) shares one pool of 4 workers, however often the layer is built, so this can only
|
|
23
|
+
* lower the layer's share. 1 to 4; default 4.
|
|
24
|
+
*/
|
|
25
|
+
readonly maxConcurrentWorkers?: number;
|
|
26
|
+
/**
|
|
27
|
+
* How long an extraction may wait for a worker slot, in ms, before it fails with
|
|
28
|
+
* `reason: 'busy'` without starting a worker. Slots are handed out first come, first served, and
|
|
29
|
+
* a slot freed after the deadline never admits the extraction. Default: the layer's `timeoutMs`.
|
|
30
|
+
* At most 2³¹−1 (Node's largest timer delay).
|
|
31
|
+
*/
|
|
32
|
+
readonly maxQueueWaitMs?: number;
|
|
33
|
+
/**
|
|
34
|
+
* The worker entry. Default: the package's self-contained `dist/node/extraction-worker.mjs`,
|
|
35
|
+
* found from this module's own location (in `dist`, or in `src` when a workspace or a bundler
|
|
36
|
+
* that keeps module locations, such as Next.js Turbopack, runs the source). When this module has
|
|
37
|
+
* been bundled into a file of another name, there is no default and extraction fails with
|
|
38
|
+
* `reason: 'worker-unavailable'` without starting a worker; point this at a copy of
|
|
39
|
+
* `@yolk-sdk/extractors/node/extraction-worker` instead.
|
|
40
|
+
*/
|
|
41
|
+
readonly workerUrl?: string | URL;
|
|
42
|
+
};
|
|
43
|
+
/**
|
|
44
|
+
* Where parsers run. `'worker'` (the default) and an options object run each PDF, DOCX, XLSX, and
|
|
45
|
+
* PPTX extraction in a fresh `worker_threads` worker with V8 heap, stack, and time limits.
|
|
46
|
+
* `'none'` runs parsers in the calling thread: only for environments without worker threads, and
|
|
47
|
+
* unsafe for untrusted input (a crafted file can exhaust the process heap or block the event loop).
|
|
48
|
+
*/
|
|
49
|
+
type FileExtractorIsolation = 'worker' | 'none' | WorkerIsolationOptions;
|
|
50
|
+
/** Resolved worker settings; every value is a positive integer, and timers fit `setTimeout`. */
|
|
51
|
+
declare const WorkerIsolationSettings: Schema.Struct<{
|
|
52
|
+
readonly maxOldGenerationSizeMb: Schema.Int;
|
|
53
|
+
readonly maxYoungGenerationSizeMb: Schema.Int;
|
|
54
|
+
readonly stackSizeMb: Schema.Int;
|
|
55
|
+
readonly timeoutMs: Schema.Int;
|
|
56
|
+
readonly maxConcurrentWorkers: Schema.Int;
|
|
57
|
+
readonly maxQueueWaitMs: Schema.Int;
|
|
58
|
+
}>;
|
|
59
|
+
type WorkerIsolationSettings = typeof WorkerIsolationSettings.Type;
|
|
60
|
+
/**
|
|
61
|
+
* Defaults. 256 MB of old generation holds SheetJS's cell objects for the default limits (100,000
|
|
62
|
+
* visited cells, 50 MiB expanded) several times over and PDF.js's working set for ordinary PDFs,
|
|
63
|
+
* and the realm-wide pool of four workers keeps their V8 heaps near 1 GB together. That bounds
|
|
64
|
+
* heap, not the process: Buffers and PDF.js's decoded data live outside it (see the README).
|
|
65
|
+
* 32 MB of young generation is twice V8's usual 64-bit default, enough for short-lived parser
|
|
66
|
+
* strings. 30 s is far above legitimate parse times at the default limits (well under a second
|
|
67
|
+
* for the research workbooks) and below common serverless request budgets; a queued extraction
|
|
68
|
+
* waits at most as long again for a slot.
|
|
69
|
+
*/
|
|
70
|
+
declare const defaultWorkerIsolation: WorkerIsolationSettings;
|
|
71
|
+
/**
|
|
72
|
+
* The built worker for an isolation module at `moduleUrl`: `dist/node/extraction-worker.mjs` of
|
|
73
|
+
* the same package, from `dist/node/extraction-isolation.mjs` (installed) or
|
|
74
|
+
* `src/node/extraction-isolation.ts` (workspace source, which must be built first). Any other
|
|
75
|
+
* location (this module bundled into a chunk) has no default: `undefined`, and nothing is started.
|
|
76
|
+
* The URL is derived from the module URL at runtime, never written as
|
|
77
|
+
* `new URL('./…', import.meta.url)`: bundlers rewrite that pattern into a copied asset.
|
|
78
|
+
*/
|
|
79
|
+
declare const extractionWorkerUrlFor: (moduleUrl: string) => URL | undefined;
|
|
80
|
+
/** The default worker entry for this module; see `extractionWorkerUrlFor`. */
|
|
81
|
+
declare const defaultExtractionWorkerUrl: () => URL | undefined;
|
|
82
|
+
type WorkerExtractor = (input: FileInput, format: ParsedFileFormat, limits: FileExtractorLimits) => Effect.Effect<ExtractedFile, FileExtractorError>;
|
|
83
|
+
/**
|
|
84
|
+
* Run each extraction in a fresh worker: the input is copied into a transferred buffer, the
|
|
85
|
+
* worker returns only text and metadata, and the worker is terminated when the result arrives,
|
|
86
|
+
* the timeout fires, or the caller is interrupted. Admission goes through this layer's share and
|
|
87
|
+
* the realm-wide pool (`worker-admission.ts`); a slot is freed only once its worker has
|
|
88
|
+
* terminated. A worker that cannot start, or a missing `workerUrl`, fails closed with
|
|
89
|
+
* `reason: 'worker-unavailable'`; there is no in-process fallback.
|
|
90
|
+
*/
|
|
91
|
+
declare const makeWorkerExtractor: (settings: WorkerIsolationSettings, workerUrl: string | URL | undefined) => Effect.Effect<WorkerExtractor>;
|
|
92
|
+
//#endregion
|
|
93
|
+
export { FileExtractorIsolation, WorkerExtractor, WorkerIsolationOptions, WorkerIsolationSettings, defaultExtractionWorkerUrl, defaultWorkerIsolation, extractionWorkerUrlFor, makeWorkerExtractor };
|
|
94
|
+
//# sourceMappingURL=extraction-isolation.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extraction-isolation.d.mts","names":[],"sources":["../../src/node/extraction-isolation.ts"],"mappings":";;;;;;;;;KAaY,sBAAA;EAAA,yEAED,sBAAA;WAEA,wBAAA,WAFA;EAAA,SAIA,WAAA;EAAA;;;;EAAA,SAKA,SAAA;EAuBqB;;AAAG;AASnC;;;EATgC,SAhBrB,oBAAA;EAyBoE;AAW/E;;;;;EAX+E,SAlBpE,cAAA;;;;;;;;;WASA,SAAA,YAAqB,GAAG;AAAA;;;;;;;KASvB,sBAAA,uBAA6C,sBAAsB;;cAWlE,uBAAA,EAAuB,MAAA,CAAA,MAAA;EAAA;;;;;;;KAWxB,uBAAA,UAAiC,uBAAA,CAAwB,IAAI;;AAAzE;;;;AAAyE;AAYzE;;;;cAAa,sBAAA,EAAwB,uBAOpC;AAaD;;;;AAA8D;AAM9D;;;AANA,cAAa,sBAAA,GAA0B,SAAA,aAAoB,GAAG;AAMvB;AAAA,cAA1B,0BAAA,QAA0B,GAAA;AAAA,KAiG3B,eAAA,IACV,KAAA,EAAO,SAAA,EACP,MAAA,EAAQ,gBAAA,EACR,MAAA,EAAQ,mBAAA,KACL,MAAA,CAAO,MAAA,CAAO,aAAA,EAAe,kBAAA;;;;;;;;;cAcrB,mBAAA,GACX,QAAA,EAAU,uBAAA,EACV,SAAA,WAAoB,GAAA,iBACnB,MAAA,CAAO,MAAA,CAAO,eAAA"}
|