@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
package/src/errors.ts ADDED
@@ -0,0 +1,96 @@
1
+ import { Match } from 'effect'
2
+ import * as Schema from 'effect/Schema'
3
+
4
+ /** The SheetJS tarball consumers install; npm `xlsx` stops at the vulnerable 0.18.5. */
5
+ export const sheetJsInstallCommand = 'pnpm add https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz'
6
+
7
+ /** Lowest SheetJS release with fixes for CVE-2023-30533 and CVE-2024-22363. */
8
+ export const minimumSheetJsVersion = '0.20.3'
9
+
10
+ /**
11
+ * Why an isolated extraction stopped before the parsers finished:
12
+ *
13
+ * - `resource-limit`: the worker ran out of its V8 heap (`maxOldGenerationSizeMb`, …);
14
+ * - `timeout`: the worker exceeded `timeoutMs` and was terminated;
15
+ * - `worker-unavailable`: the worker could not start (a missing or unloadable worker file);
16
+ * - `worker-failed`: the worker crashed or exited without a result;
17
+ * - `busy`: no worker slot freed up within `maxQueueWaitMs`, so no worker was started.
18
+ */
19
+ export const FileExtractionFailureReason = Schema.Literals([
20
+ 'resource-limit',
21
+ 'timeout',
22
+ 'worker-unavailable',
23
+ 'worker-failed',
24
+ 'busy'
25
+ ])
26
+
27
+ export type FileExtractionFailureReason = typeof FileExtractionFailureReason.Type
28
+
29
+ /**
30
+ * Reading, validating, or bounding a file failed. `message` is safe to show to users. `reason`
31
+ * is set only when an isolated worker was stopped or never admitted (see
32
+ * `FileExtractionFailureReason`).
33
+ */
34
+ export class FileExtractionError extends Schema.TaggedError<FileExtractionError>()(
35
+ 'FileExtractionError',
36
+ {
37
+ message: Schema.String,
38
+ format: Schema.String,
39
+ reason: Schema.optional(FileExtractionFailureReason),
40
+ cause: Schema.optional(Schema.Unknown)
41
+ }
42
+ ) {}
43
+
44
+ /** Neither the filename extension nor the media type maps to a supported format. */
45
+ export class UnsupportedFileFormatError extends Schema.TaggedError<UnsupportedFileFormatError>()(
46
+ 'UnsupportedFileFormatError',
47
+ {
48
+ filename: Schema.String,
49
+ mediaType: Schema.String
50
+ }
51
+ ) {
52
+ get message(): string {
53
+ return `Unsupported file format: ${this.filename}`
54
+ }
55
+ }
56
+
57
+ /** A DOCX, XLSX, or PPTX ZIP archive failed bounded validation or normalization. */
58
+ export class OfficeArchiveError extends Schema.TaggedError<OfficeArchiveError>()(
59
+ 'OfficeArchiveError',
60
+ {
61
+ message: Schema.String,
62
+ expandedBytes: Schema.optional(Schema.Number)
63
+ }
64
+ ) {}
65
+
66
+ /**
67
+ * SheetJS (`xlsx`, an optional peer) is missing, is not a usable SheetJS module, or is older than
68
+ * 0.20.3. This is host misconfiguration, not a problem with the uploaded file.
69
+ */
70
+ export class SheetJsUnavailableError extends Schema.TaggedError<SheetJsUnavailableError>()(
71
+ 'SheetJsUnavailableError',
72
+ {
73
+ reason: Schema.Literals(['missing', 'invalid', 'outdated']),
74
+ installedVersion: Schema.optional(Schema.String),
75
+ cause: Schema.optional(Schema.Unknown)
76
+ }
77
+ ) {
78
+ get message(): string {
79
+ const problem = Match.value(this.reason).pipe(
80
+ Match.when('missing', () => 'SheetJS (xlsx) is not installed'),
81
+ Match.when(
82
+ 'outdated',
83
+ () => `SheetJS ${this.installedVersion ?? 'unknown'} is older than ${minimumSheetJsVersion}`
84
+ ),
85
+ Match.when('invalid', () => 'The installed xlsx module is not a usable SheetJS build'),
86
+ Match.exhaustive
87
+ )
88
+
89
+ return `${problem}. XLSX extraction needs SheetJS ${minimumSheetJsVersion} or newer from the SheetJS CDN (npm xlsx is unmaintained and vulnerable): ${sheetJsInstallCommand}`
90
+ }
91
+ }
92
+
93
+ export type FileExtractorError =
94
+ | FileExtractionError
95
+ | UnsupportedFileFormatError
96
+ | SheetJsUnavailableError
package/src/format.ts ADDED
@@ -0,0 +1,84 @@
1
+ export const extractedFileFormats = [
2
+ 'csv',
3
+ 'docx',
4
+ 'json',
5
+ 'markdown',
6
+ 'pdf',
7
+ 'pptx',
8
+ 'text',
9
+ 'xlsx'
10
+ ] as const
11
+
12
+ export type ExtractedFileFormat = (typeof extractedFileFormats)[number]
13
+
14
+ /** Office Open XML formats; their ZIP archives are validated before any parser reads them. */
15
+ export type OfficeFileFormat = 'docx' | 'pptx' | 'xlsx'
16
+
17
+ export type FileInput = {
18
+ readonly filename: string
19
+ readonly mediaType: string
20
+ readonly bytes: Uint8Array
21
+ }
22
+
23
+ export type ExtractedFileMetadata = {
24
+ readonly format: ExtractedFileFormat
25
+ readonly title?: string
26
+ readonly pageCount?: number
27
+ readonly sheetNames?: ReadonlyArray<string>
28
+ }
29
+
30
+ export type ExtractedFile = {
31
+ readonly content: string
32
+ readonly metadata: ExtractedFileMetadata
33
+ }
34
+
35
+ const extensionFor = (filename: string) => {
36
+ const lower = filename.toLowerCase()
37
+ const dotIndex = lower.lastIndexOf('.')
38
+
39
+ return dotIndex === -1 ? '' : lower.slice(dotIndex + 1)
40
+ }
41
+
42
+ const formatsByExtension: ReadonlyMap<string, ExtractedFileFormat> = new Map([
43
+ ['txt', 'text'],
44
+ ['md', 'markdown'],
45
+ ['markdown', 'markdown'],
46
+ ['csv', 'csv'],
47
+ ['json', 'json'],
48
+ ['pdf', 'pdf'],
49
+ ['docx', 'docx'],
50
+ ['xlsx', 'xlsx'],
51
+ ['pptx', 'pptx']
52
+ ])
53
+
54
+ const formatsByMediaType: ReadonlyMap<string, ExtractedFileFormat> = new Map([
55
+ ['application/pdf', 'pdf'],
56
+ ['application/vnd.openxmlformats-officedocument.wordprocessingml.document', 'docx'],
57
+ ['application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', 'xlsx'],
58
+ ['application/vnd.openxmlformats-officedocument.presentationml.presentation', 'pptx'],
59
+ ['application/json', 'json'],
60
+ ['text/csv', 'csv'],
61
+ ['text/markdown', 'markdown']
62
+ ])
63
+
64
+ /**
65
+ * Pick the extraction format: the filename extension wins, then the media type, then any
66
+ * `text/*` media type as plain text. `undefined` means the file is unsupported.
67
+ */
68
+ export const fileFormatFor = (input: {
69
+ readonly filename: string
70
+ readonly mediaType: string
71
+ }): ExtractedFileFormat | undefined => {
72
+ const byExtension = formatsByExtension.get(extensionFor(input.filename))
73
+
74
+ if (byExtension !== undefined) return byExtension
75
+
76
+ const byMediaType = formatsByMediaType.get(input.mediaType)
77
+
78
+ if (byMediaType !== undefined) return byMediaType
79
+
80
+ return input.mediaType.startsWith('text/') ? 'text' : undefined
81
+ }
82
+
83
+ export const isOfficeFileFormat = (format: ExtractedFileFormat): format is OfficeFileFormat =>
84
+ format === 'docx' || format === 'pptx' || format === 'xlsx'
package/src/index.ts ADDED
@@ -0,0 +1,32 @@
1
+ // Runtime-portable root: types, errors, format detection, limits, and the service tag. Parsers
2
+ // live behind `@yolk-sdk/extractors/node`.
3
+
4
+ export {
5
+ FileExtractionError,
6
+ FileExtractionFailureReason,
7
+ minimumSheetJsVersion,
8
+ OfficeArchiveError,
9
+ SheetJsUnavailableError,
10
+ sheetJsInstallCommand,
11
+ UnsupportedFileFormatError
12
+ } from './errors.ts'
13
+
14
+ export type { FileExtractorError } from './errors.ts'
15
+
16
+ export { extractedFileFormats, fileFormatFor, isOfficeFileFormat } from './format.ts'
17
+
18
+ export type {
19
+ ExtractedFile,
20
+ ExtractedFileFormat,
21
+ ExtractedFileMetadata,
22
+ FileInput,
23
+ OfficeFileFormat
24
+ } from './format.ts'
25
+
26
+ export { defaultFileExtractorLimits, FileExtractorLimits } from './limits.ts'
27
+
28
+ export { sanitizeExtractedText } from './sanitize.ts'
29
+
30
+ export { FileExtractor } from './service.ts'
31
+
32
+ export type { FileExtractorApi } from './service.ts'
@@ -0,0 +1,101 @@
1
+ import { Effect, Layer, Match, Predicate } from 'effect'
2
+ import type { ExtractedKnowledgeDocument, KnowledgeSource } from '@yolk-sdk/knowledge/documents'
3
+ import { KnowledgeExtractionError } from '@yolk-sdk/knowledge/errors'
4
+ import { KnowledgeExtractor } from '@yolk-sdk/knowledge/extraction'
5
+ import type { KnowledgeExtractorApi, LoadedKnowledgeSource } from '@yolk-sdk/knowledge/extraction'
6
+ import type { ExtractedFile } from './format.ts'
7
+ import { FileExtractor } from './service.ts'
8
+ import type { FileExtractorApi } from './service.ts'
9
+
10
+ /** Percent-decode a path segment; malformed escapes (`%zz`, invalid UTF-8) stay encoded. */
11
+ const decodeSegment = (segment: string) => {
12
+ try {
13
+ return decodeURIComponent(segment)
14
+ } catch {
15
+ return segment
16
+ }
17
+ }
18
+
19
+ const urlFilename = (url: string) => {
20
+ if (!URL.canParse(url)) return url
21
+
22
+ const segments = new URL(url).pathname.split('/').filter(segment => segment.length > 0)
23
+ const last = segments.at(-1)
24
+
25
+ return last === undefined ? url : decodeSegment(last)
26
+ }
27
+
28
+ /** The filename used for format detection: file name or ref, URL path, or text label. */
29
+ const filenameFor = (source: KnowledgeSource) =>
30
+ Match.value(source).pipe(
31
+ Match.tagsExhaustive({
32
+ File: file => file.name ?? file.ref,
33
+ Url: url => urlFilename(url.url),
34
+ Text: text => text.label ?? 'text'
35
+ })
36
+ )
37
+
38
+ /** Loaded media type first, then the file source's, then `text/plain` for text sources. */
39
+ const mediaTypeFor = (loaded: LoadedKnowledgeSource) => {
40
+ if (loaded.mediaType !== undefined) return loaded.mediaType
41
+
42
+ if (Predicate.isTagged(loaded.source, 'File') && loaded.source.mediaType !== undefined)
43
+ return loaded.source.mediaType
44
+
45
+ return Predicate.isTagged(loaded.source, 'Text') ? 'text/plain' : ''
46
+ }
47
+
48
+ const documentFrom = (
49
+ loaded: LoadedKnowledgeSource,
50
+ extracted: ExtractedFile
51
+ ): ExtractedKnowledgeDocument => {
52
+ const { title, ...fileMetadata } = extracted.metadata
53
+ const metadata = { ...fileMetadata, ...loaded.metadata }
54
+
55
+ return title === undefined || title.trim().length === 0
56
+ ? { content: extracted.content, metadata }
57
+ : { content: extracted.content, title: title.trim(), metadata }
58
+ }
59
+
60
+ /**
61
+ * A `KnowledgeExtractor` backed by a `FileExtractor`. String content is already text and passes
62
+ * through unchanged (it must not be blank); bytes are extracted with the format chosen from the
63
+ * source name and media type. File metadata (`format`, `pageCount`, `sheetNames`) is merged
64
+ * under the loaded source's own metadata, and the extracted title becomes the document title.
65
+ */
66
+ export const makeFileKnowledgeExtractor = (extractor: FileExtractorApi): KnowledgeExtractorApi => ({
67
+ extract: loaded => {
68
+ if (Predicate.isString(loaded.content)) {
69
+ const content = loaded.content
70
+
71
+ return content.trim().length === 0
72
+ ? Effect.fail(new KnowledgeExtractionError({ message: 'Knowledge source text is empty' }))
73
+ : Effect.succeed(
74
+ loaded.metadata === undefined ? { content } : { content, metadata: loaded.metadata }
75
+ )
76
+ }
77
+
78
+ return extractor
79
+ .extract({
80
+ filename: filenameFor(loaded.source),
81
+ mediaType: mediaTypeFor(loaded),
82
+ bytes: loaded.content
83
+ })
84
+ .pipe(
85
+ Effect.map(extracted => documentFrom(loaded, extracted)),
86
+ Effect.mapError(
87
+ error => new KnowledgeExtractionError({ message: error.message, cause: error })
88
+ )
89
+ )
90
+ }
91
+ })
92
+
93
+ /** Provide `KnowledgeExtractor` from the `FileExtractor` in context (for example the Node layer). */
94
+ export const FileKnowledgeExtractorLayer = Layer.effect(
95
+ KnowledgeExtractor,
96
+ Effect.gen(function* () {
97
+ const extractor = yield* FileExtractor
98
+
99
+ return KnowledgeExtractor.of(makeFileKnowledgeExtractor(extractor))
100
+ })
101
+ )
package/src/limits.ts ADDED
@@ -0,0 +1,49 @@
1
+ import * as Schema from 'effect/Schema'
2
+
3
+ const PositiveSafeInteger = Schema.Int.pipe(
4
+ Schema.check(Schema.isGreaterThan(0)),
5
+ Schema.check(Schema.isLessThanOrEqualTo(Number.MAX_SAFE_INTEGER))
6
+ )
7
+
8
+ /**
9
+ * Smallest `maxXlsxTextCharacters`: one more than the space always reserved for the
10
+ * `[Some hyperlinks omitted: output limit and hyperlink limit]` marker (61 characters with its
11
+ * leading blank line), so a workbook with links can still produce text.
12
+ */
13
+ export const minimumXlsxTextCharacters = 62
14
+
15
+ /**
16
+ * Work and output bounds for one `extract` call. Every value is a positive integer, and
17
+ * `maxXlsxTextCharacters` is at least `minimumXlsxTextCharacters`.
18
+ */
19
+ export const FileExtractorLimits = Schema.Struct({
20
+ /** Input bytes accepted for any format. */
21
+ maxInputBytes: PositiveSafeInteger,
22
+ /** Entries in a DOCX, XLSX, or PPTX ZIP archive. */
23
+ maxArchiveEntries: PositiveSafeInteger,
24
+ /** Total inflated bytes of a DOCX, XLSX, or PPTX archive, counted while inflating. */
25
+ maxExpandedBytes: PositiveSafeInteger,
26
+ /** Worksheets in a workbook. */
27
+ maxXlsxSheets: PositiveSafeInteger,
28
+ /** Cells visited across all worksheet ranges (absent cells count too). */
29
+ maxXlsxCellVisits: PositiveSafeInteger,
30
+ /** Characters of XLSX text, including hyperlink annotations and the omitted-links marker. */
31
+ maxXlsxTextCharacters: PositiveSafeInteger.pipe(
32
+ Schema.check(Schema.isGreaterThanOrEqualTo(minimumXlsxTextCharacters))
33
+ ),
34
+ /** Hyperlinks read per workbook; later ones are ignored (still removed before SheetJS). */
35
+ maxXlsxHyperlinks: PositiveSafeInteger
36
+ })
37
+
38
+ export type FileExtractorLimits = typeof FileExtractorLimits.Type
39
+
40
+ /** Default limits; override any of them through `makeFileExtractorLayer({ limits })`. */
41
+ export const defaultFileExtractorLimits: FileExtractorLimits = {
42
+ maxInputBytes: 50 * 1024 * 1024,
43
+ maxArchiveEntries: 10_000,
44
+ maxExpandedBytes: 50 * 1024 * 1024,
45
+ maxXlsxSheets: 100,
46
+ maxXlsxCellVisits: 100_000,
47
+ maxXlsxTextCharacters: 512 * 1024,
48
+ maxXlsxHyperlinks: 10_000
49
+ }
@@ -0,0 +1,269 @@
1
+ import { Buffer } from 'node:buffer'
2
+ import { Effect, Option, Predicate } from 'effect'
3
+ import { FileExtractionError } from '../errors.ts'
4
+ import type { FileExtractorError } from '../errors.ts'
5
+ import type {
6
+ ExtractedFile,
7
+ ExtractedFileFormat,
8
+ ExtractedFileMetadata,
9
+ FileInput,
10
+ OfficeFileFormat
11
+ } from '../format.ts'
12
+ import type { FileExtractorLimits } from '../limits.ts'
13
+ import { sanitizeExtractedText } from '../sanitize.ts'
14
+ import { readOfficeArchive, storedArchive } from './office-archive.ts'
15
+ import type { NormalizedOfficeArchive } from './office-archive.ts'
16
+ import { extractPptxText } from './pptx-text.ts'
17
+ import { asXlsxWorkbook, loadSheetJs } from './sheetjs.ts'
18
+ import type { SheetJsLoader } from './sheetjs.ts'
19
+ import { resolveXlsxHyperlinks } from './xlsx-hyperlinks.ts'
20
+ import { buildSheetJsInput } from './xlsx-sheetjs-input.ts'
21
+ import { extractBoundedXlsxText } from './xlsx-text.ts'
22
+
23
+ /** Formats whose bytes go through a parser (and, by default, an isolated worker). */
24
+ export type ParsedFileFormat = 'pdf' | OfficeFileFormat
25
+
26
+ export const isParsedFileFormat = (format: ExtractedFileFormat): format is ParsedFileFormat =>
27
+ format === 'pdf' || format === 'docx' || format === 'pptx' || format === 'xlsx'
28
+
29
+ /** Sanitize extracted text; empty results fail. */
30
+ export const makeExtractedFile = (content: string, metadata: ExtractedFileMetadata) => {
31
+ const sanitized = sanitizeExtractedText(content)
32
+
33
+ if (sanitized.length === 0)
34
+ return Effect.fail(
35
+ new FileExtractionError({
36
+ message: 'Extracted file content is empty',
37
+ format: metadata.format
38
+ })
39
+ )
40
+
41
+ return Effect.succeed<ExtractedFile>({ content: sanitized, metadata })
42
+ }
43
+
44
+ /**
45
+ * PDF.js 6 releases a document through its loading task (`PDFDocumentProxy` no longer has
46
+ * `destroy()`); unpdf releases the documents it opens the same way.
47
+ */
48
+ type PdfDocumentResource = {
49
+ readonly loadingTask: { readonly destroy: () => Promise<void> }
50
+ }
51
+
52
+ /** Run `use` with an opened PDF document and always release the parser afterwards. */
53
+ export const withAcquiredPdfDocument = <D extends PdfDocumentResource, A, E, R>(
54
+ open: Effect.Effect<D, E, R>,
55
+ use: (document: D) => Effect.Effect<A, E, R>,
56
+ format: ExtractedFileFormat
57
+ ) =>
58
+ Effect.scoped(
59
+ Effect.gen(function* () {
60
+ const document = yield* Effect.acquireRelease(open, document =>
61
+ Effect.tryPromise({
62
+ try: () => document.loadingTask.destroy(),
63
+ catch: () => new FileExtractionError({ message: 'Could not release PDF parser', format })
64
+ }).pipe(Effect.ignore)
65
+ )
66
+
67
+ return yield* use(document)
68
+ })
69
+ )
70
+
71
+ const extractPdf = (input: FileInput) =>
72
+ Effect.gen(function* () {
73
+ // Loaded on first use, so a worker extracting another format never loads PDF.js.
74
+ const { extractText, getDocumentProxy, getMeta } = yield* Effect.tryPromise({
75
+ try: () => import('unpdf'),
76
+ catch: cause =>
77
+ new FileExtractionError({ message: 'Could not read PDF', format: 'pdf', cause })
78
+ })
79
+
80
+ return yield* withAcquiredPdfDocument(
81
+ Effect.tryPromise({
82
+ // PDF.js may detach the buffer it is given; never hand it the caller's bytes.
83
+ try: () => getDocumentProxy(new Uint8Array(input.bytes)),
84
+ catch: cause =>
85
+ new FileExtractionError({ message: 'Could not read PDF', format: 'pdf', cause })
86
+ }),
87
+ document =>
88
+ Effect.gen(function* () {
89
+ const extracted = yield* Effect.tryPromise({
90
+ try: () => extractText(document, { mergePages: true }),
91
+ catch: cause =>
92
+ new FileExtractionError({
93
+ message: 'Could not extract PDF text',
94
+ format: 'pdf',
95
+ cause
96
+ })
97
+ })
98
+
99
+ const meta = yield* Effect.tryPromise({
100
+ try: () => getMeta(document),
101
+ catch: cause =>
102
+ new FileExtractionError({
103
+ message: 'Could not read PDF metadata',
104
+ format: 'pdf',
105
+ cause
106
+ })
107
+ }).pipe(Effect.option)
108
+
109
+ const rawTitle = Option.isSome(meta) ? meta.value.info.Title : undefined
110
+ const title = Predicate.isString(rawTitle) && rawTitle.length > 0 ? rawTitle : undefined
111
+
112
+ const metadata: ExtractedFileMetadata =
113
+ title === undefined
114
+ ? { format: 'pdf', pageCount: extracted.totalPages }
115
+ : { format: 'pdf', title, pageCount: extracted.totalPages }
116
+
117
+ return yield* makeExtractedFile(extracted.text, metadata)
118
+ }),
119
+ 'pdf'
120
+ )
121
+ })
122
+
123
+ const validatedArchive = (
124
+ input: FileInput,
125
+ format: OfficeFileFormat,
126
+ limits: FileExtractorLimits
127
+ ): Effect.Effect<NormalizedOfficeArchive, FileExtractionError> =>
128
+ readOfficeArchive(
129
+ input.bytes,
130
+ format,
131
+ limits,
132
+ // One extra tag detects a workbook over the cap.
133
+ format === 'xlsx' ? { maxHyperlinkTags: limits.maxXlsxHyperlinks + 1 } : {}
134
+ ).pipe(
135
+ Effect.mapError(
136
+ error => new FileExtractionError({ message: error.message, format, cause: error })
137
+ )
138
+ )
139
+
140
+ const extractDocx = (input: FileInput, limits: FileExtractorLimits) =>
141
+ Effect.gen(function* () {
142
+ const { parts } = yield* validatedArchive(input, 'docx', limits)
143
+
144
+ const result = yield* Effect.tryPromise({
145
+ try: async () => {
146
+ const { default: mammoth } = await import('mammoth')
147
+
148
+ return await mammoth.extractRawText({ buffer: Buffer.from(storedArchive(parts)) })
149
+ },
150
+ catch: cause =>
151
+ new FileExtractionError({ message: 'Could not extract DOCX text', format: 'docx', cause })
152
+ })
153
+
154
+ return yield* makeExtractedFile(result.value, { format: 'docx', title: input.filename })
155
+ })
156
+
157
+ /** Keep the first `max` tags in archive order. */
158
+ const capHyperlinkTags = (
159
+ tagsByPart: ReadonlyMap<string, ReadonlyArray<string>>,
160
+ max: number
161
+ ): ReadonlyMap<string, ReadonlyArray<string>> => {
162
+ const capped = new Map<string, ReadonlyArray<string>>()
163
+ let remaining = max
164
+
165
+ for (const [part, tags] of tagsByPart) {
166
+ if (remaining <= 0) break
167
+
168
+ const kept = tags.slice(0, remaining)
169
+ remaining -= kept.length
170
+ capped.set(part, kept)
171
+ }
172
+
173
+ return capped
174
+ }
175
+
176
+ const extractXlsx = (input: FileInput, limits: FileExtractorLimits, loader: SheetJsLoader) =>
177
+ Effect.gen(function* () {
178
+ const normalized = yield* validatedArchive(input, 'xlsx', limits)
179
+
180
+ // SheetJS never sees the uploaded archive: only this allowlisted rebuild of its parts.
181
+ const sheetJsInput = yield* Effect.try({
182
+ try: () => buildSheetJsInput(normalized.parts, limits.maxXlsxSheets),
183
+ catch: cause =>
184
+ cause instanceof FileExtractionError
185
+ ? cause
186
+ : new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx', cause })
187
+ })
188
+
189
+ const sheetJs = yield* loadSheetJs(loader)
190
+
191
+ const parsed = yield* Effect.try({
192
+ try: () => sheetJs.read(sheetJsInput.archive),
193
+ catch: cause =>
194
+ new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx', cause })
195
+ })
196
+
197
+ const workbook = asXlsxWorkbook(parsed)
198
+
199
+ if (workbook === undefined)
200
+ return yield* Effect.fail(
201
+ new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx' })
202
+ )
203
+
204
+ // One extra tag was captured to detect that the workbook exceeds the hyperlink cap.
205
+ const capturedTags = [...normalized.hyperlinkTags.values()].reduce(
206
+ (total, tags) => total + tags.length,
207
+ 0
208
+ )
209
+
210
+ const hyperlinksTruncated = capturedTags > limits.maxXlsxHyperlinks
211
+
212
+ const hyperlinks = resolveXlsxHyperlinks(
213
+ normalized.parts,
214
+ hyperlinksTruncated
215
+ ? capHyperlinkTags(normalized.hyperlinkTags, limits.maxXlsxHyperlinks)
216
+ : normalized.hyperlinkTags,
217
+ sheetJsInput.sheets
218
+ )
219
+
220
+ const content = yield* Effect.try({
221
+ try: () => extractBoundedXlsxText(workbook, limits, { hyperlinks, hyperlinksTruncated }),
222
+ catch: cause =>
223
+ cause instanceof FileExtractionError
224
+ ? cause
225
+ : new FileExtractionError({
226
+ message: 'Could not extract XLSX text',
227
+ format: 'xlsx',
228
+ cause
229
+ })
230
+ })
231
+
232
+ return yield* makeExtractedFile(content, {
233
+ format: 'xlsx',
234
+ title: sheetJsInput.title ?? input.filename,
235
+ sheetNames: workbook.SheetNames
236
+ })
237
+ })
238
+
239
+ const extractPptx = (input: FileInput, limits: FileExtractorLimits) =>
240
+ Effect.gen(function* () {
241
+ const { parts } = yield* validatedArchive(input, 'pptx', limits)
242
+
243
+ const text = yield* Effect.try({
244
+ try: () => extractPptxText(parts),
245
+ catch: cause =>
246
+ new FileExtractionError({ message: 'Could not extract PPTX text', format: 'pptx', cause })
247
+ })
248
+
249
+ return yield* makeExtractedFile(text, { format: 'pptx', title: input.filename })
250
+ })
251
+
252
+ /**
253
+ * Extract a PDF, DOCX, XLSX, or PPTX file in the current thread. The Node layer runs this inside
254
+ * an isolated worker unless `isolation: 'none'` is configured.
255
+ */
256
+ export const extractParsedFile = (
257
+ input: FileInput,
258
+ format: ParsedFileFormat,
259
+ limits: FileExtractorLimits,
260
+ loader: SheetJsLoader
261
+ ): Effect.Effect<ExtractedFile, FileExtractorError> => {
262
+ if (format === 'pdf') return extractPdf(input)
263
+
264
+ if (format === 'docx') return extractDocx(input, limits)
265
+
266
+ if (format === 'xlsx') return extractXlsx(input, limits, loader)
267
+
268
+ return extractPptx(input, limits)
269
+ }