@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
package/src/errors.ts
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
import { Match } from 'effect'
|
|
2
|
+
import * as Schema from 'effect/Schema'
|
|
3
|
+
|
|
4
|
+
/** The SheetJS tarball consumers install; npm `xlsx` stops at the vulnerable 0.18.5. */
|
|
5
|
+
export const sheetJsInstallCommand = 'pnpm add https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz'
|
|
6
|
+
|
|
7
|
+
/** Lowest SheetJS release with fixes for CVE-2023-30533 and CVE-2024-22363. */
|
|
8
|
+
export const minimumSheetJsVersion = '0.20.3'
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Why an isolated extraction stopped before the parsers finished:
|
|
12
|
+
*
|
|
13
|
+
* - `resource-limit`: the worker ran out of its V8 heap (`maxOldGenerationSizeMb`, …);
|
|
14
|
+
* - `timeout`: the worker exceeded `timeoutMs` and was terminated;
|
|
15
|
+
* - `worker-unavailable`: the worker could not start (a missing or unloadable worker file);
|
|
16
|
+
* - `worker-failed`: the worker crashed or exited without a result;
|
|
17
|
+
* - `busy`: no worker slot freed up within `maxQueueWaitMs`, so no worker was started.
|
|
18
|
+
*/
|
|
19
|
+
export const FileExtractionFailureReason = Schema.Literals([
|
|
20
|
+
'resource-limit',
|
|
21
|
+
'timeout',
|
|
22
|
+
'worker-unavailable',
|
|
23
|
+
'worker-failed',
|
|
24
|
+
'busy'
|
|
25
|
+
])
|
|
26
|
+
|
|
27
|
+
export type FileExtractionFailureReason = typeof FileExtractionFailureReason.Type
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Reading, validating, or bounding a file failed. `message` is safe to show to users. `reason`
|
|
31
|
+
* is set only when an isolated worker was stopped or never admitted (see
|
|
32
|
+
* `FileExtractionFailureReason`).
|
|
33
|
+
*/
|
|
34
|
+
export class FileExtractionError extends Schema.TaggedError<FileExtractionError>()(
|
|
35
|
+
'FileExtractionError',
|
|
36
|
+
{
|
|
37
|
+
message: Schema.String,
|
|
38
|
+
format: Schema.String,
|
|
39
|
+
reason: Schema.optional(FileExtractionFailureReason),
|
|
40
|
+
cause: Schema.optional(Schema.Unknown)
|
|
41
|
+
}
|
|
42
|
+
) {}
|
|
43
|
+
|
|
44
|
+
/** Neither the filename extension nor the media type maps to a supported format. */
|
|
45
|
+
export class UnsupportedFileFormatError extends Schema.TaggedError<UnsupportedFileFormatError>()(
|
|
46
|
+
'UnsupportedFileFormatError',
|
|
47
|
+
{
|
|
48
|
+
filename: Schema.String,
|
|
49
|
+
mediaType: Schema.String
|
|
50
|
+
}
|
|
51
|
+
) {
|
|
52
|
+
get message(): string {
|
|
53
|
+
return `Unsupported file format: ${this.filename}`
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/** A DOCX, XLSX, or PPTX ZIP archive failed bounded validation or normalization. */
|
|
58
|
+
export class OfficeArchiveError extends Schema.TaggedError<OfficeArchiveError>()(
|
|
59
|
+
'OfficeArchiveError',
|
|
60
|
+
{
|
|
61
|
+
message: Schema.String,
|
|
62
|
+
expandedBytes: Schema.optional(Schema.Number)
|
|
63
|
+
}
|
|
64
|
+
) {}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* SheetJS (`xlsx`, an optional peer) is missing, is not a usable SheetJS module, or is older than
|
|
68
|
+
* 0.20.3. This is host misconfiguration, not a problem with the uploaded file.
|
|
69
|
+
*/
|
|
70
|
+
export class SheetJsUnavailableError extends Schema.TaggedError<SheetJsUnavailableError>()(
|
|
71
|
+
'SheetJsUnavailableError',
|
|
72
|
+
{
|
|
73
|
+
reason: Schema.Literals(['missing', 'invalid', 'outdated']),
|
|
74
|
+
installedVersion: Schema.optional(Schema.String),
|
|
75
|
+
cause: Schema.optional(Schema.Unknown)
|
|
76
|
+
}
|
|
77
|
+
) {
|
|
78
|
+
get message(): string {
|
|
79
|
+
const problem = Match.value(this.reason).pipe(
|
|
80
|
+
Match.when('missing', () => 'SheetJS (xlsx) is not installed'),
|
|
81
|
+
Match.when(
|
|
82
|
+
'outdated',
|
|
83
|
+
() => `SheetJS ${this.installedVersion ?? 'unknown'} is older than ${minimumSheetJsVersion}`
|
|
84
|
+
),
|
|
85
|
+
Match.when('invalid', () => 'The installed xlsx module is not a usable SheetJS build'),
|
|
86
|
+
Match.exhaustive
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
return `${problem}. XLSX extraction needs SheetJS ${minimumSheetJsVersion} or newer from the SheetJS CDN (npm xlsx is unmaintained and vulnerable): ${sheetJsInstallCommand}`
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export type FileExtractorError =
|
|
94
|
+
| FileExtractionError
|
|
95
|
+
| UnsupportedFileFormatError
|
|
96
|
+
| SheetJsUnavailableError
|
package/src/format.ts
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
export const extractedFileFormats = [
|
|
2
|
+
'csv',
|
|
3
|
+
'docx',
|
|
4
|
+
'json',
|
|
5
|
+
'markdown',
|
|
6
|
+
'pdf',
|
|
7
|
+
'pptx',
|
|
8
|
+
'text',
|
|
9
|
+
'xlsx'
|
|
10
|
+
] as const
|
|
11
|
+
|
|
12
|
+
export type ExtractedFileFormat = (typeof extractedFileFormats)[number]
|
|
13
|
+
|
|
14
|
+
/** Office Open XML formats; their ZIP archives are validated before any parser reads them. */
|
|
15
|
+
export type OfficeFileFormat = 'docx' | 'pptx' | 'xlsx'
|
|
16
|
+
|
|
17
|
+
export type FileInput = {
|
|
18
|
+
readonly filename: string
|
|
19
|
+
readonly mediaType: string
|
|
20
|
+
readonly bytes: Uint8Array
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export type ExtractedFileMetadata = {
|
|
24
|
+
readonly format: ExtractedFileFormat
|
|
25
|
+
readonly title?: string
|
|
26
|
+
readonly pageCount?: number
|
|
27
|
+
readonly sheetNames?: ReadonlyArray<string>
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export type ExtractedFile = {
|
|
31
|
+
readonly content: string
|
|
32
|
+
readonly metadata: ExtractedFileMetadata
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const extensionFor = (filename: string) => {
|
|
36
|
+
const lower = filename.toLowerCase()
|
|
37
|
+
const dotIndex = lower.lastIndexOf('.')
|
|
38
|
+
|
|
39
|
+
return dotIndex === -1 ? '' : lower.slice(dotIndex + 1)
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const formatsByExtension: ReadonlyMap<string, ExtractedFileFormat> = new Map([
|
|
43
|
+
['txt', 'text'],
|
|
44
|
+
['md', 'markdown'],
|
|
45
|
+
['markdown', 'markdown'],
|
|
46
|
+
['csv', 'csv'],
|
|
47
|
+
['json', 'json'],
|
|
48
|
+
['pdf', 'pdf'],
|
|
49
|
+
['docx', 'docx'],
|
|
50
|
+
['xlsx', 'xlsx'],
|
|
51
|
+
['pptx', 'pptx']
|
|
52
|
+
])
|
|
53
|
+
|
|
54
|
+
const formatsByMediaType: ReadonlyMap<string, ExtractedFileFormat> = new Map([
|
|
55
|
+
['application/pdf', 'pdf'],
|
|
56
|
+
['application/vnd.openxmlformats-officedocument.wordprocessingml.document', 'docx'],
|
|
57
|
+
['application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', 'xlsx'],
|
|
58
|
+
['application/vnd.openxmlformats-officedocument.presentationml.presentation', 'pptx'],
|
|
59
|
+
['application/json', 'json'],
|
|
60
|
+
['text/csv', 'csv'],
|
|
61
|
+
['text/markdown', 'markdown']
|
|
62
|
+
])
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Pick the extraction format: the filename extension wins, then the media type, then any
|
|
66
|
+
* `text/*` media type as plain text. `undefined` means the file is unsupported.
|
|
67
|
+
*/
|
|
68
|
+
export const fileFormatFor = (input: {
|
|
69
|
+
readonly filename: string
|
|
70
|
+
readonly mediaType: string
|
|
71
|
+
}): ExtractedFileFormat | undefined => {
|
|
72
|
+
const byExtension = formatsByExtension.get(extensionFor(input.filename))
|
|
73
|
+
|
|
74
|
+
if (byExtension !== undefined) return byExtension
|
|
75
|
+
|
|
76
|
+
const byMediaType = formatsByMediaType.get(input.mediaType)
|
|
77
|
+
|
|
78
|
+
if (byMediaType !== undefined) return byMediaType
|
|
79
|
+
|
|
80
|
+
return input.mediaType.startsWith('text/') ? 'text' : undefined
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export const isOfficeFileFormat = (format: ExtractedFileFormat): format is OfficeFileFormat =>
|
|
84
|
+
format === 'docx' || format === 'pptx' || format === 'xlsx'
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
// Runtime-portable root: types, errors, format detection, limits, and the service tag. Parsers
|
|
2
|
+
// live behind `@yolk-sdk/extractors/node`.
|
|
3
|
+
|
|
4
|
+
export {
|
|
5
|
+
FileExtractionError,
|
|
6
|
+
FileExtractionFailureReason,
|
|
7
|
+
minimumSheetJsVersion,
|
|
8
|
+
OfficeArchiveError,
|
|
9
|
+
SheetJsUnavailableError,
|
|
10
|
+
sheetJsInstallCommand,
|
|
11
|
+
UnsupportedFileFormatError
|
|
12
|
+
} from './errors.ts'
|
|
13
|
+
|
|
14
|
+
export type { FileExtractorError } from './errors.ts'
|
|
15
|
+
|
|
16
|
+
export { extractedFileFormats, fileFormatFor, isOfficeFileFormat } from './format.ts'
|
|
17
|
+
|
|
18
|
+
export type {
|
|
19
|
+
ExtractedFile,
|
|
20
|
+
ExtractedFileFormat,
|
|
21
|
+
ExtractedFileMetadata,
|
|
22
|
+
FileInput,
|
|
23
|
+
OfficeFileFormat
|
|
24
|
+
} from './format.ts'
|
|
25
|
+
|
|
26
|
+
export { defaultFileExtractorLimits, FileExtractorLimits } from './limits.ts'
|
|
27
|
+
|
|
28
|
+
export { sanitizeExtractedText } from './sanitize.ts'
|
|
29
|
+
|
|
30
|
+
export { FileExtractor } from './service.ts'
|
|
31
|
+
|
|
32
|
+
export type { FileExtractorApi } from './service.ts'
|
package/src/knowledge.ts
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
import { Effect, Layer, Match, Predicate } from 'effect'
|
|
2
|
+
import type { ExtractedKnowledgeDocument, KnowledgeSource } from '@yolk-sdk/knowledge/documents'
|
|
3
|
+
import { KnowledgeExtractionError } from '@yolk-sdk/knowledge/errors'
|
|
4
|
+
import { KnowledgeExtractor } from '@yolk-sdk/knowledge/extraction'
|
|
5
|
+
import type { KnowledgeExtractorApi, LoadedKnowledgeSource } from '@yolk-sdk/knowledge/extraction'
|
|
6
|
+
import type { ExtractedFile } from './format.ts'
|
|
7
|
+
import { FileExtractor } from './service.ts'
|
|
8
|
+
import type { FileExtractorApi } from './service.ts'
|
|
9
|
+
|
|
10
|
+
/** Percent-decode a path segment; malformed escapes (`%zz`, invalid UTF-8) stay encoded. */
|
|
11
|
+
const decodeSegment = (segment: string) => {
|
|
12
|
+
try {
|
|
13
|
+
return decodeURIComponent(segment)
|
|
14
|
+
} catch {
|
|
15
|
+
return segment
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
const urlFilename = (url: string) => {
|
|
20
|
+
if (!URL.canParse(url)) return url
|
|
21
|
+
|
|
22
|
+
const segments = new URL(url).pathname.split('/').filter(segment => segment.length > 0)
|
|
23
|
+
const last = segments.at(-1)
|
|
24
|
+
|
|
25
|
+
return last === undefined ? url : decodeSegment(last)
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** The filename used for format detection: file name or ref, URL path, or text label. */
|
|
29
|
+
const filenameFor = (source: KnowledgeSource) =>
|
|
30
|
+
Match.value(source).pipe(
|
|
31
|
+
Match.tagsExhaustive({
|
|
32
|
+
File: file => file.name ?? file.ref,
|
|
33
|
+
Url: url => urlFilename(url.url),
|
|
34
|
+
Text: text => text.label ?? 'text'
|
|
35
|
+
})
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
/** Loaded media type first, then the file source's, then `text/plain` for text sources. */
|
|
39
|
+
const mediaTypeFor = (loaded: LoadedKnowledgeSource) => {
|
|
40
|
+
if (loaded.mediaType !== undefined) return loaded.mediaType
|
|
41
|
+
|
|
42
|
+
if (Predicate.isTagged(loaded.source, 'File') && loaded.source.mediaType !== undefined)
|
|
43
|
+
return loaded.source.mediaType
|
|
44
|
+
|
|
45
|
+
return Predicate.isTagged(loaded.source, 'Text') ? 'text/plain' : ''
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
const documentFrom = (
|
|
49
|
+
loaded: LoadedKnowledgeSource,
|
|
50
|
+
extracted: ExtractedFile
|
|
51
|
+
): ExtractedKnowledgeDocument => {
|
|
52
|
+
const { title, ...fileMetadata } = extracted.metadata
|
|
53
|
+
const metadata = { ...fileMetadata, ...loaded.metadata }
|
|
54
|
+
|
|
55
|
+
return title === undefined || title.trim().length === 0
|
|
56
|
+
? { content: extracted.content, metadata }
|
|
57
|
+
: { content: extracted.content, title: title.trim(), metadata }
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* A `KnowledgeExtractor` backed by a `FileExtractor`. String content is already text and passes
|
|
62
|
+
* through unchanged (it must not be blank); bytes are extracted with the format chosen from the
|
|
63
|
+
* source name and media type. File metadata (`format`, `pageCount`, `sheetNames`) is merged
|
|
64
|
+
* under the loaded source's own metadata, and the extracted title becomes the document title.
|
|
65
|
+
*/
|
|
66
|
+
export const makeFileKnowledgeExtractor = (extractor: FileExtractorApi): KnowledgeExtractorApi => ({
|
|
67
|
+
extract: loaded => {
|
|
68
|
+
if (Predicate.isString(loaded.content)) {
|
|
69
|
+
const content = loaded.content
|
|
70
|
+
|
|
71
|
+
return content.trim().length === 0
|
|
72
|
+
? Effect.fail(new KnowledgeExtractionError({ message: 'Knowledge source text is empty' }))
|
|
73
|
+
: Effect.succeed(
|
|
74
|
+
loaded.metadata === undefined ? { content } : { content, metadata: loaded.metadata }
|
|
75
|
+
)
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
return extractor
|
|
79
|
+
.extract({
|
|
80
|
+
filename: filenameFor(loaded.source),
|
|
81
|
+
mediaType: mediaTypeFor(loaded),
|
|
82
|
+
bytes: loaded.content
|
|
83
|
+
})
|
|
84
|
+
.pipe(
|
|
85
|
+
Effect.map(extracted => documentFrom(loaded, extracted)),
|
|
86
|
+
Effect.mapError(
|
|
87
|
+
error => new KnowledgeExtractionError({ message: error.message, cause: error })
|
|
88
|
+
)
|
|
89
|
+
)
|
|
90
|
+
}
|
|
91
|
+
})
|
|
92
|
+
|
|
93
|
+
/** Provide `KnowledgeExtractor` from the `FileExtractor` in context (for example the Node layer). */
|
|
94
|
+
export const FileKnowledgeExtractorLayer = Layer.effect(
|
|
95
|
+
KnowledgeExtractor,
|
|
96
|
+
Effect.gen(function* () {
|
|
97
|
+
const extractor = yield* FileExtractor
|
|
98
|
+
|
|
99
|
+
return KnowledgeExtractor.of(makeFileKnowledgeExtractor(extractor))
|
|
100
|
+
})
|
|
101
|
+
)
|
package/src/limits.ts
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import * as Schema from 'effect/Schema'
|
|
2
|
+
|
|
3
|
+
const PositiveSafeInteger = Schema.Int.pipe(
|
|
4
|
+
Schema.check(Schema.isGreaterThan(0)),
|
|
5
|
+
Schema.check(Schema.isLessThanOrEqualTo(Number.MAX_SAFE_INTEGER))
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Smallest `maxXlsxTextCharacters`: one more than the space always reserved for the
|
|
10
|
+
* `[Some hyperlinks omitted: output limit and hyperlink limit]` marker (61 characters with its
|
|
11
|
+
* leading blank line), so a workbook with links can still produce text.
|
|
12
|
+
*/
|
|
13
|
+
export const minimumXlsxTextCharacters = 62
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Work and output bounds for one `extract` call. Every value is a positive integer, and
|
|
17
|
+
* `maxXlsxTextCharacters` is at least `minimumXlsxTextCharacters`.
|
|
18
|
+
*/
|
|
19
|
+
export const FileExtractorLimits = Schema.Struct({
|
|
20
|
+
/** Input bytes accepted for any format. */
|
|
21
|
+
maxInputBytes: PositiveSafeInteger,
|
|
22
|
+
/** Entries in a DOCX, XLSX, or PPTX ZIP archive. */
|
|
23
|
+
maxArchiveEntries: PositiveSafeInteger,
|
|
24
|
+
/** Total inflated bytes of a DOCX, XLSX, or PPTX archive, counted while inflating. */
|
|
25
|
+
maxExpandedBytes: PositiveSafeInteger,
|
|
26
|
+
/** Worksheets in a workbook. */
|
|
27
|
+
maxXlsxSheets: PositiveSafeInteger,
|
|
28
|
+
/** Cells visited across all worksheet ranges (absent cells count too). */
|
|
29
|
+
maxXlsxCellVisits: PositiveSafeInteger,
|
|
30
|
+
/** Characters of XLSX text, including hyperlink annotations and the omitted-links marker. */
|
|
31
|
+
maxXlsxTextCharacters: PositiveSafeInteger.pipe(
|
|
32
|
+
Schema.check(Schema.isGreaterThanOrEqualTo(minimumXlsxTextCharacters))
|
|
33
|
+
),
|
|
34
|
+
/** Hyperlinks read per workbook; later ones are ignored (still removed before SheetJS). */
|
|
35
|
+
maxXlsxHyperlinks: PositiveSafeInteger
|
|
36
|
+
})
|
|
37
|
+
|
|
38
|
+
export type FileExtractorLimits = typeof FileExtractorLimits.Type
|
|
39
|
+
|
|
40
|
+
/** Default limits; override any of them through `makeFileExtractorLayer({ limits })`. */
|
|
41
|
+
export const defaultFileExtractorLimits: FileExtractorLimits = {
|
|
42
|
+
maxInputBytes: 50 * 1024 * 1024,
|
|
43
|
+
maxArchiveEntries: 10_000,
|
|
44
|
+
maxExpandedBytes: 50 * 1024 * 1024,
|
|
45
|
+
maxXlsxSheets: 100,
|
|
46
|
+
maxXlsxCellVisits: 100_000,
|
|
47
|
+
maxXlsxTextCharacters: 512 * 1024,
|
|
48
|
+
maxXlsxHyperlinks: 10_000
|
|
49
|
+
}
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
import { Buffer } from 'node:buffer'
|
|
2
|
+
import { Effect, Option, Predicate } from 'effect'
|
|
3
|
+
import { FileExtractionError } from '../errors.ts'
|
|
4
|
+
import type { FileExtractorError } from '../errors.ts'
|
|
5
|
+
import type {
|
|
6
|
+
ExtractedFile,
|
|
7
|
+
ExtractedFileFormat,
|
|
8
|
+
ExtractedFileMetadata,
|
|
9
|
+
FileInput,
|
|
10
|
+
OfficeFileFormat
|
|
11
|
+
} from '../format.ts'
|
|
12
|
+
import type { FileExtractorLimits } from '../limits.ts'
|
|
13
|
+
import { sanitizeExtractedText } from '../sanitize.ts'
|
|
14
|
+
import { readOfficeArchive, storedArchive } from './office-archive.ts'
|
|
15
|
+
import type { NormalizedOfficeArchive } from './office-archive.ts'
|
|
16
|
+
import { extractPptxText } from './pptx-text.ts'
|
|
17
|
+
import { asXlsxWorkbook, loadSheetJs } from './sheetjs.ts'
|
|
18
|
+
import type { SheetJsLoader } from './sheetjs.ts'
|
|
19
|
+
import { resolveXlsxHyperlinks } from './xlsx-hyperlinks.ts'
|
|
20
|
+
import { buildSheetJsInput } from './xlsx-sheetjs-input.ts'
|
|
21
|
+
import { extractBoundedXlsxText } from './xlsx-text.ts'
|
|
22
|
+
|
|
23
|
+
/** Formats whose bytes go through a parser (and, by default, an isolated worker). */
|
|
24
|
+
export type ParsedFileFormat = 'pdf' | OfficeFileFormat
|
|
25
|
+
|
|
26
|
+
export const isParsedFileFormat = (format: ExtractedFileFormat): format is ParsedFileFormat =>
|
|
27
|
+
format === 'pdf' || format === 'docx' || format === 'pptx' || format === 'xlsx'
|
|
28
|
+
|
|
29
|
+
/** Sanitize extracted text; empty results fail. */
|
|
30
|
+
export const makeExtractedFile = (content: string, metadata: ExtractedFileMetadata) => {
|
|
31
|
+
const sanitized = sanitizeExtractedText(content)
|
|
32
|
+
|
|
33
|
+
if (sanitized.length === 0)
|
|
34
|
+
return Effect.fail(
|
|
35
|
+
new FileExtractionError({
|
|
36
|
+
message: 'Extracted file content is empty',
|
|
37
|
+
format: metadata.format
|
|
38
|
+
})
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
return Effect.succeed<ExtractedFile>({ content: sanitized, metadata })
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* PDF.js 6 releases a document through its loading task (`PDFDocumentProxy` no longer has
|
|
46
|
+
* `destroy()`); unpdf releases the documents it opens the same way.
|
|
47
|
+
*/
|
|
48
|
+
type PdfDocumentResource = {
|
|
49
|
+
readonly loadingTask: { readonly destroy: () => Promise<void> }
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Run `use` with an opened PDF document and always release the parser afterwards. */
|
|
53
|
+
export const withAcquiredPdfDocument = <D extends PdfDocumentResource, A, E, R>(
|
|
54
|
+
open: Effect.Effect<D, E, R>,
|
|
55
|
+
use: (document: D) => Effect.Effect<A, E, R>,
|
|
56
|
+
format: ExtractedFileFormat
|
|
57
|
+
) =>
|
|
58
|
+
Effect.scoped(
|
|
59
|
+
Effect.gen(function* () {
|
|
60
|
+
const document = yield* Effect.acquireRelease(open, document =>
|
|
61
|
+
Effect.tryPromise({
|
|
62
|
+
try: () => document.loadingTask.destroy(),
|
|
63
|
+
catch: () => new FileExtractionError({ message: 'Could not release PDF parser', format })
|
|
64
|
+
}).pipe(Effect.ignore)
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
return yield* use(document)
|
|
68
|
+
})
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
const extractPdf = (input: FileInput) =>
|
|
72
|
+
Effect.gen(function* () {
|
|
73
|
+
// Loaded on first use, so a worker extracting another format never loads PDF.js.
|
|
74
|
+
const { extractText, getDocumentProxy, getMeta } = yield* Effect.tryPromise({
|
|
75
|
+
try: () => import('unpdf'),
|
|
76
|
+
catch: cause =>
|
|
77
|
+
new FileExtractionError({ message: 'Could not read PDF', format: 'pdf', cause })
|
|
78
|
+
})
|
|
79
|
+
|
|
80
|
+
return yield* withAcquiredPdfDocument(
|
|
81
|
+
Effect.tryPromise({
|
|
82
|
+
// PDF.js may detach the buffer it is given; never hand it the caller's bytes.
|
|
83
|
+
try: () => getDocumentProxy(new Uint8Array(input.bytes)),
|
|
84
|
+
catch: cause =>
|
|
85
|
+
new FileExtractionError({ message: 'Could not read PDF', format: 'pdf', cause })
|
|
86
|
+
}),
|
|
87
|
+
document =>
|
|
88
|
+
Effect.gen(function* () {
|
|
89
|
+
const extracted = yield* Effect.tryPromise({
|
|
90
|
+
try: () => extractText(document, { mergePages: true }),
|
|
91
|
+
catch: cause =>
|
|
92
|
+
new FileExtractionError({
|
|
93
|
+
message: 'Could not extract PDF text',
|
|
94
|
+
format: 'pdf',
|
|
95
|
+
cause
|
|
96
|
+
})
|
|
97
|
+
})
|
|
98
|
+
|
|
99
|
+
const meta = yield* Effect.tryPromise({
|
|
100
|
+
try: () => getMeta(document),
|
|
101
|
+
catch: cause =>
|
|
102
|
+
new FileExtractionError({
|
|
103
|
+
message: 'Could not read PDF metadata',
|
|
104
|
+
format: 'pdf',
|
|
105
|
+
cause
|
|
106
|
+
})
|
|
107
|
+
}).pipe(Effect.option)
|
|
108
|
+
|
|
109
|
+
const rawTitle = Option.isSome(meta) ? meta.value.info.Title : undefined
|
|
110
|
+
const title = Predicate.isString(rawTitle) && rawTitle.length > 0 ? rawTitle : undefined
|
|
111
|
+
|
|
112
|
+
const metadata: ExtractedFileMetadata =
|
|
113
|
+
title === undefined
|
|
114
|
+
? { format: 'pdf', pageCount: extracted.totalPages }
|
|
115
|
+
: { format: 'pdf', title, pageCount: extracted.totalPages }
|
|
116
|
+
|
|
117
|
+
return yield* makeExtractedFile(extracted.text, metadata)
|
|
118
|
+
}),
|
|
119
|
+
'pdf'
|
|
120
|
+
)
|
|
121
|
+
})
|
|
122
|
+
|
|
123
|
+
const validatedArchive = (
|
|
124
|
+
input: FileInput,
|
|
125
|
+
format: OfficeFileFormat,
|
|
126
|
+
limits: FileExtractorLimits
|
|
127
|
+
): Effect.Effect<NormalizedOfficeArchive, FileExtractionError> =>
|
|
128
|
+
readOfficeArchive(
|
|
129
|
+
input.bytes,
|
|
130
|
+
format,
|
|
131
|
+
limits,
|
|
132
|
+
// One extra tag detects a workbook over the cap.
|
|
133
|
+
format === 'xlsx' ? { maxHyperlinkTags: limits.maxXlsxHyperlinks + 1 } : {}
|
|
134
|
+
).pipe(
|
|
135
|
+
Effect.mapError(
|
|
136
|
+
error => new FileExtractionError({ message: error.message, format, cause: error })
|
|
137
|
+
)
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
const extractDocx = (input: FileInput, limits: FileExtractorLimits) =>
|
|
141
|
+
Effect.gen(function* () {
|
|
142
|
+
const { parts } = yield* validatedArchive(input, 'docx', limits)
|
|
143
|
+
|
|
144
|
+
const result = yield* Effect.tryPromise({
|
|
145
|
+
try: async () => {
|
|
146
|
+
const { default: mammoth } = await import('mammoth')
|
|
147
|
+
|
|
148
|
+
return await mammoth.extractRawText({ buffer: Buffer.from(storedArchive(parts)) })
|
|
149
|
+
},
|
|
150
|
+
catch: cause =>
|
|
151
|
+
new FileExtractionError({ message: 'Could not extract DOCX text', format: 'docx', cause })
|
|
152
|
+
})
|
|
153
|
+
|
|
154
|
+
return yield* makeExtractedFile(result.value, { format: 'docx', title: input.filename })
|
|
155
|
+
})
|
|
156
|
+
|
|
157
|
+
/** Keep the first `max` tags in archive order. */
|
|
158
|
+
const capHyperlinkTags = (
|
|
159
|
+
tagsByPart: ReadonlyMap<string, ReadonlyArray<string>>,
|
|
160
|
+
max: number
|
|
161
|
+
): ReadonlyMap<string, ReadonlyArray<string>> => {
|
|
162
|
+
const capped = new Map<string, ReadonlyArray<string>>()
|
|
163
|
+
let remaining = max
|
|
164
|
+
|
|
165
|
+
for (const [part, tags] of tagsByPart) {
|
|
166
|
+
if (remaining <= 0) break
|
|
167
|
+
|
|
168
|
+
const kept = tags.slice(0, remaining)
|
|
169
|
+
remaining -= kept.length
|
|
170
|
+
capped.set(part, kept)
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
return capped
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
const extractXlsx = (input: FileInput, limits: FileExtractorLimits, loader: SheetJsLoader) =>
|
|
177
|
+
Effect.gen(function* () {
|
|
178
|
+
const normalized = yield* validatedArchive(input, 'xlsx', limits)
|
|
179
|
+
|
|
180
|
+
// SheetJS never sees the uploaded archive: only this allowlisted rebuild of its parts.
|
|
181
|
+
const sheetJsInput = yield* Effect.try({
|
|
182
|
+
try: () => buildSheetJsInput(normalized.parts, limits.maxXlsxSheets),
|
|
183
|
+
catch: cause =>
|
|
184
|
+
cause instanceof FileExtractionError
|
|
185
|
+
? cause
|
|
186
|
+
: new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx', cause })
|
|
187
|
+
})
|
|
188
|
+
|
|
189
|
+
const sheetJs = yield* loadSheetJs(loader)
|
|
190
|
+
|
|
191
|
+
const parsed = yield* Effect.try({
|
|
192
|
+
try: () => sheetJs.read(sheetJsInput.archive),
|
|
193
|
+
catch: cause =>
|
|
194
|
+
new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx', cause })
|
|
195
|
+
})
|
|
196
|
+
|
|
197
|
+
const workbook = asXlsxWorkbook(parsed)
|
|
198
|
+
|
|
199
|
+
if (workbook === undefined)
|
|
200
|
+
return yield* Effect.fail(
|
|
201
|
+
new FileExtractionError({ message: 'Could not read XLSX', format: 'xlsx' })
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
// One extra tag was captured to detect that the workbook exceeds the hyperlink cap.
|
|
205
|
+
const capturedTags = [...normalized.hyperlinkTags.values()].reduce(
|
|
206
|
+
(total, tags) => total + tags.length,
|
|
207
|
+
0
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
const hyperlinksTruncated = capturedTags > limits.maxXlsxHyperlinks
|
|
211
|
+
|
|
212
|
+
const hyperlinks = resolveXlsxHyperlinks(
|
|
213
|
+
normalized.parts,
|
|
214
|
+
hyperlinksTruncated
|
|
215
|
+
? capHyperlinkTags(normalized.hyperlinkTags, limits.maxXlsxHyperlinks)
|
|
216
|
+
: normalized.hyperlinkTags,
|
|
217
|
+
sheetJsInput.sheets
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
const content = yield* Effect.try({
|
|
221
|
+
try: () => extractBoundedXlsxText(workbook, limits, { hyperlinks, hyperlinksTruncated }),
|
|
222
|
+
catch: cause =>
|
|
223
|
+
cause instanceof FileExtractionError
|
|
224
|
+
? cause
|
|
225
|
+
: new FileExtractionError({
|
|
226
|
+
message: 'Could not extract XLSX text',
|
|
227
|
+
format: 'xlsx',
|
|
228
|
+
cause
|
|
229
|
+
})
|
|
230
|
+
})
|
|
231
|
+
|
|
232
|
+
return yield* makeExtractedFile(content, {
|
|
233
|
+
format: 'xlsx',
|
|
234
|
+
title: sheetJsInput.title ?? input.filename,
|
|
235
|
+
sheetNames: workbook.SheetNames
|
|
236
|
+
})
|
|
237
|
+
})
|
|
238
|
+
|
|
239
|
+
const extractPptx = (input: FileInput, limits: FileExtractorLimits) =>
|
|
240
|
+
Effect.gen(function* () {
|
|
241
|
+
const { parts } = yield* validatedArchive(input, 'pptx', limits)
|
|
242
|
+
|
|
243
|
+
const text = yield* Effect.try({
|
|
244
|
+
try: () => extractPptxText(parts),
|
|
245
|
+
catch: cause =>
|
|
246
|
+
new FileExtractionError({ message: 'Could not extract PPTX text', format: 'pptx', cause })
|
|
247
|
+
})
|
|
248
|
+
|
|
249
|
+
return yield* makeExtractedFile(text, { format: 'pptx', title: input.filename })
|
|
250
|
+
})
|
|
251
|
+
|
|
252
|
+
/**
|
|
253
|
+
* Extract a PDF, DOCX, XLSX, or PPTX file in the current thread. The Node layer runs this inside
|
|
254
|
+
* an isolated worker unless `isolation: 'none'` is configured.
|
|
255
|
+
*/
|
|
256
|
+
export const extractParsedFile = (
|
|
257
|
+
input: FileInput,
|
|
258
|
+
format: ParsedFileFormat,
|
|
259
|
+
limits: FileExtractorLimits,
|
|
260
|
+
loader: SheetJsLoader
|
|
261
|
+
): Effect.Effect<ExtractedFile, FileExtractorError> => {
|
|
262
|
+
if (format === 'pdf') return extractPdf(input)
|
|
263
|
+
|
|
264
|
+
if (format === 'docx') return extractDocx(input, limits)
|
|
265
|
+
|
|
266
|
+
if (format === 'xlsx') return extractXlsx(input, limits, loader)
|
|
267
|
+
|
|
268
|
+
return extractPptx(input, limits)
|
|
269
|
+
}
|