@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Yolk SDK contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,318 @@
1
+ # @yolk-sdk/extractors
2
+
3
+ Bounded text extraction for PDF, DOCX, XLSX, PPTX, CSV, JSON, Markdown, and plain-text files: an
4
+ Effect `FileExtractor` service with Office archive validation, hyperlink-safe XLSX text, and a
5
+ `KnowledgeExtractor` adapter for `@yolk-sdk/knowledge`.
6
+
7
+ ## Install
8
+
9
+ ```bash
10
+ pnpm add @yolk-sdk/extractors@canary effect@4.0.0
11
+ # Only if you extract .xlsx files: SheetJS from the SheetJS CDN, not npm.
12
+ pnpm add https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz
13
+ ```
14
+
15
+ Canary APIs are unstable. Keep all `@yolk-sdk/*` packages on the same version.
16
+ Use the SDK's matching Effect version (`4.0.0`) in host code.
17
+ Requires Node.js 22+. `@yolk-sdk/extractors/node` is server-only.
18
+
19
+ ### Why SheetJS comes from its CDN
20
+
21
+ SheetJS (`xlsx`) is an optional peer dependency (`>=0.20.3`), loaded with a dynamic import only
22
+ when an XLSX file is extracted. The `xlsx` package on npm is unmaintained at 0.18.5, which has
23
+ CVE-2023-30533 (prototype pollution from a crafted file, fixed in 0.19.3) and CVE-2024-22363
24
+ (regular-expression denial of service, fixed in 0.20.2). Fixed releases are published only as
25
+ tarballs on `cdn.sheetjs.com`. A published package must not depend on a URL, so the host installs
26
+ the tarball itself.
27
+
28
+ pnpm records the tarball URL and its integrity hash in the lockfile. A direct tarball dependency
29
+ needs no extra pnpm 11 settings. `blockExoticSubdeps` (default `true`) rejects tarball and git
30
+ dependencies only below the top level. `minimumReleaseAge` (default one day) checks registry
31
+ publish times, and a URL tarball has none. Because the peer is optional, SheetJS cannot arrive
32
+ transitively: each host adds it directly.
33
+
34
+ If SheetJS is missing, is not a SheetJS build, or is older than 0.20.3 (for example a leftover npm
35
+ `xlsx@0.18.5`, or a 0.20.3 prerelease), XLSX extraction fails with `SheetJsUnavailableError`
36
+ (`reason: 'missing' | 'invalid' | 'outdated'`). A `version` that is not strict SemVer counts as
37
+ `invalid`. Its message includes the install command. Other formats never load SheetJS.
38
+
39
+ ## Subpaths
40
+
41
+ | Subpath | Purpose |
42
+ | --------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
43
+ | `@yolk-sdk/extractors` | Runtime-portable contract: `FileInput`, `ExtractedFile`, formats, errors, `fileFormatFor`, `defaultFileExtractorLimits`, `sanitizeExtractedText`, and the `FileExtractor` service tag. No parser imports. |
44
+ | `@yolk-sdk/extractors/node` | `FileExtractorLayer` and `makeFileExtractorLayer(options)`: the Node implementation (unpdf, mammoth, SheetJS, fflate, `node:zlib`). Also `defaultWorkerIsolation`, the `FileExtractorIsolation` and `WorkerIsolationOptions` types, and `normalizeOfficeArchive` (see below). |
45
+ | `@yolk-sdk/extractors/node/extraction-worker` | The worker entry the Node layer starts per extraction (not imported directly; see Worker isolation). |
46
+ | `@yolk-sdk/extractors/knowledge` | `FileKnowledgeExtractorLayer`: a `@yolk-sdk/knowledge/extraction` `KnowledgeExtractor` backed by the `FileExtractor` in context, and `makeFileKnowledgeExtractor`. |
47
+
48
+ ## Example
49
+
50
+ ```ts
51
+ import { Effect } from 'effect'
52
+ import { FileExtractor } from '@yolk-sdk/extractors'
53
+ import { makeFileExtractorLayer } from '@yolk-sdk/extractors/node'
54
+
55
+ const FileExtractorLive = makeFileExtractorLayer({ limits: { maxInputBytes: 10 * 1024 * 1024 } })
56
+
57
+ const program = Effect.gen(function* () {
58
+ const extractor = yield* FileExtractor
59
+
60
+ return yield* extractor.extract({ filename: 'report.xlsx', mediaType: '', bytes })
61
+ }).pipe(Effect.provide(FileExtractorLive))
62
+ ```
63
+
64
+ `bytes` is a host-owned `Uint8Array`. `FileExtractorLayer` is the same layer with default limits.
65
+ The extractor copies the input before a parser reads it, so the caller's buffer is never
66
+ detached.
67
+
68
+ ### Worker isolation
69
+
70
+ PDF, DOCX, XLSX, and PPTX files are parsed in a fresh Node `worker_threads` worker per
71
+ extraction. Only the extracted text and metadata come back. Text formats are decoded in the
72
+ calling thread; no parser reads them.
73
+
74
+ ```ts
75
+ const FileExtractorLive = makeFileExtractorLayer({
76
+ isolation: { maxOldGenerationSizeMb: 512, timeoutMs: 60_000 }
77
+ })
78
+ ```
79
+
80
+ | `isolation` option | Default | Bounds |
81
+ | -------------------------- | ------------------------ | ------------------------------------------------------------------------- |
82
+ | `maxOldGenerationSizeMb` | 256 | V8 old-generation heap of each worker |
83
+ | `maxYoungGenerationSizeMb` | 32 | V8 young-generation heap of each worker |
84
+ | `stackSizeMb` | 4 | Stack of each worker (Node's default) |
85
+ | `timeoutMs` | 30,000 | Wall-clock time per worker (a real timer, not Effect's clock; ≤ 2³¹−1 ms) |
86
+ | `maxConcurrentWorkers` | 4 | This layer's share of the realm-wide pool of 4 workers (1 to 4) |
87
+ | `maxQueueWaitMs` | the layer's `timeoutMs` | Wait for a worker slot (≤ 2³¹−1 ms); then it fails with `busy` |
88
+ | `workerUrl` | the package's own worker | The worker entry, for hosts that bundle or relocate the package |
89
+
90
+ **Cap per JavaScript realm.** Every layer in a JavaScript realm (the main thread, or each worker
91
+ thread that builds the layer) shares one pool of 4 workers, however often the layer is built (a
92
+ per-request `Effect.provide` builds a new one each time). The pool is kept on `globalThis`, so
93
+ duplicated copies of the package in that realm share it too. A host that runs its request
94
+ handlers in a pool of worker threads gets 4 workers per thread. A layer's
95
+ `maxConcurrentWorkers` can only lower its own share. Slots are handed out first come, first
96
+ served. An extraction waits for a slot at most `maxQueueWaitMs`, then fails with
97
+ `reason: 'busy'` without starting a worker; a slot freed after that deadline never admits it.
98
+ A freed slot is handed to the next waiter, which resumes on a later microtask, so a long queue
99
+ of extractions that fail at once drains without deepening the stack.
100
+
101
+ **What the worker bounds.** Each worker's V8 heap, stack, and running time. When a worker runs
102
+ out of heap, times out, cannot start, or exits without a result, it is terminated and the
103
+ extraction fails with `FileExtractionError` and `reason` set to `resource-limit`, `timeout`,
104
+ `worker-unavailable`, or `worker-failed`. A worker that cannot start never falls back to parsing
105
+ in-process. Four workers at the default heap add up to about 1 GB of V8 heap.
106
+
107
+ **What it does not bound.** Total process memory (RSS) and memory outside the V8 heap: Buffers,
108
+ inflated archive parts, and native allocations. Those are limited only by the 50 MiB input and
109
+ Office expanded-size limits, per running worker. PDF.js's decoded images, fonts, and streams have
110
+ no separate cap in this package, which is a residual risk for crafted PDFs. Your host's memory
111
+ limit (container or serverless function) is the outer bound: size it for four workers plus your
112
+ own load, or lower the limits.
113
+
114
+ Legitimate files at the default limits parse in well under a second, so 30 s only stops runaway
115
+ work.
116
+
117
+ `isolation: 'none'` parses in the calling thread, for runtimes without worker threads. It is
118
+ **unsafe for untrusted input**: a crafted file can then exhaust the process heap or block the
119
+ event loop. It is also the only mode that accepts a `loadSheetJs` function (tests use it to
120
+ observe SheetJS). A worker always imports the installed `xlsx` itself, so passing `loadSheetJs`
121
+ with worker isolation is a defect when the layer is built.
122
+
123
+ Knowledge ingestion:
124
+
125
+ ```ts
126
+ import { Layer } from 'effect'
127
+ import { FileKnowledgeExtractorLayer } from '@yolk-sdk/extractors/knowledge'
128
+ import { FileExtractorLayer } from '@yolk-sdk/extractors/node'
129
+
130
+ const KnowledgeExtractorLive = FileKnowledgeExtractorLayer.pipe(Layer.provide(FileExtractorLayer))
131
+ ```
132
+
133
+ The adapter passes string `content` through unchanged (blank text fails). Bytes are extracted with
134
+ the format chosen from the source's name (`File` name or ref, `Url` path, `Text` label) and media
135
+ type (`LoadedKnowledgeSource.mediaType`, then the `File` source's, then `text/plain` for `Text`
136
+ sources). The extracted title becomes the document title. `format`, `pageCount`, and `sheetNames`
137
+ merge under the source's own metadata, so host keys win. Failures become
138
+ `KnowledgeExtractionError` with the original error as `cause`.
139
+
140
+ ## Output
141
+
142
+ | Format | Text | Metadata |
143
+ | --------------------------------- | --------------------------------------------------------------------------------------- | ---------------------------------------- |
144
+ | `text`, `markdown`, `csv`, `json` | UTF-8 decoded | `title` = filename |
145
+ | `pdf` | Text of all pages (unpdf) | PDF `Title` if set, `pageCount` |
146
+ | `docx` | Raw text (mammoth) | `title` = filename |
147
+ | `xlsx` | One `# <sheet>` section per sheet with bounded CSV rows; external links as `text <url>` | workbook title or filename, `sheetNames` |
148
+ | `pptx` | Slide text in slide order, then speaker notes | `title` = filename |
149
+
150
+ XLSX cells show their cached value as Excel displays it: SheetJS applies the number formats, and
151
+ format codes longer than Excel's own 255-character limit show as General. Formulas are not
152
+ parsed: a formula cell shows its cached result, and a formula cell without a cached value is
153
+ empty, never `=formula`. Excel, LibreOffice, and Google Sheets always store cached values; files
154
+ generated by code (openpyxl, ExcelJS, the SheetJS writer) may not. A worksheet is read when its
155
+ workbook relationship points at an `.xml` part; chart sheets and parts under another extension
156
+ are listed in `sheetNames` but have no rows. Comments, drawings, and other parts SheetJS does not
157
+ need for cell text are never parsed.
158
+
159
+ All text is sanitized: line endings are normalized, control characters dropped, and runs of
160
+ spaces, dots, and blank lines collapsed. Empty results fail with `FileExtractionError`. The format
161
+ comes from the filename extension first, then the media type, then any `text/*` media type as
162
+ plain text. Everything else fails with `UnsupportedFileFormatError`.
163
+
164
+ ## Limits
165
+
166
+ `defaultFileExtractorLimits` (conservative defaults); override any of them
167
+ through `makeFileExtractorLayer({ limits })`. Invalid values are a defect when the layer is built.
168
+
169
+ | Limit | Default | Bounds |
170
+ | ----------------------- | ------- | ------------------------------------------------------------------------- |
171
+ | `maxInputBytes` | 50 MiB | Input size, every format |
172
+ | `maxArchiveEntries` | 10,000 | ZIP entries in DOCX, XLSX, PPTX |
173
+ | `maxExpandedBytes` | 50 MiB | Inflated bytes of an Office archive, counted while inflating |
174
+ | `maxXlsxSheets` | 100 | Worksheets |
175
+ | `maxXlsxCellVisits` | 100,000 | Cells inside every sheet's declared range, absent cells included |
176
+ | `maxXlsxTextCharacters` | 512 Ki | XLSX text, annotations and the omission marker included; at least 62 |
177
+ | `maxXlsxHyperlinks` | 10,000 | Hyperlinks read per workbook. Extra links are dropped but still stripped. |
178
+
179
+ Exceeding a limit fails the extraction with `FileExtractionError`. The exceptions are hyperlink
180
+ annotations and links beyond the cap, described below.
181
+
182
+ ## Security model
183
+
184
+ - **Office archives** (DOCX, XLSX, PPTX) are validated before any parser sees them. The ZIP
185
+ directory is read only as an index and its sizes are never trusted. Each entry is inflated in
186
+ bounded chunks, and the real output is counted against its declared size and
187
+ `maxExpandedBytes`. Encrypted, ZIP64, split, and duplicate-name archives fail (names that differ
188
+ only in case count as duplicates, as in OPC), as do archives with ambiguous paths (`..`, `.`,
189
+ `//`, absolute, backslashes, control characters) or macro projects (`vbaProject.bin`). Directory
190
+ entries ending in `/` are fine. Parsers then get archives rebuilt from the validated bytes. OOXML
191
+ input missing `[Content_Types].xml` or its main part (`word/document.xml`, `xl/workbook.xml`,
192
+ `ppt/presentation.xml`) fails; this is deliberately strict.
193
+ - **Worker isolation.** Every parser runs in a worker with V8 heap, stack, and time limits and a
194
+ cap of 4 workers per JavaScript realm (see above). A parser path nobody has found yet that
195
+ exhausts the worker's heap or runs too long ends in a typed error. Memory outside the V8 heap is
196
+ bounded only by the input and expanded-size limits and your host's memory limit.
197
+ - **Generated SheetJS input.** SheetJS picks its parser from the archive, not the file name, and
198
+ several of its XML parsers do work far beyond the size of the part. So SheetJS never receives
199
+ the uploaded archive. The extractor builds a new one:
200
+ - generated: `[Content_Types].xml`, `_rels/.rels`, and `xl/_rels/workbook.xml.rels`, so no
201
+ attacker content type, part name, or relationship reaches SheetJS;
202
+ - generated `xl/workbook.xml`: the sheets in order (name, `sheetId`, hidden state, a fresh
203
+ relationship id each) and the 1904 date system, nothing else. SheetJS's defined-name handling
204
+ is quadratic, and with shared relationship ids it parses one worksheet once per sheet. A
205
+ workbook whose sheets the extractor and SheetJS's own tag grammar count or name differently,
206
+ whose names repeat ignoring case, or whose sheets share a worksheet part is rejected;
207
+ - generated `xl/styles.xml`: number formats (at most 255 characters after unescaping, at most
208
+ 1,000) and one cell format per source cell format, in order (at most 64,000). SheetJS
209
+ re-parses a cell's format for every cell, so one huge format used by many cells would
210
+ otherwise amplify;
211
+ - copied: the worksheets (any `.xml` part related as a worksheet, stored as
212
+ `xl/worksheets/sheet<n>.xml`) and `xl/sharedStrings.xml`, validated and hyperlink-stripped.
213
+ Parts where SheetJS could meet a CDATA marker are rejected: SheetJS's unescaping recurses
214
+ over an unterminated CDATA section with quadratic output, and Excel never writes CDATA there.
215
+ SheetJS reaches that unescaping in two ways the check models. It decodes: up to two of
216
+ `unescapexml` and `utf8read` (which keeps only each character's low byte, so `&#x13C;` in a
217
+ `str` cell becomes `<` between its two decodes). And it removes tags before decoding: every
218
+ `<si>` in the shared-strings table, and every `<r>` in rich text, so `A<<r>![CDATA[B` becomes
219
+ `A<![CDATA[B`. Inline strings take the rich-text path even with `cellHTML: false`. So a part is
220
+ rejected when its text, or its text after `utf8read`, contains `<<` or `<!`, or when its text,
221
+ as is or with every simple opening tag removed, meets the marker after any chain of up to two
222
+ conversions. This deliberately fails closed: XML comments, `<!DOCTYPE` and any other `<!…`
223
+ declaration, and a literal `<<` in a worksheet or the shared strings are rejected as
224
+ unsupported markup (CDATA, comments or declarations). Excel, LibreOffice, and Google Sheets
225
+ never write them there. Some forms are rejected only because they fall in this superset, not
226
+ because SheetJS 0.20.3 would expand them (for example `&lt;<r>![CDATA[`, which it decodes
227
+ after its CDATA check).
228
+
229
+ The title comes from `docProps/core.xml`, read by the extractor. With no marker entries and no
230
+ `.bin` entries, SheetJS can only take its XLSX path. Everything else (comments, VML, drawings,
231
+ `.bin` parts, external links, pivot caches, `customXml`, …) is left out.
232
+
233
+ - **Early rejection.** For a clear error instead of "Could not read XLSX", XLSX input that SheetJS
234
+ would route elsewhere fails before SheetJS loads: ODS and Numbers marker entries
235
+ (`META-INF/manifest.xml`, `objectdata.xml`, `Index/Document.iwa`, `Index.zip`, any `Root Entry/`
236
+ name), matched after SheetJS's own path normalisation (first `//` collapsed, `Root Entry/`
237
+ stripped, `\` as `/`, any case); XLSB content types in `[Content_Types].xml` overrides; and
238
+ relationship targets or override part names ending in `.bin`, unless they are types SheetJS
239
+ never parses (printer settings, OLE and ActiveX binaries, custom properties, attached toolbars).
240
+ Tags and attributes are read exactly as SheetJS reads them (quoted values may contain `<`;
241
+ attribute names are case-sensitive). `.bin` elsewhere in a name (`data.bin.xml`, an
242
+ `archive.bin/` directory, `Id="rId.bin"`) is fine.
243
+ - **Formulas off.** SheetJS runs with `cellFormula: false` (and without HTML, styles, stubs, or
244
+ VBA), so shared formulas are never copied onto every dependent cell and array formulas are never
245
+ rescanned per cell. Only cached values are read (see Output).
246
+ - **`normalizeOfficeArchive`** returns a validated, rebuilt archive of every part for storage or
247
+ for other parsers. It applies the same archive checks and, for XLSX, the hyperlink strip and the
248
+ early rejections, but it keeps parts SheetJS must never see (printer settings, comments, …).
249
+ Never run SheetJS on its output directly: extract XLSX text through `FileExtractor`, which builds
250
+ the allowlisted input.
251
+ - **XLSX hyperlinks.** Before parsing, SheetJS (0.18.5 and 0.20.3) expands every
252
+ `<hyperlink ref>` range into per-cell objects. A 6 KB file with `ref="A1:XFD1048576"`
253
+ exhausts a 1 GB heap. So the extractor reads each `<hyperlink>` itself (`ref`, `r:id` to the
254
+ worksheet relationship target, `location`, `display`), up to `maxXlsxHyperlinks` per workbook. It
255
+ removes every hyperlink tag from every archive part, replacing each with a space so fragments
256
+ cannot join into a new tag. SheetJS never sees one. In the text, every existing cell of the
257
+ visited range that falls inside a link is written as `text <url>`. An indexed lookup over the
258
+ visited range keeps this at O((links + cells) · log) instead of a scan of every link per cell.
259
+ Only `http:`, `https:`, and `mailto:` targets are shown (normalized; links whose target is
260
+ longer than 2,048 characters are dropped). `display` labels are cut to 1,024 characters ending
261
+ in `…`, and an annotation that cannot fit the remaining budget is rejected by its length before
262
+ it is built. Internal `#Sheet!A1` locations and other schemes are omitted. Plain text is
263
+ rendered first and links use only the budget left over. When an annotation does not fit, or
264
+ links exceed the cap, the cell is written plain and the output ends with one
265
+ `[Some hyperlinks omitted: output limit]` (or `hyperlink limit`) marker, inside the budget.
266
+ - **UTF-16 parts.** SheetJS decodes BOM-marked UTF-16 parts itself, and the byte-level strip
267
+ cannot see tags inside them. Any XLSX part whose stripped bytes would decode as UTF-16 with a
268
+ hyperlink tag is rejected. Excel never writes UTF-16 parts. The markup rule can also reject a
269
+ legitimate BOM-marked UTF-16 part, when a character's low byte forms `<<` or `<!`.
270
+ - **Bounded XLSX text.** Every sheet range is checked strictly against Excel's grid, and the
271
+ cell-visit total is checked before any cell is read. CSV is generated incrementally against
272
+ the character budget (never `sheet_to_csv`). Cells and sheets are read as own properties only.
273
+ The renderer uses SheetJS's display text. SheetJS evaluates number formats inside `read`, before
274
+ this budget, which is why it only ever reads the generated, bounded stylesheet.
275
+ - **SheetJS version.** Only SheetJS 0.20.3+ is used (see above). The prototype-pollution
276
+ regression (CVE-2023-30533) is covered by a test with a crafted comment: the comment never
277
+ reaches SheetJS, and SheetJS 0.20.3 parsing the file directly is not polluted either.
278
+
279
+ - **Residual risk.** SheetJS still parses the worksheets and shared strings as uploaded (after
280
+ validation, the hyperlink strip, and the CDATA check, which models SheetJS 0.20.3's decodes and
281
+ tag removals), and unpdf, mammoth, and the PPTX reader
282
+ parse their parts. No other super-linear path is known in these parsers with the options used,
283
+ but any one found later runs before the extractor's budgets. The worker's heap, stack, and time
284
+ limits are the backstop for that, and they do not bound memory outside the V8 heap (PDF.js's
285
+ decoded buffers included). With `isolation: 'none'` there is no backstop: a number format of up
286
+ to 255 tokens still multiplies SheetJS's per-cell work and memory by up to about 255.
287
+
288
+ ## Next.js
289
+
290
+ `@yolk-sdk/extractors/node` uses Node APIs and starts its worker from the package's
291
+ `dist/node/extraction-worker.mjs`: one self-contained file with Effect, fflate, mammoth, unpdf, and
292
+ the package's own code inlined. Its only other import is the optional `xlsx` peer, loaded
293
+ dynamically with its version check. Keep both packages unbundled in server code:
294
+
295
+ ```ts
296
+ // next.config.ts
297
+ const nextConfig = {
298
+ serverExternalPackages: ['@yolk-sdk/extractors', 'xlsx']
299
+ }
300
+ ```
301
+
302
+ The default worker location is found from the package's own module (`dist/node/…` installed, or
303
+ `src/node/…` in a workspace that links the source, which needs the package built first). If a
304
+ bundler moves that module into a chunk of another name, there is no default: extraction fails
305
+ closed with `reason: 'worker-unavailable'` and starts nothing. Then copy
306
+ `@yolk-sdk/extractors/node/extraction-worker` somewhere with `xlsx` resolvable beside it and pass
307
+ its URL as `isolation.workerUrl`. With the package external, Next.js 16 (Turbopack) output file
308
+ tracing includes the worker. This was checked with an `output: 'standalone'` build that runs from
309
+ a copy of the traced files. If your tracer misses `dist/node/extraction-worker.mjs`, add it (and
310
+ `xlsx`) with `outputFileTracingIncludes`.
311
+
312
+ ## Host responsibilities
313
+
314
+ - Upload size and auth policy, storage, and any per-user quotas.
315
+ - Choosing limits that fit your runtime's memory and the model context you feed the text to. The
316
+ worker limits bound V8 heap and time, not the process: your platform's memory limit is the
317
+ outer bound (see Worker isolation).
318
+ - Installing SheetJS from the CDN when XLSX extraction is needed.
@@ -0,0 +1,60 @@
1
+ import * as Schema from "effect/Schema";
2
+
3
+ //#region src/errors.d.ts
4
+ /** The SheetJS tarball consumers install; npm `xlsx` stops at the vulnerable 0.18.5. */
5
+ declare const sheetJsInstallCommand = "pnpm add https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz";
6
+ /** Lowest SheetJS release with fixes for CVE-2023-30533 and CVE-2024-22363. */
7
+ declare const minimumSheetJsVersion = "0.20.3";
8
+ /**
9
+ * Why an isolated extraction stopped before the parsers finished:
10
+ *
11
+ * - `resource-limit`: the worker ran out of its V8 heap (`maxOldGenerationSizeMb`, …);
12
+ * - `timeout`: the worker exceeded `timeoutMs` and was terminated;
13
+ * - `worker-unavailable`: the worker could not start (a missing or unloadable worker file);
14
+ * - `worker-failed`: the worker crashed or exited without a result;
15
+ * - `busy`: no worker slot freed up within `maxQueueWaitMs`, so no worker was started.
16
+ */
17
+ declare const FileExtractionFailureReason: Schema.Literals<readonly ["resource-limit", "timeout", "worker-unavailable", "worker-failed", "busy"]>;
18
+ type FileExtractionFailureReason = typeof FileExtractionFailureReason.Type;
19
+ declare const FileExtractionError_base: Schema.Class<FileExtractionError, Schema.TaggedStruct<"FileExtractionError", {
20
+ readonly message: Schema.String;
21
+ readonly format: Schema.String;
22
+ readonly reason: Schema.optional<Schema.Literals<readonly ["resource-limit", "timeout", "worker-unavailable", "worker-failed", "busy"]>>;
23
+ readonly cause: Schema.optional<Schema.Unknown>;
24
+ }>, import("effect/Cause").YieldableError>;
25
+ /**
26
+ * Reading, validating, or bounding a file failed. `message` is safe to show to users. `reason`
27
+ * is set only when an isolated worker was stopped or never admitted (see
28
+ * `FileExtractionFailureReason`).
29
+ */
30
+ declare class FileExtractionError extends FileExtractionError_base {}
31
+ declare const UnsupportedFileFormatError_base: Schema.Class<UnsupportedFileFormatError, Schema.TaggedStruct<"UnsupportedFileFormatError", {
32
+ readonly filename: Schema.String;
33
+ readonly mediaType: Schema.String;
34
+ }>, import("effect/Cause").YieldableError>;
35
+ /** Neither the filename extension nor the media type maps to a supported format. */
36
+ declare class UnsupportedFileFormatError extends UnsupportedFileFormatError_base {
37
+ get message(): string;
38
+ }
39
+ declare const OfficeArchiveError_base: Schema.Class<OfficeArchiveError, Schema.TaggedStruct<"OfficeArchiveError", {
40
+ readonly message: Schema.String;
41
+ readonly expandedBytes: Schema.optional<Schema.Number>;
42
+ }>, import("effect/Cause").YieldableError>;
43
+ /** A DOCX, XLSX, or PPTX ZIP archive failed bounded validation or normalization. */
44
+ declare class OfficeArchiveError extends OfficeArchiveError_base {}
45
+ declare const SheetJsUnavailableError_base: Schema.Class<SheetJsUnavailableError, Schema.TaggedStruct<"SheetJsUnavailableError", {
46
+ readonly reason: Schema.Literals<readonly ["missing", "invalid", "outdated"]>;
47
+ readonly installedVersion: Schema.optional<Schema.String>;
48
+ readonly cause: Schema.optional<Schema.Unknown>;
49
+ }>, import("effect/Cause").YieldableError>;
50
+ /**
51
+ * SheetJS (`xlsx`, an optional peer) is missing, is not a usable SheetJS module, or is older than
52
+ * 0.20.3. This is host misconfiguration, not a problem with the uploaded file.
53
+ */
54
+ declare class SheetJsUnavailableError extends SheetJsUnavailableError_base {
55
+ get message(): string;
56
+ }
57
+ type FileExtractorError = FileExtractionError | UnsupportedFileFormatError | SheetJsUnavailableError;
58
+ //#endregion
59
+ export { FileExtractionError, FileExtractionFailureReason, FileExtractorError, OfficeArchiveError, SheetJsUnavailableError, UnsupportedFileFormatError, minimumSheetJsVersion, sheetJsInstallCommand };
60
+ //# sourceMappingURL=errors.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"errors.d.mts","names":[],"sources":["../src/errors.ts"],"mappings":";;;;cAIa,qBAAA;AAAb;AAAA,cAGa,qBAAA;;;AAHqB;AAGlC;;;;AAAkC;AAWlC;cAAa,2BAAA,EAA2B,MAAA,CAAA,QAAA;AAAA,KAQ5B,2BAAA,UAAqC,2BAAA,CAA4B,IAAI;AAAA,cAAA,wBAAA;;;;;;AAAA;AAAA;;;;AAAA,cAOpE,mBAAA,SAA4B,wBAQxC;AAAA,cAAG,+BAAA;;;;;cAGS,0BAAA,SAAmC,+BAM/C;EAAA,IACK,OAAO,CAAA;AAAA;AAAA,cAGZ,uBAAA;;;;;cAGY,kBAAA,SAA2B,uBAMvC;AAAA,cAAG,4BAAA;;;;;;;;;cAMS,uBAAA,SAAgC,4BAO5C;EAAA,IACK,OAAO,CAAA;AAAA;AAAA,KAeD,kBAAA,GACR,mBAAA,GACA,0BAAA,GACA,uBAAA"}
@@ -0,0 +1,69 @@
1
+ import { Match } from "effect";
2
+ import * as Schema from "effect/Schema";
3
+ //#region src/errors.ts
4
+ /** The SheetJS tarball consumers install; npm `xlsx` stops at the vulnerable 0.18.5. */
5
+ const sheetJsInstallCommand = "pnpm add https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz";
6
+ /** Lowest SheetJS release with fixes for CVE-2023-30533 and CVE-2024-22363. */
7
+ const minimumSheetJsVersion = "0.20.3";
8
+ /**
9
+ * Why an isolated extraction stopped before the parsers finished:
10
+ *
11
+ * - `resource-limit`: the worker ran out of its V8 heap (`maxOldGenerationSizeMb`, …);
12
+ * - `timeout`: the worker exceeded `timeoutMs` and was terminated;
13
+ * - `worker-unavailable`: the worker could not start (a missing or unloadable worker file);
14
+ * - `worker-failed`: the worker crashed or exited without a result;
15
+ * - `busy`: no worker slot freed up within `maxQueueWaitMs`, so no worker was started.
16
+ */
17
+ const FileExtractionFailureReason = Schema.Literals([
18
+ "resource-limit",
19
+ "timeout",
20
+ "worker-unavailable",
21
+ "worker-failed",
22
+ "busy"
23
+ ]);
24
+ /**
25
+ * Reading, validating, or bounding a file failed. `message` is safe to show to users. `reason`
26
+ * is set only when an isolated worker was stopped or never admitted (see
27
+ * `FileExtractionFailureReason`).
28
+ */
29
+ var FileExtractionError = class extends Schema.TaggedError()("FileExtractionError", {
30
+ message: Schema.String,
31
+ format: Schema.String,
32
+ reason: Schema.optional(FileExtractionFailureReason),
33
+ cause: Schema.optional(Schema.Unknown)
34
+ }) {};
35
+ /** Neither the filename extension nor the media type maps to a supported format. */
36
+ var UnsupportedFileFormatError = class extends Schema.TaggedError()("UnsupportedFileFormatError", {
37
+ filename: Schema.String,
38
+ mediaType: Schema.String
39
+ }) {
40
+ get message() {
41
+ return `Unsupported file format: ${this.filename}`;
42
+ }
43
+ };
44
+ /** A DOCX, XLSX, or PPTX ZIP archive failed bounded validation or normalization. */
45
+ var OfficeArchiveError = class extends Schema.TaggedError()("OfficeArchiveError", {
46
+ message: Schema.String,
47
+ expandedBytes: Schema.optional(Schema.Number)
48
+ }) {};
49
+ /**
50
+ * SheetJS (`xlsx`, an optional peer) is missing, is not a usable SheetJS module, or is older than
51
+ * 0.20.3. This is host misconfiguration, not a problem with the uploaded file.
52
+ */
53
+ var SheetJsUnavailableError = class extends Schema.TaggedError()("SheetJsUnavailableError", {
54
+ reason: Schema.Literals([
55
+ "missing",
56
+ "invalid",
57
+ "outdated"
58
+ ]),
59
+ installedVersion: Schema.optional(Schema.String),
60
+ cause: Schema.optional(Schema.Unknown)
61
+ }) {
62
+ get message() {
63
+ return `${Match.value(this.reason).pipe(Match.when("missing", () => "SheetJS (xlsx) is not installed"), Match.when("outdated", () => `SheetJS ${this.installedVersion ?? "unknown"} is older than ${minimumSheetJsVersion}`), Match.when("invalid", () => "The installed xlsx module is not a usable SheetJS build"), Match.exhaustive)}. XLSX extraction needs SheetJS ${minimumSheetJsVersion} or newer from the SheetJS CDN (npm xlsx is unmaintained and vulnerable): ${sheetJsInstallCommand}`;
64
+ }
65
+ };
66
+ //#endregion
67
+ export { FileExtractionError, FileExtractionFailureReason, OfficeArchiveError, SheetJsUnavailableError, UnsupportedFileFormatError, minimumSheetJsVersion, sheetJsInstallCommand };
68
+
69
+ //# sourceMappingURL=errors.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"errors.mjs","names":[],"sources":["../src/errors.ts"],"sourcesContent":["import { Match } from 'effect'\nimport * as Schema from 'effect/Schema'\n\n/** The SheetJS tarball consumers install; npm `xlsx` stops at the vulnerable 0.18.5. */\nexport const sheetJsInstallCommand = 'pnpm add https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz'\n\n/** Lowest SheetJS release with fixes for CVE-2023-30533 and CVE-2024-22363. */\nexport const minimumSheetJsVersion = '0.20.3'\n\n/**\n * Why an isolated extraction stopped before the parsers finished:\n *\n * - `resource-limit`: the worker ran out of its V8 heap (`maxOldGenerationSizeMb`, …);\n * - `timeout`: the worker exceeded `timeoutMs` and was terminated;\n * - `worker-unavailable`: the worker could not start (a missing or unloadable worker file);\n * - `worker-failed`: the worker crashed or exited without a result;\n * - `busy`: no worker slot freed up within `maxQueueWaitMs`, so no worker was started.\n */\nexport const FileExtractionFailureReason = Schema.Literals([\n 'resource-limit',\n 'timeout',\n 'worker-unavailable',\n 'worker-failed',\n 'busy'\n])\n\nexport type FileExtractionFailureReason = typeof FileExtractionFailureReason.Type\n\n/**\n * Reading, validating, or bounding a file failed. `message` is safe to show to users. `reason`\n * is set only when an isolated worker was stopped or never admitted (see\n * `FileExtractionFailureReason`).\n */\nexport class FileExtractionError extends Schema.TaggedError<FileExtractionError>()(\n 'FileExtractionError',\n {\n message: Schema.String,\n format: Schema.String,\n reason: Schema.optional(FileExtractionFailureReason),\n cause: Schema.optional(Schema.Unknown)\n }\n) {}\n\n/** Neither the filename extension nor the media type maps to a supported format. */\nexport class UnsupportedFileFormatError extends Schema.TaggedError<UnsupportedFileFormatError>()(\n 'UnsupportedFileFormatError',\n {\n filename: Schema.String,\n mediaType: Schema.String\n }\n) {\n get message(): string {\n return `Unsupported file format: ${this.filename}`\n }\n}\n\n/** A DOCX, XLSX, or PPTX ZIP archive failed bounded validation or normalization. */\nexport class OfficeArchiveError extends Schema.TaggedError<OfficeArchiveError>()(\n 'OfficeArchiveError',\n {\n message: Schema.String,\n expandedBytes: Schema.optional(Schema.Number)\n }\n) {}\n\n/**\n * SheetJS (`xlsx`, an optional peer) is missing, is not a usable SheetJS module, or is older than\n * 0.20.3. This is host misconfiguration, not a problem with the uploaded file.\n */\nexport class SheetJsUnavailableError extends Schema.TaggedError<SheetJsUnavailableError>()(\n 'SheetJsUnavailableError',\n {\n reason: Schema.Literals(['missing', 'invalid', 'outdated']),\n installedVersion: Schema.optional(Schema.String),\n cause: Schema.optional(Schema.Unknown)\n }\n) {\n get message(): string {\n const problem = Match.value(this.reason).pipe(\n Match.when('missing', () => 'SheetJS (xlsx) is not installed'),\n Match.when(\n 'outdated',\n () => `SheetJS ${this.installedVersion ?? 'unknown'} is older than ${minimumSheetJsVersion}`\n ),\n Match.when('invalid', () => 'The installed xlsx module is not a usable SheetJS build'),\n Match.exhaustive\n )\n\n return `${problem}. XLSX extraction needs SheetJS ${minimumSheetJsVersion} or newer from the SheetJS CDN (npm xlsx is unmaintained and vulnerable): ${sheetJsInstallCommand}`\n }\n}\n\nexport type FileExtractorError =\n | FileExtractionError\n | UnsupportedFileFormatError\n | SheetJsUnavailableError\n"],"mappings":";;;;AAIA,MAAa,wBAAwB;;AAGrC,MAAa,wBAAwB;;;;;;;;;;AAWrC,MAAa,8BAA8B,OAAO,SAAS;CACzD;CACA;CACA;CACA;CACA;AACF,CAAC;;;;;;AASD,IAAa,sBAAb,cAAyC,OAAO,YAAiC,EAC/E,uBACA;CACE,SAAS,OAAO;CAChB,QAAQ,OAAO;CACf,QAAQ,OAAO,SAAS,2BAA2B;CACnD,OAAO,OAAO,SAAS,OAAO,OAAO;AACvC,CACF,EAAE,CAAC;;AAGH,IAAa,6BAAb,cAAgD,OAAO,YAAwC,EAC7F,8BACA;CACE,UAAU,OAAO;CACjB,WAAW,OAAO;AACpB,CACF,EAAE;CACA,IAAI,UAAkB;EACpB,OAAO,4BAA4B,KAAK;CAC1C;AACF;;AAGA,IAAa,qBAAb,cAAwC,OAAO,YAAgC,EAC7E,sBACA;CACE,SAAS,OAAO;CAChB,eAAe,OAAO,SAAS,OAAO,MAAM;AAC9C,CACF,EAAE,CAAC;;;;;AAMH,IAAa,0BAAb,cAA6C,OAAO,YAAqC,EACvF,2BACA;CACE,QAAQ,OAAO,SAAS;EAAC;EAAW;EAAW;CAAU,CAAC;CAC1D,kBAAkB,OAAO,SAAS,OAAO,MAAM;CAC/C,OAAO,OAAO,SAAS,OAAO,OAAO;AACvC,CACF,EAAE;CACA,IAAI,UAAkB;EAWpB,OAAO,GAVS,MAAM,MAAM,KAAK,MAAM,EAAE,KACvC,MAAM,KAAK,iBAAiB,iCAAiC,GAC7D,MAAM,KACJ,kBACM,WAAW,KAAK,oBAAoB,UAAU,iBAAiB,uBACvE,GACA,MAAM,KAAK,iBAAiB,yDAAyD,GACrF,MAAM,UAGQ,EAAE,kCAAkC,sBAAsB,4EAA4E;CACxJ;AACF"}
@@ -0,0 +1,32 @@
1
+ //#region src/format.d.ts
2
+ declare const extractedFileFormats: readonly ["csv", "docx", "json", "markdown", "pdf", "pptx", "text", "xlsx"];
3
+ type ExtractedFileFormat = (typeof extractedFileFormats)[number];
4
+ /** Office Open XML formats; their ZIP archives are validated before any parser reads them. */
5
+ type OfficeFileFormat = 'docx' | 'pptx' | 'xlsx';
6
+ type FileInput = {
7
+ readonly filename: string;
8
+ readonly mediaType: string;
9
+ readonly bytes: Uint8Array;
10
+ };
11
+ type ExtractedFileMetadata = {
12
+ readonly format: ExtractedFileFormat;
13
+ readonly title?: string;
14
+ readonly pageCount?: number;
15
+ readonly sheetNames?: ReadonlyArray<string>;
16
+ };
17
+ type ExtractedFile = {
18
+ readonly content: string;
19
+ readonly metadata: ExtractedFileMetadata;
20
+ };
21
+ /**
22
+ * Pick the extraction format: the filename extension wins, then the media type, then any
23
+ * `text/*` media type as plain text. `undefined` means the file is unsupported.
24
+ */
25
+ declare const fileFormatFor: (input: {
26
+ readonly filename: string;
27
+ readonly mediaType: string;
28
+ }) => ExtractedFileFormat | undefined;
29
+ declare const isOfficeFileFormat: (format: ExtractedFileFormat) => format is OfficeFileFormat;
30
+ //#endregion
31
+ export { ExtractedFile, ExtractedFileFormat, ExtractedFileMetadata, FileInput, OfficeFileFormat, extractedFileFormats, fileFormatFor, isOfficeFileFormat };
32
+ //# sourceMappingURL=format.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"format.d.mts","names":[],"sources":["../src/format.ts"],"mappings":";cAAa,oBAAA;AAAA,KAWD,mBAAA,WAA8B,oBAAoB;;KAGlD,gBAAA;AAAA,KAEA,SAAA;EAAA,SACD,QAAA;EAAA,SACA,SAAA;EAAA,SACA,KAAA,EAAO,UAAU;AAAA;AAAA,KAGhB,qBAAA;EAAA,SACD,MAAA,EAAQ,mBAAA;EAAA,SACR,KAAA;EAAA,SACA,SAAA;EAAA,SACA,UAAA,GAAa,aAAa;AAAA;AAAA,KAGzB,aAAA;EAAA,SACD,OAAA;EAAA,SACA,QAAA,EAAU,qBAAqB;AAAA;;;;;cAoC7B,aAAA,GAAiB,KAAA;EAAA,SACnB,QAAA;EAAA,SACA,SAAA;AAAA,MACP,mBAAmB;AAAA,cAYV,kBAAA,GAAsB,MAAA,EAAQ,mBAAA,KAAsB,MAAA,IAAU,gBACd"}
@@ -0,0 +1,52 @@
1
+ //#region src/format.ts
2
+ const extractedFileFormats = [
3
+ "csv",
4
+ "docx",
5
+ "json",
6
+ "markdown",
7
+ "pdf",
8
+ "pptx",
9
+ "text",
10
+ "xlsx"
11
+ ];
12
+ const extensionFor = (filename) => {
13
+ const lower = filename.toLowerCase();
14
+ const dotIndex = lower.lastIndexOf(".");
15
+ return dotIndex === -1 ? "" : lower.slice(dotIndex + 1);
16
+ };
17
+ const formatsByExtension = new Map([
18
+ ["txt", "text"],
19
+ ["md", "markdown"],
20
+ ["markdown", "markdown"],
21
+ ["csv", "csv"],
22
+ ["json", "json"],
23
+ ["pdf", "pdf"],
24
+ ["docx", "docx"],
25
+ ["xlsx", "xlsx"],
26
+ ["pptx", "pptx"]
27
+ ]);
28
+ const formatsByMediaType = new Map([
29
+ ["application/pdf", "pdf"],
30
+ ["application/vnd.openxmlformats-officedocument.wordprocessingml.document", "docx"],
31
+ ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", "xlsx"],
32
+ ["application/vnd.openxmlformats-officedocument.presentationml.presentation", "pptx"],
33
+ ["application/json", "json"],
34
+ ["text/csv", "csv"],
35
+ ["text/markdown", "markdown"]
36
+ ]);
37
+ /**
38
+ * Pick the extraction format: the filename extension wins, then the media type, then any
39
+ * `text/*` media type as plain text. `undefined` means the file is unsupported.
40
+ */
41
+ const fileFormatFor = (input) => {
42
+ const byExtension = formatsByExtension.get(extensionFor(input.filename));
43
+ if (byExtension !== void 0) return byExtension;
44
+ const byMediaType = formatsByMediaType.get(input.mediaType);
45
+ if (byMediaType !== void 0) return byMediaType;
46
+ return input.mediaType.startsWith("text/") ? "text" : void 0;
47
+ };
48
+ const isOfficeFileFormat = (format) => format === "docx" || format === "pptx" || format === "xlsx";
49
+ //#endregion
50
+ export { extractedFileFormats, fileFormatFor, isOfficeFileFormat };
51
+
52
+ //# sourceMappingURL=format.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"format.mjs","names":[],"sources":["../src/format.ts"],"sourcesContent":["export const extractedFileFormats = [\n 'csv',\n 'docx',\n 'json',\n 'markdown',\n 'pdf',\n 'pptx',\n 'text',\n 'xlsx'\n] as const\n\nexport type ExtractedFileFormat = (typeof extractedFileFormats)[number]\n\n/** Office Open XML formats; their ZIP archives are validated before any parser reads them. */\nexport type OfficeFileFormat = 'docx' | 'pptx' | 'xlsx'\n\nexport type FileInput = {\n readonly filename: string\n readonly mediaType: string\n readonly bytes: Uint8Array\n}\n\nexport type ExtractedFileMetadata = {\n readonly format: ExtractedFileFormat\n readonly title?: string\n readonly pageCount?: number\n readonly sheetNames?: ReadonlyArray<string>\n}\n\nexport type ExtractedFile = {\n readonly content: string\n readonly metadata: ExtractedFileMetadata\n}\n\nconst extensionFor = (filename: string) => {\n const lower = filename.toLowerCase()\n const dotIndex = lower.lastIndexOf('.')\n\n return dotIndex === -1 ? '' : lower.slice(dotIndex + 1)\n}\n\nconst formatsByExtension: ReadonlyMap<string, ExtractedFileFormat> = new Map([\n ['txt', 'text'],\n ['md', 'markdown'],\n ['markdown', 'markdown'],\n ['csv', 'csv'],\n ['json', 'json'],\n ['pdf', 'pdf'],\n ['docx', 'docx'],\n ['xlsx', 'xlsx'],\n ['pptx', 'pptx']\n])\n\nconst formatsByMediaType: ReadonlyMap<string, ExtractedFileFormat> = new Map([\n ['application/pdf', 'pdf'],\n ['application/vnd.openxmlformats-officedocument.wordprocessingml.document', 'docx'],\n ['application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', 'xlsx'],\n ['application/vnd.openxmlformats-officedocument.presentationml.presentation', 'pptx'],\n ['application/json', 'json'],\n ['text/csv', 'csv'],\n ['text/markdown', 'markdown']\n])\n\n/**\n * Pick the extraction format: the filename extension wins, then the media type, then any\n * `text/*` media type as plain text. `undefined` means the file is unsupported.\n */\nexport const fileFormatFor = (input: {\n readonly filename: string\n readonly mediaType: string\n}): ExtractedFileFormat | undefined => {\n const byExtension = formatsByExtension.get(extensionFor(input.filename))\n\n if (byExtension !== undefined) return byExtension\n\n const byMediaType = formatsByMediaType.get(input.mediaType)\n\n if (byMediaType !== undefined) return byMediaType\n\n return input.mediaType.startsWith('text/') ? 'text' : undefined\n}\n\nexport const isOfficeFileFormat = (format: ExtractedFileFormat): format is OfficeFileFormat =>\n format === 'docx' || format === 'pptx' || format === 'xlsx'\n"],"mappings":";AAAA,MAAa,uBAAuB;CAClC;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAyBA,MAAM,gBAAgB,aAAqB;CACzC,MAAM,QAAQ,SAAS,YAAY;CACnC,MAAM,WAAW,MAAM,YAAY,GAAG;CAEtC,OAAO,aAAa,KAAK,KAAK,MAAM,MAAM,WAAW,CAAC;AACxD;AAEA,MAAM,qBAA+D,IAAI,IAAI;CAC3E,CAAC,OAAO,MAAM;CACd,CAAC,MAAM,UAAU;CACjB,CAAC,YAAY,UAAU;CACvB,CAAC,OAAO,KAAK;CACb,CAAC,QAAQ,MAAM;CACf,CAAC,OAAO,KAAK;CACb,CAAC,QAAQ,MAAM;CACf,CAAC,QAAQ,MAAM;CACf,CAAC,QAAQ,MAAM;AACjB,CAAC;AAED,MAAM,qBAA+D,IAAI,IAAI;CAC3E,CAAC,mBAAmB,KAAK;CACzB,CAAC,2EAA2E,MAAM;CAClF,CAAC,qEAAqE,MAAM;CAC5E,CAAC,6EAA6E,MAAM;CACpF,CAAC,oBAAoB,MAAM;CAC3B,CAAC,YAAY,KAAK;CAClB,CAAC,iBAAiB,UAAU;AAC9B,CAAC;;;;;AAMD,MAAa,iBAAiB,UAGS;CACrC,MAAM,cAAc,mBAAmB,IAAI,aAAa,MAAM,QAAQ,CAAC;CAEvE,IAAI,gBAAgB,KAAA,GAAW,OAAO;CAEtC,MAAM,cAAc,mBAAmB,IAAI,MAAM,SAAS;CAE1D,IAAI,gBAAgB,KAAA,GAAW,OAAO;CAEtC,OAAO,MAAM,UAAU,WAAW,OAAO,IAAI,SAAS,KAAA;AACxD;AAEA,MAAa,sBAAsB,WACjC,WAAW,UAAU,WAAW,UAAU,WAAW"}
@@ -0,0 +1,6 @@
1
+ import { FileExtractionError, FileExtractionFailureReason, FileExtractorError, OfficeArchiveError, SheetJsUnavailableError, UnsupportedFileFormatError, minimumSheetJsVersion, sheetJsInstallCommand } from "./errors.mjs";
2
+ import { ExtractedFile, ExtractedFileFormat, ExtractedFileMetadata, FileInput, OfficeFileFormat, extractedFileFormats, fileFormatFor, isOfficeFileFormat } from "./format.mjs";
3
+ import { FileExtractorLimits, defaultFileExtractorLimits } from "./limits.mjs";
4
+ import { sanitizeExtractedText } from "./sanitize.mjs";
5
+ import { FileExtractor, FileExtractorApi } from "./service.mjs";
6
+ export { type ExtractedFile, type ExtractedFileFormat, type ExtractedFileMetadata, FileExtractionError, FileExtractionFailureReason, FileExtractor, type FileExtractorApi, type FileExtractorError, FileExtractorLimits, type FileInput, OfficeArchiveError, type OfficeFileFormat, SheetJsUnavailableError, UnsupportedFileFormatError, defaultFileExtractorLimits, extractedFileFormats, fileFormatFor, isOfficeFileFormat, minimumSheetJsVersion, sanitizeExtractedText, sheetJsInstallCommand };
package/dist/index.mjs ADDED
@@ -0,0 +1,6 @@
1
+ import { FileExtractionError, FileExtractionFailureReason, OfficeArchiveError, SheetJsUnavailableError, UnsupportedFileFormatError, minimumSheetJsVersion, sheetJsInstallCommand } from "./errors.mjs";
2
+ import { extractedFileFormats, fileFormatFor, isOfficeFileFormat } from "./format.mjs";
3
+ import { FileExtractorLimits, defaultFileExtractorLimits } from "./limits.mjs";
4
+ import { sanitizeExtractedText } from "./sanitize.mjs";
5
+ import { FileExtractor } from "./service.mjs";
6
+ export { FileExtractionError, FileExtractionFailureReason, FileExtractor, FileExtractorLimits, OfficeArchiveError, SheetJsUnavailableError, UnsupportedFileFormatError, defaultFileExtractorLimits, extractedFileFormats, fileFormatFor, isOfficeFileFormat, minimumSheetJsVersion, sanitizeExtractedText, sheetJsInstallCommand };
@@ -0,0 +1,17 @@
1
+ import { FileExtractor, FileExtractorApi } from "./service.mjs";
2
+ import { Layer } from "effect";
3
+ import { KnowledgeExtractor, KnowledgeExtractorApi } from "@yolk-sdk/knowledge/extraction";
4
+
5
+ //#region src/knowledge.d.ts
6
+ /**
7
+ * A `KnowledgeExtractor` backed by a `FileExtractor`. String content is already text and passes
8
+ * through unchanged (it must not be blank); bytes are extracted with the format chosen from the
9
+ * source name and media type. File metadata (`format`, `pageCount`, `sheetNames`) is merged
10
+ * under the loaded source's own metadata, and the extracted title becomes the document title.
11
+ */
12
+ declare const makeFileKnowledgeExtractor: (extractor: FileExtractorApi) => KnowledgeExtractorApi;
13
+ /** Provide `KnowledgeExtractor` from the `FileExtractor` in context (for example the Node layer). */
14
+ declare const FileKnowledgeExtractorLayer: Layer.Layer<KnowledgeExtractor, never, FileExtractor>;
15
+ //#endregion
16
+ export { FileKnowledgeExtractorLayer, makeFileKnowledgeExtractor };
17
+ //# sourceMappingURL=knowledge.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"knowledge.d.mts","names":[],"sources":["../src/knowledge.ts"],"mappings":";;;;;AAiEA;;;;;;AAAA,cAAa,0BAAA,GAA8B,SAAA,EAAW,gBAAA,KAAmB,qBAyBvE;;cAGW,2BAAA,EAA2B,KAAA,CAAA,KAAA,CAAA,kBAAA,SAAA,aAAA"}