@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
import { FileExtractionError } from "../errors.mjs";
|
|
2
|
+
import { cellCount, encodeCellAddress, parseCellRange } from "./xlsx-range.mjs";
|
|
3
|
+
import { makeHyperlinkLookup } from "./xlsx-hyperlinks.mjs";
|
|
4
|
+
import { Predicate } from "effect";
|
|
5
|
+
//#region src/node/xlsx-text.ts
|
|
6
|
+
const outputLimitReason = "output limit";
|
|
7
|
+
const hyperlinkLimitReason = "hyperlink limit";
|
|
8
|
+
const omittedHyperlinksMarker = (reasons) => `\n\n[Some hyperlinks omitted: ${reasons.join(" and ")}]`;
|
|
9
|
+
/**
|
|
10
|
+
* Space reserved for the marker whenever a workbook has hyperlinks, so it always fits.
|
|
11
|
+
* `maxXlsxTextCharacters` must exceed it (`minimumXlsxTextCharacters`).
|
|
12
|
+
*/
|
|
13
|
+
const omittedHyperlinksMarkerReserve = omittedHyperlinksMarker([outputLimitReason, hyperlinkLimitReason]).length;
|
|
14
|
+
const invalidRange = () => new FileExtractionError({
|
|
15
|
+
format: "xlsx",
|
|
16
|
+
message: "Invalid XLSX worksheet range."
|
|
17
|
+
});
|
|
18
|
+
const tooLarge = () => new FileExtractionError({
|
|
19
|
+
format: "xlsx",
|
|
20
|
+
message: "XLSX exceeds the worksheet or cell-visit limit."
|
|
21
|
+
});
|
|
22
|
+
const outputTooLarge = () => new FileExtractionError({
|
|
23
|
+
format: "xlsx",
|
|
24
|
+
message: "XLSX extracted text exceeds the output limit."
|
|
25
|
+
});
|
|
26
|
+
/** Read an own property without walking the prototype chain (or trusting a polluted one). */
|
|
27
|
+
const ownProperty = (target, key) => {
|
|
28
|
+
const descriptor = Object.getOwnPropertyDescriptor(target, key);
|
|
29
|
+
if (descriptor === void 0) return void 0;
|
|
30
|
+
return "value" in descriptor ? descriptor.value : descriptor.get?.call(target);
|
|
31
|
+
};
|
|
32
|
+
const cellText = (cell) => {
|
|
33
|
+
if (ownProperty(cell, "t") === "z") return "";
|
|
34
|
+
const value = ownProperty(cell, "v");
|
|
35
|
+
if (value === void 0 || value === null) return "";
|
|
36
|
+
const display = ownProperty(cell, "w");
|
|
37
|
+
if (Predicate.isString(display)) return display;
|
|
38
|
+
if (Predicate.isString(value)) return value;
|
|
39
|
+
if (Predicate.isBoolean(value)) return value ? "TRUE" : "FALSE";
|
|
40
|
+
if (value instanceof Date && Number.isFinite(value.getTime())) return value.toISOString();
|
|
41
|
+
if (Predicate.isNumber(value) && Number.isFinite(value)) return String(value);
|
|
42
|
+
throw new FileExtractionError({
|
|
43
|
+
format: "xlsx",
|
|
44
|
+
message: "Invalid XLSX cell value."
|
|
45
|
+
});
|
|
46
|
+
};
|
|
47
|
+
/**
|
|
48
|
+
* CSV length of `text` (quotes doubled, wrapped when needed). The scan stops once the length
|
|
49
|
+
* exceeds `limit`; the returned length is then only known to be above it.
|
|
50
|
+
*/
|
|
51
|
+
const csvField = (text, limit = Number.POSITIVE_INFINITY) => {
|
|
52
|
+
let quoteCount = 0;
|
|
53
|
+
let quote = text === "ID";
|
|
54
|
+
for (let index = 0; index < text.length; index += 1) {
|
|
55
|
+
const character = text.charCodeAt(index);
|
|
56
|
+
if (character === 34) quoteCount += 1;
|
|
57
|
+
if (character === 34 || character === 44 || character === 10 || character === 13) {
|
|
58
|
+
quote = true;
|
|
59
|
+
if (text.length + quoteCount + 2 > limit) break;
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
return {
|
|
63
|
+
quote,
|
|
64
|
+
length: text.length + quoteCount + (quote ? 2 : 0)
|
|
65
|
+
};
|
|
66
|
+
};
|
|
67
|
+
/** Write bounded CSV for every sheet; throws once `budget` characters would be exceeded. */
|
|
68
|
+
const renderSheets = (sheets, budget, decorate) => {
|
|
69
|
+
let characters = 0;
|
|
70
|
+
const output = [];
|
|
71
|
+
const append = (text) => {
|
|
72
|
+
if (text.length > budget - characters) throw outputTooLarge();
|
|
73
|
+
characters += text.length;
|
|
74
|
+
output.push(text);
|
|
75
|
+
};
|
|
76
|
+
const appendCell = (text) => {
|
|
77
|
+
if (text.length > budget - characters) throw outputTooLarge();
|
|
78
|
+
const field = csvField(text, budget - characters);
|
|
79
|
+
if (field.length > budget - characters) throw outputTooLarge();
|
|
80
|
+
append(field.quote ? `"${text.replaceAll("\"", "\"\"")}"` : text);
|
|
81
|
+
};
|
|
82
|
+
for (const visited of sheets) {
|
|
83
|
+
const { name, sheet, range } = visited;
|
|
84
|
+
if (sheet === void 0) continue;
|
|
85
|
+
if (output.length > 0) append("\n\n");
|
|
86
|
+
append("# ");
|
|
87
|
+
append(name);
|
|
88
|
+
append("\n");
|
|
89
|
+
if (range === void 0) continue;
|
|
90
|
+
for (let row = range.start.r; row <= range.end.r; row += 1) {
|
|
91
|
+
if (row > range.start.r) append("\n");
|
|
92
|
+
for (let column = range.start.c; column <= range.end.c; column += 1) {
|
|
93
|
+
if (column > range.start.c) append(",");
|
|
94
|
+
const cell = ownProperty(sheet, encodeCellAddress({
|
|
95
|
+
r: row,
|
|
96
|
+
c: column
|
|
97
|
+
}));
|
|
98
|
+
if (!Predicate.isObject(cell)) {
|
|
99
|
+
appendCell("");
|
|
100
|
+
continue;
|
|
101
|
+
}
|
|
102
|
+
const text = cellText(cell);
|
|
103
|
+
appendCell(decorate?.({
|
|
104
|
+
sheet: visited,
|
|
105
|
+
row,
|
|
106
|
+
column,
|
|
107
|
+
text
|
|
108
|
+
}) ?? text);
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return output.join("");
|
|
113
|
+
};
|
|
114
|
+
/**
|
|
115
|
+
* Validate every sheet before touching any cell, then generate bounded CSV incrementally, never
|
|
116
|
+
* with `sheet_to_csv` (which can walk billions of absent cells or allocate an unbounded quoted
|
|
117
|
+
* string).
|
|
118
|
+
*
|
|
119
|
+
* Existing cells inside an external hyperlink are written as `text <url>`. Plain text is
|
|
120
|
+
* rendered first, so links only use budget the plain text leaves over: in document order, an
|
|
121
|
+
* annotation that no longer fits leaves its cell plain and one marker ends the output.
|
|
122
|
+
*/
|
|
123
|
+
const extractBoundedXlsxText = (workbook, limits, options = {}) => {
|
|
124
|
+
if (workbook.SheetNames.length > limits.maxXlsxSheets) throw tooLarge();
|
|
125
|
+
let visits = 0;
|
|
126
|
+
const sheets = workbook.SheetNames.map((name) => {
|
|
127
|
+
const candidate = ownProperty(workbook.Sheets, name);
|
|
128
|
+
const sheet = Predicate.isObject(candidate) ? candidate : void 0;
|
|
129
|
+
const ref = sheet === void 0 ? void 0 : ownProperty(sheet, "!ref");
|
|
130
|
+
if (ref !== void 0 && !Predicate.isString(ref)) throw invalidRange();
|
|
131
|
+
const range = ref === void 0 ? void 0 : parseCellRange(ref);
|
|
132
|
+
if (ref !== void 0 && range === void 0) throw invalidRange();
|
|
133
|
+
visits += range === void 0 ? 0 : cellCount(range);
|
|
134
|
+
if (!Number.isSafeInteger(visits) || visits > limits.maxXlsxCellVisits) throw tooLarge();
|
|
135
|
+
return {
|
|
136
|
+
name,
|
|
137
|
+
sheet,
|
|
138
|
+
range
|
|
139
|
+
};
|
|
140
|
+
});
|
|
141
|
+
const links = options.hyperlinks ?? /* @__PURE__ */ new Map();
|
|
142
|
+
const truncated = options.hyperlinksTruncated === true;
|
|
143
|
+
if (links.size === 0 && !truncated) return renderSheets(sheets, limits.maxXlsxTextCharacters);
|
|
144
|
+
const budget = limits.maxXlsxTextCharacters - omittedHyperlinksMarkerReserve;
|
|
145
|
+
let spare = budget - renderSheets(sheets, budget).length;
|
|
146
|
+
let dropped = false;
|
|
147
|
+
const lookups = /* @__PURE__ */ new Map();
|
|
148
|
+
const lookupFor = (sheet) => {
|
|
149
|
+
const sheetLinks = links.get(sheet.name);
|
|
150
|
+
if (sheetLinks === void 0 || sheet.range === void 0) return void 0;
|
|
151
|
+
const existing = lookups.get(sheet.name);
|
|
152
|
+
if (existing !== void 0) return existing;
|
|
153
|
+
const lookup = makeHyperlinkLookup(sheetLinks, sheet.range);
|
|
154
|
+
lookups.set(sheet.name, lookup);
|
|
155
|
+
return lookup;
|
|
156
|
+
};
|
|
157
|
+
const output = renderSheets(sheets, budget, ({ sheet, row, column, text }) => {
|
|
158
|
+
const link = lookupFor(sheet)?.at(row, column);
|
|
159
|
+
if (link === void 0 || link.target === text) return void 0;
|
|
160
|
+
const label = text.length > 0 ? text : link.display ?? "";
|
|
161
|
+
const plainLength = csvField(text).length;
|
|
162
|
+
if (label.length + (label.length > 0 ? 3 : 2) + link.target.length - plainLength > spare) {
|
|
163
|
+
dropped = true;
|
|
164
|
+
return;
|
|
165
|
+
}
|
|
166
|
+
const annotated = label.length > 0 ? `${label} <${link.target}>` : `<${link.target}>`;
|
|
167
|
+
const extra = csvField(annotated, plainLength + spare).length - plainLength;
|
|
168
|
+
if (extra > spare) {
|
|
169
|
+
dropped = true;
|
|
170
|
+
return;
|
|
171
|
+
}
|
|
172
|
+
spare -= extra;
|
|
173
|
+
return annotated;
|
|
174
|
+
});
|
|
175
|
+
const reasons = [...dropped ? [outputLimitReason] : [], ...truncated ? [hyperlinkLimitReason] : []];
|
|
176
|
+
return reasons.length === 0 ? output : output + omittedHyperlinksMarker(reasons);
|
|
177
|
+
};
|
|
178
|
+
//#endregion
|
|
179
|
+
export { extractBoundedXlsxText, omittedHyperlinksMarkerReserve };
|
|
180
|
+
|
|
181
|
+
//# sourceMappingURL=xlsx-text.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-text.mjs","names":[],"sources":["../../src/node/xlsx-text.ts"],"sourcesContent":["import { Predicate } from 'effect'\nimport { FileExtractionError } from '../errors.ts'\nimport type { FileExtractorLimits } from '../limits.ts'\nimport { makeHyperlinkLookup } from './xlsx-hyperlinks.ts'\nimport type { HyperlinkLookup, XlsxHyperlinks } from './xlsx-hyperlinks.ts'\nimport type { CellRange } from './xlsx-range.ts'\nimport { cellCount, encodeCellAddress, parseCellRange } from './xlsx-range.ts'\n\nexport type XlsxTextLimits = Pick<\n FileExtractorLimits,\n 'maxXlsxCellVisits' | 'maxXlsxSheets' | 'maxXlsxTextCharacters'\n>\n\n/** The parsed workbook as SheetJS returns it; sheets and cells are read as own properties. */\nexport type XlsxWorkbook = {\n readonly SheetNames: ReadonlyArray<string>\n readonly Sheets: object\n}\n\nexport type XlsxTextOptions = {\n readonly hyperlinks?: XlsxHyperlinks\n /** Some hyperlinks were never read because the workbook exceeded `maxXlsxHyperlinks`. */\n readonly hyperlinksTruncated?: boolean\n}\n\nconst outputLimitReason = 'output limit'\n\nconst hyperlinkLimitReason = 'hyperlink limit'\n\nconst omittedHyperlinksMarker = (reasons: ReadonlyArray<string>) =>\n `\\n\\n[Some hyperlinks omitted: ${reasons.join(' and ')}]`\n\n/**\n * Space reserved for the marker whenever a workbook has hyperlinks, so it always fits.\n * `maxXlsxTextCharacters` must exceed it (`minimumXlsxTextCharacters`).\n */\nexport const omittedHyperlinksMarkerReserve = omittedHyperlinksMarker([\n outputLimitReason,\n hyperlinkLimitReason\n]).length\n\nconst invalidRange = () =>\n new FileExtractionError({ format: 'xlsx', message: 'Invalid XLSX worksheet range.' })\n\nconst tooLarge = () =>\n new FileExtractionError({\n format: 'xlsx',\n message: 'XLSX exceeds the worksheet or cell-visit limit.'\n })\n\nconst outputTooLarge = () =>\n new FileExtractionError({\n format: 'xlsx',\n message: 'XLSX extracted text exceeds the output limit.'\n })\n\n/** Read an own property without walking the prototype chain (or trusting a polluted one). */\nconst ownProperty = (target: object, key: string): unknown => {\n const descriptor = Object.getOwnPropertyDescriptor(target, key)\n\n if (descriptor === undefined) return undefined\n\n return 'value' in descriptor ? descriptor.value : descriptor.get?.call(target)\n}\n\nconst cellText = (cell: object): string => {\n if (ownProperty(cell, 't') === 'z') return ''\n\n const value = ownProperty(cell, 'v')\n\n // SheetJS runs with `cellFormula: false`: only cached values exist, formula-only cells are empty.\n if (value === undefined || value === null) return ''\n\n // Prefer parser-provided display text. Do not run an untrusted format template here.\n const display = ownProperty(cell, 'w')\n\n if (Predicate.isString(display)) return display\n\n if (Predicate.isString(value)) return value\n\n if (Predicate.isBoolean(value)) return value ? 'TRUE' : 'FALSE'\n\n if (value instanceof Date && Number.isFinite(value.getTime())) return value.toISOString()\n\n if (Predicate.isNumber(value) && Number.isFinite(value)) return String(value)\n\n throw new FileExtractionError({ format: 'xlsx', message: 'Invalid XLSX cell value.' })\n}\n\n/**\n * CSV length of `text` (quotes doubled, wrapped when needed). The scan stops once the length\n * exceeds `limit`; the returned length is then only known to be above it.\n */\nconst csvField = (text: string, limit = Number.POSITIVE_INFINITY) => {\n let quoteCount = 0\n let quote = text === 'ID'\n\n for (let index = 0; index < text.length; index += 1) {\n const character = text.charCodeAt(index)\n\n // '\"', ',', '\\n', '\\r'\n if (character === 34) quoteCount += 1\n\n if (character === 34 || character === 44 || character === 10 || character === 13) {\n quote = true\n\n if (text.length + quoteCount + 2 > limit) break\n }\n }\n\n return { quote, length: text.length + quoteCount + (quote ? 2 : 0) }\n}\n\ntype VisitedSheet = {\n readonly name: string\n readonly sheet: object | undefined\n readonly range: CellRange | undefined\n}\n\n/** Annotated text for a cell, or `undefined` to write it plain. */\ntype CellDecorator = (input: {\n readonly sheet: VisitedSheet\n readonly row: number\n readonly column: number\n readonly text: string\n}) => string | undefined\n\n/** Write bounded CSV for every sheet; throws once `budget` characters would be exceeded. */\nconst renderSheets = (\n sheets: ReadonlyArray<VisitedSheet>,\n budget: number,\n decorate?: CellDecorator\n) => {\n let characters = 0\n const output: Array<string> = []\n\n const append = (text: string) => {\n if (text.length > budget - characters) throw outputTooLarge()\n\n characters += text.length\n output.push(text)\n }\n\n const appendCell = (text: string) => {\n if (text.length > budget - characters) throw outputTooLarge()\n\n const field = csvField(text, budget - characters)\n\n if (field.length > budget - characters) throw outputTooLarge()\n\n append(field.quote ? `\"${text.replaceAll('\"', '\"\"')}\"` : text)\n }\n\n for (const visited of sheets) {\n const { name, sheet, range } = visited\n\n if (sheet === undefined) continue\n\n if (output.length > 0) append('\\n\\n')\n\n append('# ')\n append(name)\n append('\\n')\n\n if (range === undefined) continue\n\n for (let row = range.start.r; row <= range.end.r; row += 1) {\n if (row > range.start.r) append('\\n')\n\n for (let column = range.start.c; column <= range.end.c; column += 1) {\n if (column > range.start.c) append(',')\n\n const cell = ownProperty(sheet, encodeCellAddress({ r: row, c: column }))\n\n if (!Predicate.isObject(cell)) {\n appendCell('')\n continue\n }\n\n const text = cellText(cell)\n\n appendCell(decorate?.({ sheet: visited, row, column, text }) ?? text)\n }\n }\n }\n\n return output.join('')\n}\n\n/**\n * Validate every sheet before touching any cell, then generate bounded CSV incrementally, never\n * with `sheet_to_csv` (which can walk billions of absent cells or allocate an unbounded quoted\n * string).\n *\n * Existing cells inside an external hyperlink are written as `text <url>`. Plain text is\n * rendered first, so links only use budget the plain text leaves over: in document order, an\n * annotation that no longer fits leaves its cell plain and one marker ends the output.\n */\nexport const extractBoundedXlsxText = (\n workbook: XlsxWorkbook,\n limits: XlsxTextLimits,\n options: XlsxTextOptions = {}\n) => {\n if (workbook.SheetNames.length > limits.maxXlsxSheets) throw tooLarge()\n\n let visits = 0\n\n const sheets = workbook.SheetNames.map((name): VisitedSheet => {\n const candidate = ownProperty(workbook.Sheets, name)\n const sheet = Predicate.isObject(candidate) ? candidate : undefined\n const ref = sheet === undefined ? undefined : ownProperty(sheet, '!ref')\n\n if (ref !== undefined && !Predicate.isString(ref)) throw invalidRange()\n\n const range = ref === undefined ? undefined : parseCellRange(ref)\n\n if (ref !== undefined && range === undefined) throw invalidRange()\n\n visits += range === undefined ? 0 : cellCount(range)\n\n if (!Number.isSafeInteger(visits) || visits > limits.maxXlsxCellVisits) throw tooLarge()\n\n return { name, sheet, range }\n })\n\n const links = options.hyperlinks ?? new Map()\n const truncated = options.hyperlinksTruncated === true\n\n if (links.size === 0 && !truncated) return renderSheets(sheets, limits.maxXlsxTextCharacters)\n\n // Reserve the marker's space so it always fits inside the character limit.\n const budget = limits.maxXlsxTextCharacters - omittedHyperlinksMarkerReserve\n let spare = budget - renderSheets(sheets, budget).length\n let dropped = false\n const lookups = new Map<string, HyperlinkLookup>()\n\n const lookupFor = (sheet: VisitedSheet) => {\n const sheetLinks = links.get(sheet.name)\n\n if (sheetLinks === undefined || sheet.range === undefined) return undefined\n\n const existing = lookups.get(sheet.name)\n\n if (existing !== undefined) return existing\n\n const lookup = makeHyperlinkLookup(sheetLinks, sheet.range)\n lookups.set(sheet.name, lookup)\n\n return lookup\n }\n\n const output = renderSheets(sheets, budget, ({ sheet, row, column, text }) => {\n const link = lookupFor(sheet)?.at(row, column)\n\n if (link === undefined || link.target === text) return undefined\n\n const label = text.length > 0 ? text : (link.display ?? '')\n const plainLength = csvField(text).length\n const annotatedLength = label.length + (label.length > 0 ? 3 : 2) + link.target.length\n\n // Constant-time bound first (CSV escaping only adds), then a scan capped at the budget.\n if (annotatedLength - plainLength > spare) {\n dropped = true\n\n return undefined\n }\n\n const annotated = label.length > 0 ? `${label} <${link.target}>` : `<${link.target}>`\n const extra = csvField(annotated, plainLength + spare).length - plainLength\n\n if (extra > spare) {\n dropped = true\n\n return undefined\n }\n\n spare -= extra\n\n return annotated\n })\n\n const reasons = [\n ...(dropped ? [outputLimitReason] : []),\n ...(truncated ? [hyperlinkLimitReason] : [])\n ]\n\n return reasons.length === 0 ? output : output + omittedHyperlinksMarker(reasons)\n}\n"],"mappings":";;;;;AAyBA,MAAM,oBAAoB;AAE1B,MAAM,uBAAuB;AAE7B,MAAM,2BAA2B,YAC/B,iCAAiC,QAAQ,KAAK,OAAO,EAAE;;;;;AAMzD,MAAa,iCAAiC,wBAAwB,CACpE,mBACA,oBACF,CAAC,EAAE;AAEH,MAAM,qBACJ,IAAI,oBAAoB;CAAE,QAAQ;CAAQ,SAAS;AAAgC,CAAC;AAEtF,MAAM,iBACJ,IAAI,oBAAoB;CACtB,QAAQ;CACR,SAAS;AACX,CAAC;AAEH,MAAM,uBACJ,IAAI,oBAAoB;CACtB,QAAQ;CACR,SAAS;AACX,CAAC;;AAGH,MAAM,eAAe,QAAgB,QAAyB;CAC5D,MAAM,aAAa,OAAO,yBAAyB,QAAQ,GAAG;CAE9D,IAAI,eAAe,KAAA,GAAW,OAAO,KAAA;CAErC,OAAO,WAAW,aAAa,WAAW,QAAQ,WAAW,KAAK,KAAK,MAAM;AAC/E;AAEA,MAAM,YAAY,SAAyB;CACzC,IAAI,YAAY,MAAM,GAAG,MAAM,KAAK,OAAO;CAE3C,MAAM,QAAQ,YAAY,MAAM,GAAG;CAGnC,IAAI,UAAU,KAAA,KAAa,UAAU,MAAM,OAAO;CAGlD,MAAM,UAAU,YAAY,MAAM,GAAG;CAErC,IAAI,UAAU,SAAS,OAAO,GAAG,OAAO;CAExC,IAAI,UAAU,SAAS,KAAK,GAAG,OAAO;CAEtC,IAAI,UAAU,UAAU,KAAK,GAAG,OAAO,QAAQ,SAAS;CAExD,IAAI,iBAAiB,QAAQ,OAAO,SAAS,MAAM,QAAQ,CAAC,GAAG,OAAO,MAAM,YAAY;CAExF,IAAI,UAAU,SAAS,KAAK,KAAK,OAAO,SAAS,KAAK,GAAG,OAAO,OAAO,KAAK;CAE5E,MAAM,IAAI,oBAAoB;EAAE,QAAQ;EAAQ,SAAS;CAA2B,CAAC;AACvF;;;;;AAMA,MAAM,YAAY,MAAc,QAAQ,OAAO,sBAAsB;CACnE,IAAI,aAAa;CACjB,IAAI,QAAQ,SAAS;CAErB,KAAK,IAAI,QAAQ,GAAG,QAAQ,KAAK,QAAQ,SAAS,GAAG;EACnD,MAAM,YAAY,KAAK,WAAW,KAAK;EAGvC,IAAI,cAAc,IAAI,cAAc;EAEpC,IAAI,cAAc,MAAM,cAAc,MAAM,cAAc,MAAM,cAAc,IAAI;GAChF,QAAQ;GAER,IAAI,KAAK,SAAS,aAAa,IAAI,OAAO;EAC5C;CACF;CAEA,OAAO;EAAE;EAAO,QAAQ,KAAK,SAAS,cAAc,QAAQ,IAAI;CAAG;AACrE;;AAiBA,MAAM,gBACJ,QACA,QACA,aACG;CACH,IAAI,aAAa;CACjB,MAAM,SAAwB,CAAC;CAE/B,MAAM,UAAU,SAAiB;EAC/B,IAAI,KAAK,SAAS,SAAS,YAAY,MAAM,eAAe;EAE5D,cAAc,KAAK;EACnB,OAAO,KAAK,IAAI;CAClB;CAEA,MAAM,cAAc,SAAiB;EACnC,IAAI,KAAK,SAAS,SAAS,YAAY,MAAM,eAAe;EAE5D,MAAM,QAAQ,SAAS,MAAM,SAAS,UAAU;EAEhD,IAAI,MAAM,SAAS,SAAS,YAAY,MAAM,eAAe;EAE7D,OAAO,MAAM,QAAQ,IAAI,KAAK,WAAW,MAAK,MAAI,EAAE,KAAK,IAAI;CAC/D;CAEA,KAAK,MAAM,WAAW,QAAQ;EAC5B,MAAM,EAAE,MAAM,OAAO,UAAU;EAE/B,IAAI,UAAU,KAAA,GAAW;EAEzB,IAAI,OAAO,SAAS,GAAG,OAAO,MAAM;EAEpC,OAAO,IAAI;EACX,OAAO,IAAI;EACX,OAAO,IAAI;EAEX,IAAI,UAAU,KAAA,GAAW;EAEzB,KAAK,IAAI,MAAM,MAAM,MAAM,GAAG,OAAO,MAAM,IAAI,GAAG,OAAO,GAAG;GAC1D,IAAI,MAAM,MAAM,MAAM,GAAG,OAAO,IAAI;GAEpC,KAAK,IAAI,SAAS,MAAM,MAAM,GAAG,UAAU,MAAM,IAAI,GAAG,UAAU,GAAG;IACnE,IAAI,SAAS,MAAM,MAAM,GAAG,OAAO,GAAG;IAEtC,MAAM,OAAO,YAAY,OAAO,kBAAkB;KAAE,GAAG;KAAK,GAAG;IAAO,CAAC,CAAC;IAExE,IAAI,CAAC,UAAU,SAAS,IAAI,GAAG;KAC7B,WAAW,EAAE;KACb;IACF;IAEA,MAAM,OAAO,SAAS,IAAI;IAE1B,WAAW,WAAW;KAAE,OAAO;KAAS;KAAK;KAAQ;IAAK,CAAC,KAAK,IAAI;GACtE;EACF;CACF;CAEA,OAAO,OAAO,KAAK,EAAE;AACvB;;;;;;;;;;AAWA,MAAa,0BACX,UACA,QACA,UAA2B,CAAC,MACzB;CACH,IAAI,SAAS,WAAW,SAAS,OAAO,eAAe,MAAM,SAAS;CAEtE,IAAI,SAAS;CAEb,MAAM,SAAS,SAAS,WAAW,KAAK,SAAuB;EAC7D,MAAM,YAAY,YAAY,SAAS,QAAQ,IAAI;EACnD,MAAM,QAAQ,UAAU,SAAS,SAAS,IAAI,YAAY,KAAA;EAC1D,MAAM,MAAM,UAAU,KAAA,IAAY,KAAA,IAAY,YAAY,OAAO,MAAM;EAEvE,IAAI,QAAQ,KAAA,KAAa,CAAC,UAAU,SAAS,GAAG,GAAG,MAAM,aAAa;EAEtE,MAAM,QAAQ,QAAQ,KAAA,IAAY,KAAA,IAAY,eAAe,GAAG;EAEhE,IAAI,QAAQ,KAAA,KAAa,UAAU,KAAA,GAAW,MAAM,aAAa;EAEjE,UAAU,UAAU,KAAA,IAAY,IAAI,UAAU,KAAK;EAEnD,IAAI,CAAC,OAAO,cAAc,MAAM,KAAK,SAAS,OAAO,mBAAmB,MAAM,SAAS;EAEvF,OAAO;GAAE;GAAM;GAAO;EAAM;CAC9B,CAAC;CAED,MAAM,QAAQ,QAAQ,8BAAc,IAAI,IAAI;CAC5C,MAAM,YAAY,QAAQ,wBAAwB;CAElD,IAAI,MAAM,SAAS,KAAK,CAAC,WAAW,OAAO,aAAa,QAAQ,OAAO,qBAAqB;CAG5F,MAAM,SAAS,OAAO,wBAAwB;CAC9C,IAAI,QAAQ,SAAS,aAAa,QAAQ,MAAM,EAAE;CAClD,IAAI,UAAU;CACd,MAAM,0BAAU,IAAI,IAA6B;CAEjD,MAAM,aAAa,UAAwB;EACzC,MAAM,aAAa,MAAM,IAAI,MAAM,IAAI;EAEvC,IAAI,eAAe,KAAA,KAAa,MAAM,UAAU,KAAA,GAAW,OAAO,KAAA;EAElE,MAAM,WAAW,QAAQ,IAAI,MAAM,IAAI;EAEvC,IAAI,aAAa,KAAA,GAAW,OAAO;EAEnC,MAAM,SAAS,oBAAoB,YAAY,MAAM,KAAK;EAC1D,QAAQ,IAAI,MAAM,MAAM,MAAM;EAE9B,OAAO;CACT;CAEA,MAAM,SAAS,aAAa,QAAQ,SAAS,EAAE,OAAO,KAAK,QAAQ,WAAW;EAC5E,MAAM,OAAO,UAAU,KAAK,GAAG,GAAG,KAAK,MAAM;EAE7C,IAAI,SAAS,KAAA,KAAa,KAAK,WAAW,MAAM,OAAO,KAAA;EAEvD,MAAM,QAAQ,KAAK,SAAS,IAAI,OAAQ,KAAK,WAAW;EACxD,MAAM,cAAc,SAAS,IAAI,EAAE;EAInC,IAHwB,MAAM,UAAU,MAAM,SAAS,IAAI,IAAI,KAAK,KAAK,OAAO,SAG1D,cAAc,OAAO;GACzC,UAAU;GAEV;EACF;EAEA,MAAM,YAAY,MAAM,SAAS,IAAI,GAAG,MAAM,IAAI,KAAK,OAAO,KAAK,IAAI,KAAK,OAAO;EACnF,MAAM,QAAQ,SAAS,WAAW,cAAc,KAAK,EAAE,SAAS;EAEhE,IAAI,QAAQ,OAAO;GACjB,UAAU;GAEV;EACF;EAEA,SAAS;EAET,OAAO;CACT,CAAC;CAED,MAAM,UAAU,CACd,GAAI,UAAU,CAAC,iBAAiB,IAAI,CAAC,GACrC,GAAI,YAAY,CAAC,oBAAoB,IAAI,CAAC,CAC5C;CAEA,OAAO,QAAQ,WAAW,IAAI,SAAS,SAAS,wBAAwB,OAAO;AACjF"}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
//#region src/node/xlsx-workbook.d.ts
|
|
2
|
+
/**
|
|
3
|
+
* `xl/workbook.xml` is never handed to SheetJS as uploaded. SheetJS's workbook parser slices and
|
|
4
|
+
* decodes the whole prefix of the part at every `</definedName>` (quadratic), and walks every
|
|
5
|
+
* `<sheet>` its own tag pattern finds, resolving each through its `r:id`, so one worksheet can be
|
|
6
|
+
* parsed once per declaration. The extractor reads the sheet list itself, checks that its strict
|
|
7
|
+
* scan and SheetJS's tag grammar see the same sheets, and writes a minimal workbook: the sheets in
|
|
8
|
+
* order (name, `sheetId`, hidden state, a fresh `r:id`) and the 1904 date system.
|
|
9
|
+
*/
|
|
10
|
+
type WorkbookSheetDeclaration = {
|
|
11
|
+
/** The sheet name exactly as SheetJS reads it (`unescapexml(utf8read(name))`). */readonly name: string;
|
|
12
|
+
readonly sheetId: string;
|
|
13
|
+
readonly state: 'hidden' | 'veryHidden' | undefined; /** The `r:id` of the sheet's relationship in the uploaded workbook, entity-decoded. */
|
|
14
|
+
readonly relationshipId: string | undefined;
|
|
15
|
+
};
|
|
16
|
+
type WorkbookModel = {
|
|
17
|
+
readonly sheets: ReadonlyArray<WorkbookSheetDeclaration>;
|
|
18
|
+
readonly date1904: boolean;
|
|
19
|
+
};
|
|
20
|
+
declare const malformedWorkbookMessage = "XLSX workbook is malformed.";
|
|
21
|
+
/**
|
|
22
|
+
* The sheets of `xl/workbook.xml` and its date system. Throws `FileExtractionError` when the
|
|
23
|
+
* workbook declares more than `maxSheets` sheets (counted by either scan), when the strict scan
|
|
24
|
+
* and SheetJS's grammar disagree on the number or names of sheets, when a name is missing, holds
|
|
25
|
+
* CDATA, or repeats another ignoring case, or when there are no sheets.
|
|
26
|
+
*/
|
|
27
|
+
declare const readWorkbookModel: (bytes: Uint8Array, maxSheets: number) => WorkbookModel;
|
|
28
|
+
/** The generated workbook: sheet `n` has `r:id="rId<n>"`. */
|
|
29
|
+
declare const workbookXml: (model: WorkbookModel) => string;
|
|
30
|
+
//#endregion
|
|
31
|
+
export { WorkbookModel, WorkbookSheetDeclaration, malformedWorkbookMessage, readWorkbookModel, workbookXml };
|
|
32
|
+
//# sourceMappingURL=xlsx-workbook.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-workbook.d.mts","names":[],"sources":["../../src/node/xlsx-workbook.ts"],"mappings":";;AAuBA;;;;;;;KAAY,wBAAA;EAMa,2FAJd,IAAA;EAAA,SACA,OAAA;EAAA,SACA,KAAA;WAEA,cAAA;AAAA;AAAA,KAGC,aAAA;EAAA,SACD,MAAA,EAAQ,aAAa,CAAC,wBAAA;EAAA,SACtB,QAAA;AAAA;AAAA,cAGE,wBAAA;AAAb;;;;AAAqC;AAsBrC;AAtBA,cAsBa,iBAAA,GAAqB,KAAA,EAAO,UAAA,EAAY,SAAA,aAAoB,aA4DxE;;cAGY,WAAA,GAAe,KAAoB,EAAb,aAAa"}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { FileExtractionError } from "../errors.mjs";
|
|
2
|
+
import { officeDocumentRelationshipsNamespace, sheetJsAttributeEscape, sheetJsAttributeText, sheetJsTags, sheetJsXmlHeader, spreadsheetMainNamespace, stripSheetJsNamespace } from "./sheetjs-xml.mjs";
|
|
3
|
+
import { decodeXmlEntities, prefixedAttribute, rawXmlAttributes } from "./xml-text.mjs";
|
|
4
|
+
import { Buffer } from "node:buffer";
|
|
5
|
+
//#region src/node/xlsx-workbook.ts
|
|
6
|
+
const malformedWorkbookMessage = "XLSX workbook is malformed.";
|
|
7
|
+
const malformed = () => new FileExtractionError({
|
|
8
|
+
format: "xlsx",
|
|
9
|
+
message: malformedWorkbookMessage
|
|
10
|
+
});
|
|
11
|
+
const tooManySheets = () => new FileExtractionError({
|
|
12
|
+
format: "xlsx",
|
|
13
|
+
message: "XLSX exceeds the worksheet or cell-visit limit."
|
|
14
|
+
});
|
|
15
|
+
const strictSheetTag = /<(?:[\w.-]+:)?sheet(?=[\s/>])[^<>]*>/g;
|
|
16
|
+
const plainSheetId = /^[1-9]\d{0,8}$/;
|
|
17
|
+
/**
|
|
18
|
+
* The sheets of `xl/workbook.xml` and its date system. Throws `FileExtractionError` when the
|
|
19
|
+
* workbook declares more than `maxSheets` sheets (counted by either scan), when the strict scan
|
|
20
|
+
* and SheetJS's grammar disagree on the number or names of sheets, when a name is missing, holds
|
|
21
|
+
* CDATA, or repeats another ignoring case, or when there are no sheets.
|
|
22
|
+
*/
|
|
23
|
+
const readWorkbookModel = (bytes, maxSheets) => {
|
|
24
|
+
const text = Buffer.from(bytes.buffer, bytes.byteOffset, bytes.byteLength).toString("latin1");
|
|
25
|
+
const sheetJsSheets = [];
|
|
26
|
+
let date1904 = false;
|
|
27
|
+
for (const tag of sheetJsTags(text)) {
|
|
28
|
+
const head = stripSheetJsNamespace(tag.head);
|
|
29
|
+
if (head === "<sheet") {
|
|
30
|
+
sheetJsSheets.push(tag);
|
|
31
|
+
if (sheetJsSheets.length > maxSheets) throw tooManySheets();
|
|
32
|
+
} else if (head === "<workbookPr" || head === "<workbookPr/>") {
|
|
33
|
+
const value = tag.attributes.get("date1904");
|
|
34
|
+
if (value !== void 0) date1904 = value === "1" || value === "true";
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
const strictSheets = [];
|
|
38
|
+
for (const [tag] of text.matchAll(strictSheetTag)) {
|
|
39
|
+
strictSheets.push(rawXmlAttributes(tag));
|
|
40
|
+
if (strictSheets.length > maxSheets) throw tooManySheets();
|
|
41
|
+
}
|
|
42
|
+
if (sheetJsSheets.length === 0 || sheetJsSheets.length !== strictSheets.length) throw malformed();
|
|
43
|
+
const seen = /* @__PURE__ */ new Set();
|
|
44
|
+
return {
|
|
45
|
+
sheets: sheetJsSheets.map((tag, index) => {
|
|
46
|
+
const strict = strictSheets[index];
|
|
47
|
+
const rawName = tag.attributes.get("name");
|
|
48
|
+
if (strict === void 0 || rawName === void 0 || strict.get("name") !== rawName) throw malformed();
|
|
49
|
+
const name = sheetJsAttributeText(rawName);
|
|
50
|
+
if (name === void 0 || seen.has(name.toLowerCase())) throw malformed();
|
|
51
|
+
seen.add(name.toLowerCase());
|
|
52
|
+
const state = tag.attributes.get("state");
|
|
53
|
+
const sheetId = tag.attributes.get("sheetId");
|
|
54
|
+
const relationshipId = prefixedAttribute(strict, "id");
|
|
55
|
+
return {
|
|
56
|
+
name,
|
|
57
|
+
sheetId: sheetId !== void 0 && plainSheetId.test(sheetId) ? sheetId : `${index + 1}`,
|
|
58
|
+
state: state === "hidden" || state === "veryHidden" ? state : void 0,
|
|
59
|
+
relationshipId: relationshipId === void 0 ? void 0 : decodeXmlEntities(relationshipId)
|
|
60
|
+
};
|
|
61
|
+
}),
|
|
62
|
+
date1904
|
|
63
|
+
};
|
|
64
|
+
};
|
|
65
|
+
/** The generated workbook: sheet `n` has `r:id="rId<n>"`. */
|
|
66
|
+
const workbookXml = (model) => `${sheetJsXmlHeader}<workbook xmlns="${spreadsheetMainNamespace}" xmlns:r="${officeDocumentRelationshipsNamespace}">${model.date1904 ? "<workbookPr date1904=\"1\"/>" : ""}<sheets>${model.sheets.map((sheet, index) => `<sheet name="${sheetJsAttributeEscape(sheet.name)}" sheetId="${sheet.sheetId}"${sheet.state === void 0 ? "" : ` state="${sheet.state}"`} r:id="rId${index + 1}"/>`).join("")}</sheets></workbook>`;
|
|
67
|
+
//#endregion
|
|
68
|
+
export { malformedWorkbookMessage, readWorkbookModel, workbookXml };
|
|
69
|
+
|
|
70
|
+
//# sourceMappingURL=xlsx-workbook.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-workbook.mjs","names":[],"sources":["../../src/node/xlsx-workbook.ts"],"sourcesContent":["import { Buffer } from 'node:buffer'\nimport { FileExtractionError } from '../errors.ts'\nimport {\n sheetJsAttributeEscape,\n sheetJsAttributeText,\n sheetJsTags,\n stripSheetJsNamespace,\n sheetJsXmlHeader,\n spreadsheetMainNamespace,\n officeDocumentRelationshipsNamespace\n} from './sheetjs-xml.ts'\nimport type { SheetJsTag } from './sheetjs-xml.ts'\nimport { decodeXmlEntities, prefixedAttribute, rawXmlAttributes } from './xml-text.ts'\n\n/**\n * `xl/workbook.xml` is never handed to SheetJS as uploaded. SheetJS's workbook parser slices and\n * decodes the whole prefix of the part at every `</definedName>` (quadratic), and walks every\n * `<sheet>` its own tag pattern finds, resolving each through its `r:id`, so one worksheet can be\n * parsed once per declaration. The extractor reads the sheet list itself, checks that its strict\n * scan and SheetJS's tag grammar see the same sheets, and writes a minimal workbook: the sheets in\n * order (name, `sheetId`, hidden state, a fresh `r:id`) and the 1904 date system.\n */\n\nexport type WorkbookSheetDeclaration = {\n /** The sheet name exactly as SheetJS reads it (`unescapexml(utf8read(name))`). */\n readonly name: string\n readonly sheetId: string\n readonly state: 'hidden' | 'veryHidden' | undefined\n /** The `r:id` of the sheet's relationship in the uploaded workbook, entity-decoded. */\n readonly relationshipId: string | undefined\n}\n\nexport type WorkbookModel = {\n readonly sheets: ReadonlyArray<WorkbookSheetDeclaration>\n readonly date1904: boolean\n}\n\nexport const malformedWorkbookMessage = 'XLSX workbook is malformed.'\n\nconst malformed = () =>\n new FileExtractionError({ format: 'xlsx', message: malformedWorkbookMessage })\n\nconst tooManySheets = () =>\n new FileExtractionError({\n format: 'xlsx',\n message: 'XLSX exceeds the worksheet or cell-visit limit.'\n })\n\n// `[^<>]` keeps each candidate inside one tag, so the strict scan stays linear.\nconst strictSheetTag = /<(?:[\\w.-]+:)?sheet(?=[\\s/>])[^<>]*>/g\n\nconst plainSheetId = /^[1-9]\\d{0,8}$/\n\n/**\n * The sheets of `xl/workbook.xml` and its date system. Throws `FileExtractionError` when the\n * workbook declares more than `maxSheets` sheets (counted by either scan), when the strict scan\n * and SheetJS's grammar disagree on the number or names of sheets, when a name is missing, holds\n * CDATA, or repeats another ignoring case, or when there are no sheets.\n */\nexport const readWorkbookModel = (bytes: Uint8Array, maxSheets: number): WorkbookModel => {\n // SheetJS parses the Latin-1 (\"binary\") view and decodes attribute text as UTF-8 afterwards.\n const text = Buffer.from(bytes.buffer, bytes.byteOffset, bytes.byteLength).toString('latin1')\n const sheetJsSheets: Array<SheetJsTag> = []\n let date1904 = false\n\n for (const tag of sheetJsTags(text)) {\n const head = stripSheetJsNamespace(tag.head)\n\n if (head === '<sheet') {\n sheetJsSheets.push(tag)\n\n if (sheetJsSheets.length > maxSheets) throw tooManySheets()\n } else if (head === '<workbookPr' || head === '<workbookPr/>') {\n // Each `workbookPr` that carries the attribute overrides the last (SheetJS `parsexmlbool`).\n const value = tag.attributes.get('date1904')\n\n if (value !== undefined) date1904 = value === '1' || value === 'true'\n }\n }\n\n const strictSheets: Array<ReadonlyMap<string, string>> = []\n\n for (const [tag] of text.matchAll(strictSheetTag)) {\n strictSheets.push(rawXmlAttributes(tag))\n\n if (strictSheets.length > maxSheets) throw tooManySheets()\n }\n\n if (sheetJsSheets.length === 0 || sheetJsSheets.length !== strictSheets.length) throw malformed()\n\n const seen = new Set<string>()\n\n const sheets = sheetJsSheets.map((tag, index): WorkbookSheetDeclaration => {\n const strict = strictSheets[index]\n const rawName = tag.attributes.get('name')\n\n if (strict === undefined || rawName === undefined || strict.get('name') !== rawName)\n throw malformed()\n\n const name = sheetJsAttributeText(rawName)\n\n // Excel compares sheet names ignoring case.\n if (name === undefined || seen.has(name.toLowerCase())) throw malformed()\n\n seen.add(name.toLowerCase())\n\n const state = tag.attributes.get('state')\n const sheetId = tag.attributes.get('sheetId')\n const relationshipId = prefixedAttribute(strict, 'id')\n\n return {\n name,\n sheetId: sheetId !== undefined && plainSheetId.test(sheetId) ? sheetId : `${index + 1}`,\n state: state === 'hidden' || state === 'veryHidden' ? state : undefined,\n relationshipId: relationshipId === undefined ? undefined : decodeXmlEntities(relationshipId)\n }\n })\n\n return { sheets, date1904 }\n}\n\n/** The generated workbook: sheet `n` has `r:id=\"rId<n>\"`. */\nexport const workbookXml = (model: WorkbookModel) =>\n `${sheetJsXmlHeader}<workbook xmlns=\"${spreadsheetMainNamespace}\" xmlns:r=\"${officeDocumentRelationshipsNamespace}\">${\n model.date1904 ? '<workbookPr date1904=\"1\"/>' : ''\n }<sheets>${model.sheets\n .map(\n (sheet, index) =>\n `<sheet name=\"${sheetJsAttributeEscape(sheet.name)}\" sheetId=\"${sheet.sheetId}\"${\n sheet.state === undefined ? '' : ` state=\"${sheet.state}\"`\n } r:id=\"rId${index + 1}\"/>`\n )\n .join('')}</sheets></workbook>`\n"],"mappings":";;;;;AAqCA,MAAa,2BAA2B;AAExC,MAAM,kBACJ,IAAI,oBAAoB;CAAE,QAAQ;CAAQ,SAAS;AAAyB,CAAC;AAE/E,MAAM,sBACJ,IAAI,oBAAoB;CACtB,QAAQ;CACR,SAAS;AACX,CAAC;AAGH,MAAM,iBAAiB;AAEvB,MAAM,eAAe;;;;;;;AAQrB,MAAa,qBAAqB,OAAmB,cAAqC;CAExF,MAAM,OAAO,OAAO,KAAK,MAAM,QAAQ,MAAM,YAAY,MAAM,UAAU,EAAE,SAAS,QAAQ;CAC5F,MAAM,gBAAmC,CAAC;CAC1C,IAAI,WAAW;CAEf,KAAK,MAAM,OAAO,YAAY,IAAI,GAAG;EACnC,MAAM,OAAO,sBAAsB,IAAI,IAAI;EAE3C,IAAI,SAAS,UAAU;GACrB,cAAc,KAAK,GAAG;GAEtB,IAAI,cAAc,SAAS,WAAW,MAAM,cAAc;EAC5D,OAAO,IAAI,SAAS,iBAAiB,SAAS,iBAAiB;GAE7D,MAAM,QAAQ,IAAI,WAAW,IAAI,UAAU;GAE3C,IAAI,UAAU,KAAA,GAAW,WAAW,UAAU,OAAO,UAAU;EACjE;CACF;CAEA,MAAM,eAAmD,CAAC;CAE1D,KAAK,MAAM,CAAC,QAAQ,KAAK,SAAS,cAAc,GAAG;EACjD,aAAa,KAAK,iBAAiB,GAAG,CAAC;EAEvC,IAAI,aAAa,SAAS,WAAW,MAAM,cAAc;CAC3D;CAEA,IAAI,cAAc,WAAW,KAAK,cAAc,WAAW,aAAa,QAAQ,MAAM,UAAU;CAEhG,MAAM,uBAAO,IAAI,IAAY;CA4B7B,OAAO;EAAE,QA1BM,cAAc,KAAK,KAAK,UAAoC;GACzE,MAAM,SAAS,aAAa;GAC5B,MAAM,UAAU,IAAI,WAAW,IAAI,MAAM;GAEzC,IAAI,WAAW,KAAA,KAAa,YAAY,KAAA,KAAa,OAAO,IAAI,MAAM,MAAM,SAC1E,MAAM,UAAU;GAElB,MAAM,OAAO,qBAAqB,OAAO;GAGzC,IAAI,SAAS,KAAA,KAAa,KAAK,IAAI,KAAK,YAAY,CAAC,GAAG,MAAM,UAAU;GAExE,KAAK,IAAI,KAAK,YAAY,CAAC;GAE3B,MAAM,QAAQ,IAAI,WAAW,IAAI,OAAO;GACxC,MAAM,UAAU,IAAI,WAAW,IAAI,SAAS;GAC5C,MAAM,iBAAiB,kBAAkB,QAAQ,IAAI;GAErD,OAAO;IACL;IACA,SAAS,YAAY,KAAA,KAAa,aAAa,KAAK,OAAO,IAAI,UAAU,GAAG,QAAQ;IACpF,OAAO,UAAU,YAAY,UAAU,eAAe,QAAQ,KAAA;IAC9D,gBAAgB,mBAAmB,KAAA,IAAY,KAAA,IAAY,kBAAkB,cAAc;GAC7F;EACF,CAEc;EAAG;CAAS;AAC5B;;AAGA,MAAa,eAAe,UAC1B,GAAG,iBAAiB,mBAAmB,yBAAyB,aAAa,qCAAqC,IAChH,MAAM,WAAW,iCAA+B,GACjD,UAAU,MAAM,OACd,KACE,OAAO,UACN,gBAAgB,uBAAuB,MAAM,IAAI,EAAE,aAAa,MAAM,QAAQ,GAC5E,MAAM,UAAU,KAAA,IAAY,KAAK,WAAW,MAAM,MAAM,GACzD,YAAY,QAAQ,EAAE,IAC3B,EACC,KAAK,EAAE,EAAE"}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
//#region src/node/xml-text.d.ts
|
|
2
|
+
/** Decode the predefined XML entities and numeric character references; keep unknown ones. */
|
|
3
|
+
declare const decodeXmlEntities: (text: string) => string;
|
|
4
|
+
/** Attributes of one start tag as written (not decoded), keyed by qualified name; first wins. */
|
|
5
|
+
declare const rawXmlAttributes: (tag: string) => ReadonlyMap<string, string>;
|
|
6
|
+
/** Attributes of one start tag, entity-decoded, keyed by their qualified name. */
|
|
7
|
+
declare const xmlAttributes: (tag: string) => ReadonlyMap<string, string>;
|
|
8
|
+
/** The value of a namespace-prefixed attribute such as `r:id`, whatever the prefix. */
|
|
9
|
+
declare const prefixedAttribute: (attributes: ReadonlyMap<string, string>, localName: string) => string | undefined;
|
|
10
|
+
//#endregion
|
|
11
|
+
export { decodeXmlEntities, prefixedAttribute, rawXmlAttributes, xmlAttributes };
|
|
12
|
+
//# sourceMappingURL=xml-text.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xml-text.d.mts","names":[],"sources":["../../src/node/xml-text.ts"],"mappings":";;cAuCa,iBAAA,GAAqB,IAAY;;cAQjC,gBAAA,GAAoB,GAAA,aAAc,WAAW;;cAe7C,aAAA,GAAiB,GAAA,aAAc,WAAW;AAfvD;AAAA,cAqBa,iBAAA,GAAqB,UAAA,EAAY,WAAW,kBAAkB,SAAA"}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
//#region src/node/xml-text.ts
|
|
2
|
+
const xmlEntity = /&([^;&\s]{1,16});/g;
|
|
3
|
+
const hexEntity = /^#x([0-9a-fA-F]+)$/;
|
|
4
|
+
const decimalEntity = /^#(\d+)$/;
|
|
5
|
+
const namedEntities = new Map([
|
|
6
|
+
["amp", "&"],
|
|
7
|
+
["lt", "<"],
|
|
8
|
+
["gt", ">"],
|
|
9
|
+
["quot", "\""],
|
|
10
|
+
["apos", "'"]
|
|
11
|
+
]);
|
|
12
|
+
const decodeCodePoint = (raw, codePointText, radix) => {
|
|
13
|
+
const codePoint = Number.parseInt(codePointText, radix);
|
|
14
|
+
if (!Number.isInteger(codePoint) || codePoint < 0 || codePoint > 1114111) return raw;
|
|
15
|
+
return String.fromCodePoint(codePoint);
|
|
16
|
+
};
|
|
17
|
+
const decodeXmlEntity = (raw, entity) => {
|
|
18
|
+
const named = namedEntities.get(entity);
|
|
19
|
+
if (named !== void 0) return named;
|
|
20
|
+
const hex = hexEntity.exec(entity)?.[1];
|
|
21
|
+
if (hex !== void 0) return decodeCodePoint(raw, hex, 16);
|
|
22
|
+
const decimal = decimalEntity.exec(entity)?.[1];
|
|
23
|
+
if (decimal !== void 0) return decodeCodePoint(raw, decimal, 10);
|
|
24
|
+
return raw;
|
|
25
|
+
};
|
|
26
|
+
/** Decode the predefined XML entities and numeric character references; keep unknown ones. */
|
|
27
|
+
const decodeXmlEntities = (text) => text.replace(xmlEntity, (raw, entity) => decodeXmlEntity(raw, entity));
|
|
28
|
+
const attributePattern = /(?<![\w.:-])([\w.:-]+)\s*=\s*(?:"([^"]*)"|'([^']*)')/g;
|
|
29
|
+
/** Attributes of one start tag as written (not decoded), keyed by qualified name; first wins. */
|
|
30
|
+
const rawXmlAttributes = (tag) => {
|
|
31
|
+
const attributes = /* @__PURE__ */ new Map();
|
|
32
|
+
for (const match of tag.matchAll(attributePattern)) {
|
|
33
|
+
const name = match[1];
|
|
34
|
+
const value = match[2] ?? match[3];
|
|
35
|
+
if (name !== void 0 && value !== void 0 && !attributes.has(name)) attributes.set(name, value);
|
|
36
|
+
}
|
|
37
|
+
return attributes;
|
|
38
|
+
};
|
|
39
|
+
/** Attributes of one start tag, entity-decoded, keyed by their qualified name. */
|
|
40
|
+
const xmlAttributes = (tag) => new Map(Array.from(rawXmlAttributes(tag), ([name, value]) => [name, decodeXmlEntities(value)]));
|
|
41
|
+
/** The value of a namespace-prefixed attribute such as `r:id`, whatever the prefix. */
|
|
42
|
+
const prefixedAttribute = (attributes, localName) => {
|
|
43
|
+
for (const [name, value] of attributes) {
|
|
44
|
+
const separator = name.indexOf(":");
|
|
45
|
+
if (separator > 0 && name.slice(separator + 1) === localName) return value;
|
|
46
|
+
}
|
|
47
|
+
};
|
|
48
|
+
//#endregion
|
|
49
|
+
export { decodeXmlEntities, prefixedAttribute, rawXmlAttributes, xmlAttributes };
|
|
50
|
+
|
|
51
|
+
//# sourceMappingURL=xml-text.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xml-text.mjs","names":[],"sources":["../../src/node/xml-text.ts"],"sourcesContent":["const xmlEntity = /&([^;&\\s]{1,16});/g\n\nconst hexEntity = /^#x([0-9a-fA-F]+)$/\n\nconst decimalEntity = /^#(\\d+)$/\n\nconst namedEntities: ReadonlyMap<string, string> = new Map([\n ['amp', '&'],\n ['lt', '<'],\n ['gt', '>'],\n ['quot', '\"'],\n ['apos', \"'\"]\n])\n\nconst decodeCodePoint = (raw: string, codePointText: string, radix: number) => {\n const codePoint = Number.parseInt(codePointText, radix)\n\n if (!Number.isInteger(codePoint) || codePoint < 0 || codePoint > 0x10ffff) return raw\n\n return String.fromCodePoint(codePoint)\n}\n\nconst decodeXmlEntity = (raw: string, entity: string) => {\n const named = namedEntities.get(entity)\n\n if (named !== undefined) return named\n\n const hex = hexEntity.exec(entity)?.[1]\n\n if (hex !== undefined) return decodeCodePoint(raw, hex, 16)\n\n const decimal = decimalEntity.exec(entity)?.[1]\n\n if (decimal !== undefined) return decodeCodePoint(raw, decimal, 10)\n\n return raw\n}\n\n/** Decode the predefined XML entities and numeric character references; keep unknown ones. */\nexport const decodeXmlEntities = (text: string) =>\n text.replace(xmlEntity, (raw, entity: string) => decodeXmlEntity(raw, entity))\n\n// A name may only start after a non-name character, so a long run of name characters is scanned\n// once rather than once per starting position.\nconst attributePattern = /(?<![\\w.:-])([\\w.:-]+)\\s*=\\s*(?:\"([^\"]*)\"|'([^']*)')/g\n\n/** Attributes of one start tag as written (not decoded), keyed by qualified name; first wins. */\nexport const rawXmlAttributes = (tag: string): ReadonlyMap<string, string> => {\n const attributes = new Map<string, string>()\n\n for (const match of tag.matchAll(attributePattern)) {\n const name = match[1]\n const value = match[2] ?? match[3]\n\n if (name !== undefined && value !== undefined && !attributes.has(name))\n attributes.set(name, value)\n }\n\n return attributes\n}\n\n/** Attributes of one start tag, entity-decoded, keyed by their qualified name. */\nexport const xmlAttributes = (tag: string): ReadonlyMap<string, string> =>\n new Map(\n Array.from(rawXmlAttributes(tag), ([name, value]) => [name, decodeXmlEntities(value)] as const)\n )\n\n/** The value of a namespace-prefixed attribute such as `r:id`, whatever the prefix. */\nexport const prefixedAttribute = (attributes: ReadonlyMap<string, string>, localName: string) => {\n for (const [name, value] of attributes) {\n const separator = name.indexOf(':')\n\n if (separator > 0 && name.slice(separator + 1) === localName) return value\n }\n\n return undefined\n}\n"],"mappings":";AAAA,MAAM,YAAY;AAElB,MAAM,YAAY;AAElB,MAAM,gBAAgB;AAEtB,MAAM,gBAA6C,IAAI,IAAI;CACzD,CAAC,OAAO,GAAG;CACX,CAAC,MAAM,GAAG;CACV,CAAC,MAAM,GAAG;CACV,CAAC,QAAQ,IAAG;CACZ,CAAC,QAAQ,GAAG;AACd,CAAC;AAED,MAAM,mBAAmB,KAAa,eAAuB,UAAkB;CAC7E,MAAM,YAAY,OAAO,SAAS,eAAe,KAAK;CAEtD,IAAI,CAAC,OAAO,UAAU,SAAS,KAAK,YAAY,KAAK,YAAY,SAAU,OAAO;CAElF,OAAO,OAAO,cAAc,SAAS;AACvC;AAEA,MAAM,mBAAmB,KAAa,WAAmB;CACvD,MAAM,QAAQ,cAAc,IAAI,MAAM;CAEtC,IAAI,UAAU,KAAA,GAAW,OAAO;CAEhC,MAAM,MAAM,UAAU,KAAK,MAAM,IAAI;CAErC,IAAI,QAAQ,KAAA,GAAW,OAAO,gBAAgB,KAAK,KAAK,EAAE;CAE1D,MAAM,UAAU,cAAc,KAAK,MAAM,IAAI;CAE7C,IAAI,YAAY,KAAA,GAAW,OAAO,gBAAgB,KAAK,SAAS,EAAE;CAElE,OAAO;AACT;;AAGA,MAAa,qBAAqB,SAChC,KAAK,QAAQ,YAAY,KAAK,WAAmB,gBAAgB,KAAK,MAAM,CAAC;AAI/E,MAAM,mBAAmB;;AAGzB,MAAa,oBAAoB,QAA6C;CAC5E,MAAM,6BAAa,IAAI,IAAoB;CAE3C,KAAK,MAAM,SAAS,IAAI,SAAS,gBAAgB,GAAG;EAClD,MAAM,OAAO,MAAM;EACnB,MAAM,QAAQ,MAAM,MAAM,MAAM;EAEhC,IAAI,SAAS,KAAA,KAAa,UAAU,KAAA,KAAa,CAAC,WAAW,IAAI,IAAI,GACnE,WAAW,IAAI,MAAM,KAAK;CAC9B;CAEA,OAAO;AACT;;AAGA,MAAa,iBAAiB,QAC5B,IAAI,IACF,MAAM,KAAK,iBAAiB,GAAG,IAAI,CAAC,MAAM,WAAW,CAAC,MAAM,kBAAkB,KAAK,CAAC,CAAU,CAChG;;AAGF,MAAa,qBAAqB,YAAyC,cAAsB;CAC/F,KAAK,MAAM,CAAC,MAAM,UAAU,YAAY;EACtC,MAAM,YAAY,KAAK,QAAQ,GAAG;EAElC,IAAI,YAAY,KAAK,KAAK,MAAM,YAAY,CAAC,MAAM,WAAW,OAAO;CACvE;AAGF"}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
//#region src/sanitize.d.ts
|
|
2
|
+
/** Normalize line endings, drop control characters, and collapse layout noise. */
|
|
3
|
+
declare const sanitizeExtractedText: (text: string) => string;
|
|
4
|
+
//#endregion
|
|
5
|
+
export { sanitizeExtractedText };
|
|
6
|
+
//# sourceMappingURL=sanitize.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sanitize.d.mts","names":[],"sources":["../src/sanitize.ts"],"mappings":";;cASa,qBAAA,GAAyB,IAAY"}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
//#region src/sanitize.ts
|
|
2
|
+
const nonPrintableCharacters = /[\u0000-\u0008\u000B\u000C\u000E-\u001F\u007F]/g;
|
|
3
|
+
const longDotRuns = /\.{4,}/g;
|
|
4
|
+
const horizontalWhitespaceRuns = /[\t ]{2,}/g;
|
|
5
|
+
const blankLineRuns = /\n{3,}/g;
|
|
6
|
+
/** Normalize line endings, drop control characters, and collapse layout noise. */
|
|
7
|
+
const sanitizeExtractedText = (text) => text.replaceAll("\r\n", "\n").replaceAll("\r", "\n").replace(nonPrintableCharacters, "").replace(longDotRuns, "…").replace(horizontalWhitespaceRuns, " ").replace(blankLineRuns, "\n\n").trim();
|
|
8
|
+
//#endregion
|
|
9
|
+
export { sanitizeExtractedText };
|
|
10
|
+
|
|
11
|
+
//# sourceMappingURL=sanitize.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sanitize.mjs","names":[],"sources":["../src/sanitize.ts"],"sourcesContent":["const nonPrintableCharacters = /[\\u0000-\\u0008\\u000B\\u000C\\u000E-\\u001F\\u007F]/g\n\nconst longDotRuns = /\\.{4,}/g\n\nconst horizontalWhitespaceRuns = /[\\t ]{2,}/g\n\nconst blankLineRuns = /\\n{3,}/g\n\n/** Normalize line endings, drop control characters, and collapse layout noise. */\nexport const sanitizeExtractedText = (text: string) =>\n text\n .replaceAll('\\r\\n', '\\n')\n .replaceAll('\\r', '\\n')\n .replace(nonPrintableCharacters, '')\n .replace(longDotRuns, '…')\n .replace(horizontalWhitespaceRuns, ' ')\n .replace(blankLineRuns, '\\n\\n')\n .trim()\n"],"mappings":";AAAA,MAAM,yBAAyB;AAE/B,MAAM,cAAc;AAEpB,MAAM,2BAA2B;AAEjC,MAAM,gBAAgB;;AAGtB,MAAa,yBAAyB,SACpC,KACG,WAAW,QAAQ,IAAI,EACvB,WAAW,MAAM,IAAI,EACrB,QAAQ,wBAAwB,EAAE,EAClC,QAAQ,aAAa,GAAG,EACxB,QAAQ,0BAA0B,GAAG,EACrC,QAAQ,eAAe,MAAM,EAC7B,KAAK"}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import { FileExtractorError } from "./errors.mjs";
|
|
2
|
+
import { ExtractedFile, FileInput } from "./format.mjs";
|
|
3
|
+
import { Context, Effect } from "effect";
|
|
4
|
+
|
|
5
|
+
//#region src/service.d.ts
|
|
6
|
+
type FileExtractorApi = {
|
|
7
|
+
/**
|
|
8
|
+
* Extract sanitized text from one file. Fails with `UnsupportedFileFormatError` for unknown
|
|
9
|
+
* formats, `FileExtractionError` for unreadable, oversized, or empty files, and
|
|
10
|
+
* `SheetJsUnavailableError` when an XLSX file arrives but SheetJS 0.20.3+ is not installed.
|
|
11
|
+
*/
|
|
12
|
+
readonly extract: (input: FileInput) => Effect.Effect<ExtractedFile, FileExtractorError>;
|
|
13
|
+
};
|
|
14
|
+
declare const FileExtractor_base: Context.ServiceClass<FileExtractor, "@yolk-sdk/extractors/FileExtractor", FileExtractorApi>;
|
|
15
|
+
/**
|
|
16
|
+
* The file extractor service. The tag is runtime-portable; the Node implementation lives in
|
|
17
|
+
* `@yolk-sdk/extractors/node`.
|
|
18
|
+
*/
|
|
19
|
+
declare class FileExtractor extends FileExtractor_base {}
|
|
20
|
+
//#endregion
|
|
21
|
+
export { FileExtractor, FileExtractorApi };
|
|
22
|
+
//# sourceMappingURL=service.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"service.d.mts","names":[],"sources":["../src/service.ts"],"mappings":";;;;;KAKY,gBAAA;EAAA;;;;;EAAA,SAMD,OAAA,GAAU,KAAA,EAAO,SAAA,KAAc,MAAA,CAAO,MAAA,CAAO,aAAA,EAAe,kBAAA;AAAA;AAAA,cACtE,kBAAA;;;;;cAMY,aAAA,SAAsB,kBAElC"}
|
package/dist/service.mjs
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import { Context } from "effect";
|
|
2
|
+
//#region src/service.ts
|
|
3
|
+
/**
|
|
4
|
+
* The file extractor service. The tag is runtime-portable; the Node implementation lives in
|
|
5
|
+
* `@yolk-sdk/extractors/node`.
|
|
6
|
+
*/
|
|
7
|
+
var FileExtractor = class extends Context.Service()("@yolk-sdk/extractors/FileExtractor") {};
|
|
8
|
+
//#endregion
|
|
9
|
+
export { FileExtractor };
|
|
10
|
+
|
|
11
|
+
//# sourceMappingURL=service.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"service.mjs","names":[],"sources":["../src/service.ts"],"sourcesContent":["import { Context } from 'effect'\nimport type { Effect } from 'effect'\nimport type { FileExtractorError } from './errors.ts'\nimport type { ExtractedFile, FileInput } from './format.ts'\n\nexport type FileExtractorApi = {\n /**\n * Extract sanitized text from one file. Fails with `UnsupportedFileFormatError` for unknown\n * formats, `FileExtractionError` for unreadable, oversized, or empty files, and\n * `SheetJsUnavailableError` when an XLSX file arrives but SheetJS 0.20.3+ is not installed.\n */\n readonly extract: (input: FileInput) => Effect.Effect<ExtractedFile, FileExtractorError>\n}\n\n/**\n * The file extractor service. The tag is runtime-portable; the Node implementation lives in\n * `@yolk-sdk/extractors/node`.\n */\nexport class FileExtractor extends Context.Service<FileExtractor, FileExtractorApi>()(\n '@yolk-sdk/extractors/FileExtractor'\n) {}\n"],"mappings":";;;;;;AAkBA,IAAa,gBAAb,cAAmC,QAAQ,QAAyC,EAClF,oCACF,EAAE,CAAC"}
|
package/package.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@yolk-sdk/extractors",
|
|
3
|
+
"version": "0.1.0-canary.98",
|
|
4
|
+
"description": "Bounded text extraction for PDF, DOCX, XLSX, PPTX, and text files: an Effect FileExtractor service with Office archive validation, hyperlink-safe XLSX text, and a knowledge extractor adapter.",
|
|
5
|
+
"license": "MIT",
|
|
6
|
+
"type": "module",
|
|
7
|
+
"sideEffects": false,
|
|
8
|
+
"repository": {
|
|
9
|
+
"type": "git",
|
|
10
|
+
"url": "git+https://github.com/magoz/yolk-sdk.git",
|
|
11
|
+
"directory": "packages/extractors"
|
|
12
|
+
},
|
|
13
|
+
"bugs": {
|
|
14
|
+
"url": "https://github.com/magoz/yolk-sdk/issues"
|
|
15
|
+
},
|
|
16
|
+
"homepage": "https://github.com/magoz/yolk-sdk#readme",
|
|
17
|
+
"keywords": [
|
|
18
|
+
"extraction",
|
|
19
|
+
"pdf",
|
|
20
|
+
"docx",
|
|
21
|
+
"xlsx",
|
|
22
|
+
"pptx",
|
|
23
|
+
"knowledge",
|
|
24
|
+
"effect"
|
|
25
|
+
],
|
|
26
|
+
"engines": {
|
|
27
|
+
"node": ">=22"
|
|
28
|
+
},
|
|
29
|
+
"exports": {
|
|
30
|
+
"./package.json": "./package.json",
|
|
31
|
+
".": {
|
|
32
|
+
"types": "./dist/index.d.mts",
|
|
33
|
+
"import": "./dist/index.mjs",
|
|
34
|
+
"default": "./dist/index.mjs"
|
|
35
|
+
},
|
|
36
|
+
"./knowledge": {
|
|
37
|
+
"types": "./dist/knowledge.d.mts",
|
|
38
|
+
"import": "./dist/knowledge.mjs",
|
|
39
|
+
"default": "./dist/knowledge.mjs"
|
|
40
|
+
},
|
|
41
|
+
"./node": {
|
|
42
|
+
"types": "./dist/node/index.d.mts",
|
|
43
|
+
"import": "./dist/node/index.mjs",
|
|
44
|
+
"default": "./dist/node/index.mjs"
|
|
45
|
+
},
|
|
46
|
+
"./node/extraction-worker": {
|
|
47
|
+
"types": "./dist/node/extraction-worker.d.mts",
|
|
48
|
+
"import": "./dist/node/extraction-worker.mjs",
|
|
49
|
+
"default": "./dist/node/extraction-worker.mjs"
|
|
50
|
+
}
|
|
51
|
+
},
|
|
52
|
+
"files": [
|
|
53
|
+
"src/**/*.ts",
|
|
54
|
+
"!src/**/*.test.ts",
|
|
55
|
+
"!src/**/*.test.tsx",
|
|
56
|
+
"dist/**/*",
|
|
57
|
+
"README.md"
|
|
58
|
+
],
|
|
59
|
+
"publishConfig": {
|
|
60
|
+
"access": "public",
|
|
61
|
+
"provenance": true
|
|
62
|
+
},
|
|
63
|
+
"dependencies": {
|
|
64
|
+
"effect": "4.0.0",
|
|
65
|
+
"fflate": "^0.8.3",
|
|
66
|
+
"mammoth": "^1.13.0",
|
|
67
|
+
"unpdf": "^1.8.1",
|
|
68
|
+
"@yolk-sdk/knowledge": "^0.1.0-canary.98"
|
|
69
|
+
},
|
|
70
|
+
"peerDependencies": {
|
|
71
|
+
"xlsx": ">=0.20.3"
|
|
72
|
+
},
|
|
73
|
+
"peerDependenciesMeta": {
|
|
74
|
+
"xlsx": {
|
|
75
|
+
"optional": true
|
|
76
|
+
}
|
|
77
|
+
},
|
|
78
|
+
"devDependencies": {
|
|
79
|
+
"xlsx": "https://cdn.sheetjs.com/xlsx-0.20.3/xlsx-0.20.3.tgz"
|
|
80
|
+
},
|
|
81
|
+
"scripts": {
|
|
82
|
+
"build": "tsdown",
|
|
83
|
+
"check": "tsc -p tsconfig.json --noEmit",
|
|
84
|
+
"test": "vitest run --passWithNoTests",
|
|
85
|
+
"test:run": "vitest run --passWithNoTests"
|
|
86
|
+
}
|
|
87
|
+
}
|