@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
import { sheetJsTags, sheetJsTextViews } from "./sheetjs-xml.mjs";
|
|
2
|
+
//#region src/node/xlsx-routing.ts
|
|
3
|
+
/**
|
|
4
|
+
* Early, exact rejection of XLSX input that SheetJS 0.20.3 would hand to its ODS, Numbers, or
|
|
5
|
+
* binary (XLSB) parsers, so users get a clear error instead of "Could not read XLSX". These checks
|
|
6
|
+
* are not the security guarantee: SheetJS only ever receives the allowlisted archive built by
|
|
7
|
+
* `buildSheetJsInput` (`xlsx-sheetjs-input.ts`), which contains no marker entries and no `.bin`
|
|
8
|
+
* parts whatever these checks decide.
|
|
9
|
+
*/
|
|
10
|
+
/**
|
|
11
|
+
* An entry name as SheetJS looks it up: its ZIP reader (`cfb_add`) keeps a name that already
|
|
12
|
+
* starts with `Root Entry/` and otherwise stores `"Root Entry/" + name` with the first `//`
|
|
13
|
+
* collapsed; `safegetzipfile` strips `Root Entry/`, treats `\` and `/` alike, and compares
|
|
14
|
+
* lower-cased names.
|
|
15
|
+
*/
|
|
16
|
+
const sheetJsEntryPath = (name) => (name.startsWith("Root Entry/") ? name : `Root Entry/${name}`.replace("//", "/")).replace(/^Root Entry\//, "").replaceAll("\\", "/").toLowerCase();
|
|
17
|
+
/** Entries `parse_zip` checks before content types (SheetJS-normalized paths). */
|
|
18
|
+
const alternateFormatEntries = new Set([
|
|
19
|
+
"meta-inf/manifest.xml",
|
|
20
|
+
"objectdata.xml",
|
|
21
|
+
"index/document.iwa"
|
|
22
|
+
]);
|
|
23
|
+
/**
|
|
24
|
+
* Whether SheetJS would treat this entry as an ODS, UOC, or Numbers marker. `CFB.find` matches
|
|
25
|
+
* `Index.zip` by base name, and any `Root Entry/` name (any case) is rejected outright.
|
|
26
|
+
*/
|
|
27
|
+
const isAlternateFormatEntry = (name) => {
|
|
28
|
+
const path = sheetJsEntryPath(name);
|
|
29
|
+
return name.toLowerCase().startsWith("root entry/") || alternateFormatEntries.has(path) || path.slice(path.lastIndexOf("/") + 1) === "index.zip";
|
|
30
|
+
};
|
|
31
|
+
/** Whether any tag SheetJS would parse from any view of `content` satisfies `test`. */
|
|
32
|
+
const someSheetJsTag = (content, test) => sheetJsTextViews(content).some((text) => {
|
|
33
|
+
for (const tag of sheetJsTags(text)) if (test(tag)) return true;
|
|
34
|
+
return false;
|
|
35
|
+
});
|
|
36
|
+
const namedEntities = new Map([
|
|
37
|
+
["quot", "\""],
|
|
38
|
+
["apos", "'"],
|
|
39
|
+
["gt", ">"],
|
|
40
|
+
["lt", "<"],
|
|
41
|
+
["amp", "&"]
|
|
42
|
+
]);
|
|
43
|
+
/** SheetJS `unescapexml`: XML entities, numeric references, `_xHHHH_` codes, CDATA. */
|
|
44
|
+
const unescapeLikeSheetJs = (text) => text.replace(/<!\[CDATA\[|\]\]>/g, "").replace(/&(quot|apos|gt|lt|amp|#x?[\da-f]+);/gi, (raw, entity) => {
|
|
45
|
+
const named = namedEntities.get(entity.toLowerCase());
|
|
46
|
+
if (named !== void 0) return named;
|
|
47
|
+
const hex = entity[1] === "x" || entity[1] === "X";
|
|
48
|
+
const code = Number.parseInt(entity.slice(hex ? 2 : 1), hex ? 16 : 10);
|
|
49
|
+
return Number.isInteger(code) && code <= 65535 ? String.fromCharCode(code) : raw;
|
|
50
|
+
}).replace(/_x([\da-f]{4})_/gi, (_, code) => String.fromCharCode(Number.parseInt(code, 16)));
|
|
51
|
+
/** SheetJS `resolve_path` from the directory of `xl/…` (only the final segment matters here). */
|
|
52
|
+
const resolveLikeSheetJs = (target) => {
|
|
53
|
+
if (target.startsWith("/")) return target.slice(1);
|
|
54
|
+
const segments = ["xl"];
|
|
55
|
+
for (const step of target.split("/")) if (step === "..") segments.pop();
|
|
56
|
+
else if (step !== ".") segments.push(step);
|
|
57
|
+
return segments.join("/");
|
|
58
|
+
};
|
|
59
|
+
/**
|
|
60
|
+
* Whether a `Target` or `PartName` names a path ending in `.bin`, the only suffix SheetJS hands
|
|
61
|
+
* to a binary parser. Checked raw and SheetJS-unescaped, as written and resolved.
|
|
62
|
+
*/
|
|
63
|
+
const namesBinaryPart = (value) => [value, unescapeLikeSheetJs(value)].some((path) => path.toLowerCase().endsWith(".bin") || resolveLikeSheetJs(path).toLowerCase().endsWith(".bin"));
|
|
64
|
+
/**
|
|
65
|
+
* Relationship types whose `.bin` targets SheetJS never parses: it follows workbook
|
|
66
|
+
* relationships only as sheets (worksheet, chartsheet, dialogsheet, macrosheet, or no type) and
|
|
67
|
+
* worksheet relationships only as comments, drawings, and legacy drawings.
|
|
68
|
+
*/
|
|
69
|
+
const binaryRelationshipTypes = new Set([
|
|
70
|
+
"printerSettings",
|
|
71
|
+
"oleObject",
|
|
72
|
+
"activeXControlBinary",
|
|
73
|
+
"customProperty",
|
|
74
|
+
"attachedToolbars",
|
|
75
|
+
"image",
|
|
76
|
+
"hyperlink"
|
|
77
|
+
]);
|
|
78
|
+
/**
|
|
79
|
+
* Whether a relationships part has a `<Relationship>` whose `Target` ends in `.bin` and whose
|
|
80
|
+
* `Type` (read exactly as SheetJS reads it: case-sensitive, missing counts as a sheet) is not on
|
|
81
|
+
* the allowlist.
|
|
82
|
+
*/
|
|
83
|
+
const relationshipsRouteToBinary = (content) => someSheetJsTag(content, ({ head, attributes }) => {
|
|
84
|
+
const target = attributes.get("Target");
|
|
85
|
+
if (head !== "<Relationship" || target === void 0 || !namesBinaryPart(target)) return false;
|
|
86
|
+
const type = attributes.get("Type");
|
|
87
|
+
return type === void 0 || !binaryRelationshipTypes.has(type.slice(type.lastIndexOf("/") + 1));
|
|
88
|
+
});
|
|
89
|
+
/** Content types of `.bin` parts SheetJS never parses, lower-cased. */
|
|
90
|
+
const binaryPartContentTypes = new Set([
|
|
91
|
+
"application/vnd.openxmlformats-officedocument.spreadsheetml.printersettings",
|
|
92
|
+
"application/vnd.ms-office.activex",
|
|
93
|
+
"application/vnd.openxmlformats-officedocument.oleobject",
|
|
94
|
+
"application/vnd.openxmlformats-officedocument.spreadsheetml.customproperty",
|
|
95
|
+
"application/vnd.ms-excel.attachedtoolbars"
|
|
96
|
+
]);
|
|
97
|
+
/** XLSB part types: `application/vnd.ms-excel.*` without an `+xml` suffix (toolbars excepted). */
|
|
98
|
+
const isBinarySpreadsheetType = (contentType) => contentType.startsWith("application/vnd.ms-excel.") && !contentType.endsWith("+xml") && !binaryPartContentTypes.has(contentType);
|
|
99
|
+
/**
|
|
100
|
+
* Whether `[Content_Types].xml` has an `<Override>` (any prefix, as SheetJS reads it) with an
|
|
101
|
+
* XLSB content type, or a `PartName` ending in `.bin` whose content type is not on the allowlist.
|
|
102
|
+
* `<Default>` entries are ignored: SheetJS does not route by them, and its own XLSX writer emits
|
|
103
|
+
* `<Default Extension="bin">` with the XLSB workbook type.
|
|
104
|
+
*/
|
|
105
|
+
const contentTypesRouteToBinary = (content) => someSheetJsTag(content, ({ head, attributes }) => {
|
|
106
|
+
if (head.replace(/<\w*:/, "<") !== "<Override") return false;
|
|
107
|
+
const contentType = attributes.get("ContentType")?.toLowerCase();
|
|
108
|
+
const partName = attributes.get("PartName");
|
|
109
|
+
if (contentType !== void 0 && isBinarySpreadsheetType(contentType)) return true;
|
|
110
|
+
return partName !== void 0 && namesBinaryPart(partName) && (contentType === void 0 || !binaryPartContentTypes.has(contentType));
|
|
111
|
+
});
|
|
112
|
+
//#endregion
|
|
113
|
+
export { contentTypesRouteToBinary, isAlternateFormatEntry, relationshipsRouteToBinary, sheetJsEntryPath };
|
|
114
|
+
|
|
115
|
+
//# sourceMappingURL=xlsx-routing.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-routing.mjs","names":[],"sources":["../../src/node/xlsx-routing.ts"],"sourcesContent":["import { sheetJsTags, sheetJsTextViews } from './sheetjs-xml.ts'\nimport type { SheetJsTag } from './sheetjs-xml.ts'\n\n/**\n * Early, exact rejection of XLSX input that SheetJS 0.20.3 would hand to its ODS, Numbers, or\n * binary (XLSB) parsers, so users get a clear error instead of \"Could not read XLSX\". These checks\n * are not the security guarantee: SheetJS only ever receives the allowlisted archive built by\n * `buildSheetJsInput` (`xlsx-sheetjs-input.ts`), which contains no marker entries and no `.bin`\n * parts whatever these checks decide.\n */\n\n/**\n * An entry name as SheetJS looks it up: its ZIP reader (`cfb_add`) keeps a name that already\n * starts with `Root Entry/` and otherwise stores `\"Root Entry/\" + name` with the first `//`\n * collapsed; `safegetzipfile` strips `Root Entry/`, treats `\\` and `/` alike, and compares\n * lower-cased names.\n */\nexport const sheetJsEntryPath = (name: string) =>\n (name.startsWith('Root Entry/') ? name : `Root Entry/${name}`.replace('//', '/'))\n .replace(/^Root Entry\\//, '')\n .replaceAll('\\\\', '/')\n .toLowerCase()\n\n/** Entries `parse_zip` checks before content types (SheetJS-normalized paths). */\nconst alternateFormatEntries = new Set([\n 'meta-inf/manifest.xml',\n 'objectdata.xml',\n 'index/document.iwa'\n])\n\n/**\n * Whether SheetJS would treat this entry as an ODS, UOC, or Numbers marker. `CFB.find` matches\n * `Index.zip` by base name, and any `Root Entry/` name (any case) is rejected outright.\n */\nexport const isAlternateFormatEntry = (name: string) => {\n const path = sheetJsEntryPath(name)\n\n return (\n name.toLowerCase().startsWith('root entry/') ||\n alternateFormatEntries.has(path) ||\n path.slice(path.lastIndexOf('/') + 1) === 'index.zip'\n )\n}\n\n/** Whether any tag SheetJS would parse from any view of `content` satisfies `test`. */\nconst someSheetJsTag = (content: Uint8Array, test: (tag: SheetJsTag) => boolean) =>\n sheetJsTextViews(content).some(text => {\n for (const tag of sheetJsTags(text)) {\n if (test(tag)) return true\n }\n\n return false\n })\n\nconst namedEntities: ReadonlyMap<string, string> = new Map([\n ['quot', '\"'],\n ['apos', \"'\"],\n ['gt', '>'],\n ['lt', '<'],\n ['amp', '&']\n])\n\n/** SheetJS `unescapexml`: XML entities, numeric references, `_xHHHH_` codes, CDATA. */\nconst unescapeLikeSheetJs = (text: string): string =>\n text\n .replace(/<!\\[CDATA\\[|\\]\\]>/g, '')\n .replace(/&(quot|apos|gt|lt|amp|#x?[\\da-f]+);/gi, (raw, entity: string) => {\n const named = namedEntities.get(entity.toLowerCase())\n\n if (named !== undefined) return named\n\n const hex = entity[1] === 'x' || entity[1] === 'X'\n const code = Number.parseInt(entity.slice(hex ? 2 : 1), hex ? 16 : 10)\n\n return Number.isInteger(code) && code <= 0xffff ? String.fromCharCode(code) : raw\n })\n .replace(/_x([\\da-f]{4})_/gi, (_, code: string) =>\n String.fromCharCode(Number.parseInt(code, 16))\n )\n\n/** SheetJS `resolve_path` from the directory of `xl/…` (only the final segment matters here). */\nconst resolveLikeSheetJs = (target: string) => {\n if (target.startsWith('/')) return target.slice(1)\n\n const segments = ['xl']\n\n for (const step of target.split('/')) {\n if (step === '..') segments.pop()\n else if (step !== '.') segments.push(step)\n }\n\n return segments.join('/')\n}\n\n/**\n * Whether a `Target` or `PartName` names a path ending in `.bin`, the only suffix SheetJS hands\n * to a binary parser. Checked raw and SheetJS-unescaped, as written and resolved.\n */\nconst namesBinaryPart = (value: string) =>\n [value, unescapeLikeSheetJs(value)].some(\n path =>\n path.toLowerCase().endsWith('.bin') || resolveLikeSheetJs(path).toLowerCase().endsWith('.bin')\n )\n\n/**\n * Relationship types whose `.bin` targets SheetJS never parses: it follows workbook\n * relationships only as sheets (worksheet, chartsheet, dialogsheet, macrosheet, or no type) and\n * worksheet relationships only as comments, drawings, and legacy drawings.\n */\nconst binaryRelationshipTypes = new Set([\n 'printerSettings',\n 'oleObject',\n 'activeXControlBinary',\n 'customProperty',\n 'attachedToolbars',\n 'image',\n 'hyperlink'\n])\n\n/**\n * Whether a relationships part has a `<Relationship>` whose `Target` ends in `.bin` and whose\n * `Type` (read exactly as SheetJS reads it: case-sensitive, missing counts as a sheet) is not on\n * the allowlist.\n */\nexport const relationshipsRouteToBinary = (content: Uint8Array) =>\n someSheetJsTag(content, ({ head, attributes }) => {\n const target = attributes.get('Target')\n\n if (head !== '<Relationship' || target === undefined || !namesBinaryPart(target)) return false\n\n const type = attributes.get('Type')\n\n return type === undefined || !binaryRelationshipTypes.has(type.slice(type.lastIndexOf('/') + 1))\n })\n\n/** Content types of `.bin` parts SheetJS never parses, lower-cased. */\nconst binaryPartContentTypes = new Set([\n 'application/vnd.openxmlformats-officedocument.spreadsheetml.printersettings',\n 'application/vnd.ms-office.activex',\n 'application/vnd.openxmlformats-officedocument.oleobject',\n 'application/vnd.openxmlformats-officedocument.spreadsheetml.customproperty',\n 'application/vnd.ms-excel.attachedtoolbars'\n])\n\n/** XLSB part types: `application/vnd.ms-excel.*` without an `+xml` suffix (toolbars excepted). */\nconst isBinarySpreadsheetType = (contentType: string) =>\n contentType.startsWith('application/vnd.ms-excel.') &&\n !contentType.endsWith('+xml') &&\n !binaryPartContentTypes.has(contentType)\n\n/**\n * Whether `[Content_Types].xml` has an `<Override>` (any prefix, as SheetJS reads it) with an\n * XLSB content type, or a `PartName` ending in `.bin` whose content type is not on the allowlist.\n * `<Default>` entries are ignored: SheetJS does not route by them, and its own XLSX writer emits\n * `<Default Extension=\"bin\">` with the XLSB workbook type.\n */\nexport const contentTypesRouteToBinary = (content: Uint8Array) =>\n someSheetJsTag(content, ({ head, attributes }) => {\n if (head.replace(/<\\w*:/, '<') !== '<Override') return false\n\n const contentType = attributes.get('ContentType')?.toLowerCase()\n const partName = attributes.get('PartName')\n\n if (contentType !== undefined && isBinarySpreadsheetType(contentType)) return true\n\n return (\n partName !== undefined &&\n namesBinaryPart(partName) &&\n (contentType === undefined || !binaryPartContentTypes.has(contentType))\n )\n })\n"],"mappings":";;;;;;;;;;;;;;;AAiBA,MAAa,oBAAoB,UAC9B,KAAK,WAAW,aAAa,IAAI,OAAO,cAAc,OAAO,QAAQ,MAAM,GAAG,GAC5E,QAAQ,iBAAiB,EAAE,EAC3B,WAAW,MAAM,GAAG,EACpB,YAAY;;AAGjB,MAAM,yBAAyB,IAAI,IAAI;CACrC;CACA;CACA;AACF,CAAC;;;;;AAMD,MAAa,0BAA0B,SAAiB;CACtD,MAAM,OAAO,iBAAiB,IAAI;CAElC,OACE,KAAK,YAAY,EAAE,WAAW,aAAa,KAC3C,uBAAuB,IAAI,IAAI,KAC/B,KAAK,MAAM,KAAK,YAAY,GAAG,IAAI,CAAC,MAAM;AAE9C;;AAGA,MAAM,kBAAkB,SAAqB,SAC3C,iBAAiB,OAAO,EAAE,MAAK,SAAQ;CACrC,KAAK,MAAM,OAAO,YAAY,IAAI,GAChC,IAAI,KAAK,GAAG,GAAG,OAAO;CAGxB,OAAO;AACT,CAAC;AAEH,MAAM,gBAA6C,IAAI,IAAI;CACzD,CAAC,QAAQ,IAAG;CACZ,CAAC,QAAQ,GAAG;CACZ,CAAC,MAAM,GAAG;CACV,CAAC,MAAM,GAAG;CACV,CAAC,OAAO,GAAG;AACb,CAAC;;AAGD,MAAM,uBAAuB,SAC3B,KACG,QAAQ,sBAAsB,EAAE,EAChC,QAAQ,0CAA0C,KAAK,WAAmB;CACzE,MAAM,QAAQ,cAAc,IAAI,OAAO,YAAY,CAAC;CAEpD,IAAI,UAAU,KAAA,GAAW,OAAO;CAEhC,MAAM,MAAM,OAAO,OAAO,OAAO,OAAO,OAAO;CAC/C,MAAM,OAAO,OAAO,SAAS,OAAO,MAAM,MAAM,IAAI,CAAC,GAAG,MAAM,KAAK,EAAE;CAErE,OAAO,OAAO,UAAU,IAAI,KAAK,QAAQ,QAAS,OAAO,aAAa,IAAI,IAAI;AAChF,CAAC,EACA,QAAQ,sBAAsB,GAAG,SAChC,OAAO,aAAa,OAAO,SAAS,MAAM,EAAE,CAAC,CAC/C;;AAGJ,MAAM,sBAAsB,WAAmB;CAC7C,IAAI,OAAO,WAAW,GAAG,GAAG,OAAO,OAAO,MAAM,CAAC;CAEjD,MAAM,WAAW,CAAC,IAAI;CAEtB,KAAK,MAAM,QAAQ,OAAO,MAAM,GAAG,GACjC,IAAI,SAAS,MAAM,SAAS,IAAI;MAC3B,IAAI,SAAS,KAAK,SAAS,KAAK,IAAI;CAG3C,OAAO,SAAS,KAAK,GAAG;AAC1B;;;;;AAMA,MAAM,mBAAmB,UACvB,CAAC,OAAO,oBAAoB,KAAK,CAAC,EAAE,MAClC,SACE,KAAK,YAAY,EAAE,SAAS,MAAM,KAAK,mBAAmB,IAAI,EAAE,YAAY,EAAE,SAAS,MAAM,CACjG;;;;;;AAOF,MAAM,0BAA0B,IAAI,IAAI;CACtC;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;;;;;;AAOD,MAAa,8BAA8B,YACzC,eAAe,UAAU,EAAE,MAAM,iBAAiB;CAChD,MAAM,SAAS,WAAW,IAAI,QAAQ;CAEtC,IAAI,SAAS,mBAAmB,WAAW,KAAA,KAAa,CAAC,gBAAgB,MAAM,GAAG,OAAO;CAEzF,MAAM,OAAO,WAAW,IAAI,MAAM;CAElC,OAAO,SAAS,KAAA,KAAa,CAAC,wBAAwB,IAAI,KAAK,MAAM,KAAK,YAAY,GAAG,IAAI,CAAC,CAAC;AACjG,CAAC;;AAGH,MAAM,yBAAyB,IAAI,IAAI;CACrC;CACA;CACA;CACA;CACA;AACF,CAAC;;AAGD,MAAM,2BAA2B,gBAC/B,YAAY,WAAW,2BAA2B,KAClD,CAAC,YAAY,SAAS,MAAM,KAC5B,CAAC,uBAAuB,IAAI,WAAW;;;;;;;AAQzC,MAAa,6BAA6B,YACxC,eAAe,UAAU,EAAE,MAAM,iBAAiB;CAChD,IAAI,KAAK,QAAQ,SAAS,GAAG,MAAM,aAAa,OAAO;CAEvD,MAAM,cAAc,WAAW,IAAI,aAAa,GAAG,YAAY;CAC/D,MAAM,WAAW,WAAW,IAAI,UAAU;CAE1C,IAAI,gBAAgB,KAAA,KAAa,wBAAwB,WAAW,GAAG,OAAO;CAE9E,OACE,aAAa,KAAA,KACb,gBAAgB,QAAQ,MACvB,gBAAgB,KAAA,KAAa,CAAC,uBAAuB,IAAI,WAAW;AAEzE,CAAC"}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
//#region src/node/xlsx-sheetjs-input.d.ts
|
|
2
|
+
/** Canonical names SheetJS can receive besides `xl/worksheets/sheet<n>.xml`. */
|
|
3
|
+
declare const sheetJsFixedParts: readonly ["[Content_Types].xml", "_rels/.rels", "xl/workbook.xml", "xl/_rels/workbook.xml.rels", "xl/sharedStrings.xml", "xl/styles.xml"];
|
|
4
|
+
/** Whether `name` is one of the canonical names `buildSheetJsInput` emits. */
|
|
5
|
+
declare const isSheetJsInputName: (name: string) => boolean;
|
|
6
|
+
type SheetJsInputSheet = {
|
|
7
|
+
/** The sheet name exactly as SheetJS reads it from the generated workbook. */readonly name: string; /** The validated source part handed over as this sheet's worksheet, if any. */
|
|
8
|
+
readonly part: string | undefined;
|
|
9
|
+
};
|
|
10
|
+
type SheetJsInput = {
|
|
11
|
+
/** The stored-entry ZIP handed to SheetJS. */readonly archive: Uint8Array; /** Entry names in `archive`, all canonical (see `isSheetJsInputName`). */
|
|
12
|
+
readonly names: ReadonlyArray<string>; /** Every sheet of the generated workbook, in order. */
|
|
13
|
+
readonly sheets: ReadonlyArray<SheetJsInputSheet>; /** The workbook title from its core properties, read without SheetJS. */
|
|
14
|
+
readonly title: string | undefined;
|
|
15
|
+
};
|
|
16
|
+
/**
|
|
17
|
+
* Build SheetJS's input from validated parts. Sheet `n` of the workbook gets relationship
|
|
18
|
+
* `rId<n>` to `xl/worksheets/sheet<n>.xml`. That entry holds the sheet's worksheet when its
|
|
19
|
+
* workbook relationship is an internal worksheet resolving to an `.xml` part; other sheets (chart
|
|
20
|
+
* sheets, missing parts) keep their name and order but have no entry, so SheetJS skips them.
|
|
21
|
+
*
|
|
22
|
+
* Throws `FileExtractionError` when the workbook declares more than `maxSheets` sheets, when it
|
|
23
|
+
* is malformed (see `readWorkbookModel`; two sheets on one worksheet part count too), or when a
|
|
24
|
+
* copied part holds CDATA, all before SheetJS runs.
|
|
25
|
+
*/
|
|
26
|
+
declare const buildSheetJsInput: (parts: Readonly<Record<string, Uint8Array>>, maxSheets: number) => SheetJsInput;
|
|
27
|
+
//#endregion
|
|
28
|
+
export { SheetJsInput, SheetJsInputSheet, buildSheetJsInput, isSheetJsInputName, sheetJsFixedParts };
|
|
29
|
+
//# sourceMappingURL=xlsx-sheetjs-input.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-sheetjs-input.d.mts","names":[],"sources":["../../src/node/xlsx-sheetjs-input.ts"],"mappings":";;cA8Da,iBAAA;;cAcA,kBAAA,GAAsB,IAAY;AAAA,KA4DnC,iBAAA;EAnEF,uFAqEC,IAAA,UA7DgD;EAAA,SA+DhD,IAAI;AAAA;AAAA,KAGH,YAAA;EAPA,uDASD,OAAA,EAAS,UAAA;WAET,KAAA,EAAO,aAAA,UAPH;EAAA,SASJ,MAAA,EAAQ,aAAA,CAAc,iBAAA,GANT;EAAA,SAQb,KAAA;AAAA;;;;;;;;;;;cAaE,iBAAA,GACX,KAAA,EAAO,QAAA,CAAS,MAAA,SAAe,UAAA,IAC/B,SAAA,aACC,YAAA"}
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
import { FileExtractionError } from "../errors.mjs";
|
|
2
|
+
import { coreTitle, officeDocumentRelationshipsNamespace, sheetJsCouldReadCdata, sheetJsXmlHeader } from "./sheetjs-xml.mjs";
|
|
3
|
+
import { directoryOf, indexParts, relationships, relationshipsPathFor, resolvePartPath, utf8Text, workbookPartPath } from "./xlsx-parts.mjs";
|
|
4
|
+
import { readStyles, stylesXml } from "./xlsx-styles.mjs";
|
|
5
|
+
import { readWorkbookModel, workbookXml } from "./xlsx-workbook.mjs";
|
|
6
|
+
import { strToU8, zipSync } from "fflate";
|
|
7
|
+
//#region src/node/xlsx-sheetjs-input.ts
|
|
8
|
+
/**
|
|
9
|
+
* The archive SheetJS reads is built here from scratch, never passed through. Every part SheetJS
|
|
10
|
+
* reads is generated or checked:
|
|
11
|
+
*
|
|
12
|
+
* - generated: `[Content_Types].xml`, `_rels/.rels`, `xl/_rels/workbook.xml.rels`,
|
|
13
|
+
* `xl/workbook.xml` (`xlsx-workbook.ts`), and `xl/styles.xml` (`xlsx-styles.ts`);
|
|
14
|
+
* - copied after checks: worksheets (stored as `xl/worksheets/sheet<n>.xml`) and
|
|
15
|
+
* `xl/sharedStrings.xml`, validated, hyperlink-stripped, and free of anything SheetJS could
|
|
16
|
+
* turn into a CDATA marker (`sheetJsCouldReadCdata`).
|
|
17
|
+
*
|
|
18
|
+
* With these entries SheetJS 0.20.3 `parse_zip` can only take its XLSX path:
|
|
19
|
+
*
|
|
20
|
+
* - no `META-INF/manifest.xml`, `objectdata.xml`, or `Index/Document.iwa`, so it never reaches
|
|
21
|
+
* `parse_ods` or `parse_numbers_iwa`; `[Content_Types].xml` exists, so `Index.zip` is never read;
|
|
22
|
+
* - every entry ends in `.xml` or `.rels`, so no binary (XLSB) parser can receive data: SheetJS
|
|
23
|
+
* dispatches on the requested path ending in `.bin`, and `safegetzipfile` only returns an entry
|
|
24
|
+
* whose name equals that path ignoring case;
|
|
25
|
+
* - the generated content types name the XML workbook, so `xlsb` stays false, and no attacker
|
|
26
|
+
* `Override`, `PartName`, relationship `Type`, or `Target` reaches SheetJS;
|
|
27
|
+
* - the generated workbook lists exactly the sheets the extractor counted, each with its own
|
|
28
|
+
* relationship to its own `xl/worksheets/sheet<n>.xml`, so every part is parsed at most once.
|
|
29
|
+
*
|
|
30
|
+
* Comments, threaded comments, VML, drawings, worksheet relationships, external links, pivot
|
|
31
|
+
* caches, calculation chains, metadata, themes, `customXml`, and `docProps/*` (the title is read
|
|
32
|
+
* by the extractor) are left out; SheetJS reads none of them to produce cell values or display
|
|
33
|
+
* text.
|
|
34
|
+
*/
|
|
35
|
+
const strictOfficeDocumentRelationships = "http://purl.oclc.org/ooxml/officeDocument/relationships";
|
|
36
|
+
const corePropertiesTypes = new Set(["http://schemas.openxmlformats.org/package/2006/relationships/metadata/core-properties", "http://schemas.openxmlformats.org/officedocument/2006/relationships/metadata/core-properties"]);
|
|
37
|
+
/** A relationship of `kind` in the transitional or strict namespace. */
|
|
38
|
+
const hasKind = (relationship, kind) => relationship.type === `http://schemas.openxmlformats.org/officeDocument/2006/relationships/${kind}` || relationship.type === `${strictOfficeDocumentRelationships}/${kind}`;
|
|
39
|
+
/** Canonical names SheetJS can receive besides `xl/worksheets/sheet<n>.xml`. */
|
|
40
|
+
const sheetJsFixedParts = [
|
|
41
|
+
"[Content_Types].xml",
|
|
42
|
+
"_rels/.rels",
|
|
43
|
+
"xl/workbook.xml",
|
|
44
|
+
"xl/_rels/workbook.xml.rels",
|
|
45
|
+
"xl/sharedStrings.xml",
|
|
46
|
+
"xl/styles.xml"
|
|
47
|
+
];
|
|
48
|
+
const canonicalWorksheet = /^xl\/worksheets\/sheet[1-9]\d*\.xml$/;
|
|
49
|
+
const fixedPartNames = new Set(sheetJsFixedParts);
|
|
50
|
+
/** Whether `name` is one of the canonical names `buildSheetJsInput` emits. */
|
|
51
|
+
const isSheetJsInputName = (name) => fixedPartNames.has(name) || canonicalWorksheet.test(name);
|
|
52
|
+
const contentType = {
|
|
53
|
+
workbook: "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml",
|
|
54
|
+
worksheet: "application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml",
|
|
55
|
+
sharedStrings: "application/vnd.openxmlformats-officedocument.spreadsheetml.sharedStrings+xml",
|
|
56
|
+
styles: "application/vnd.openxmlformats-officedocument.spreadsheetml.styles+xml"
|
|
57
|
+
};
|
|
58
|
+
const contentTypesXml = (overrides) => `${sheetJsXmlHeader}<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"><Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/><Default Extension="xml" ContentType="application/xml"/>${overrides.map(([path, type]) => `<Override PartName="/${path}" ContentType="${type}"/>`).join("")}</Types>`;
|
|
59
|
+
const relationshipsXml = (entries) => `${sheetJsXmlHeader}<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">${entries.map(([id, type, target]) => `<Relationship Id="${id}" Type="${type}" Target="${target}"/>`).join("")}</Relationships>`;
|
|
60
|
+
/** An XML part found by name ignoring case; `.xml` only, so no binary bytes are renamed. */
|
|
61
|
+
const xmlPart = (index, path) => {
|
|
62
|
+
const part = path === void 0 ? void 0 : index.find(path);
|
|
63
|
+
return part !== void 0 && part.name.toLowerCase().endsWith(".xml") ? part : void 0;
|
|
64
|
+
};
|
|
65
|
+
/** The part a relationship of `kind` points at, else the conventional path. */
|
|
66
|
+
const relatedXmlPart = (index, related, baseDirectory, matches, conventionalPath) => {
|
|
67
|
+
for (const relationship of related.values()) if (!relationship.external && matches(relationship)) return xmlPart(index, resolvePartPath(baseDirectory, relationship.target));
|
|
68
|
+
return xmlPart(index, conventionalPath);
|
|
69
|
+
};
|
|
70
|
+
const malformed = () => new FileExtractionError({
|
|
71
|
+
format: "xlsx",
|
|
72
|
+
message: "XLSX workbook is malformed."
|
|
73
|
+
});
|
|
74
|
+
const cdataRejected = () => new FileExtractionError({
|
|
75
|
+
format: "xlsx",
|
|
76
|
+
message: "XLSX contains unsupported markup (CDATA, comments or declarations) in worksheet or shared-strings parts."
|
|
77
|
+
});
|
|
78
|
+
/** A copied part, unless SheetJS could meet a CDATA marker in it (decoded or tag-removed). */
|
|
79
|
+
const withoutCdata = (bytes) => {
|
|
80
|
+
if (sheetJsCouldReadCdata(bytes)) throw cdataRejected();
|
|
81
|
+
return bytes;
|
|
82
|
+
};
|
|
83
|
+
/**
|
|
84
|
+
* Build SheetJS's input from validated parts. Sheet `n` of the workbook gets relationship
|
|
85
|
+
* `rId<n>` to `xl/worksheets/sheet<n>.xml`. That entry holds the sheet's worksheet when its
|
|
86
|
+
* workbook relationship is an internal worksheet resolving to an `.xml` part; other sheets (chart
|
|
87
|
+
* sheets, missing parts) keep their name and order but have no entry, so SheetJS skips them.
|
|
88
|
+
*
|
|
89
|
+
* Throws `FileExtractionError` when the workbook declares more than `maxSheets` sheets, when it
|
|
90
|
+
* is malformed (see `readWorkbookModel`; two sheets on one worksheet part count too), or when a
|
|
91
|
+
* copied part holds CDATA, all before SheetJS runs.
|
|
92
|
+
*/
|
|
93
|
+
const buildSheetJsInput = (parts, maxSheets) => {
|
|
94
|
+
const index = indexParts(parts);
|
|
95
|
+
const workbook = index.find(workbookPartPath);
|
|
96
|
+
if (workbook === void 0) throw new FileExtractionError({
|
|
97
|
+
format: "xlsx",
|
|
98
|
+
message: "Invalid Office archive."
|
|
99
|
+
});
|
|
100
|
+
const workbookDirectory = directoryOf(workbookPartPath);
|
|
101
|
+
const workbookRelationships = relationships(utf8Text(index.find(relationshipsPathFor(workbookPartPath))?.bytes));
|
|
102
|
+
const model = readWorkbookModel(workbook.bytes, maxSheets);
|
|
103
|
+
const files = Object.create(null);
|
|
104
|
+
const overrides = [];
|
|
105
|
+
const sheetRelationships = [];
|
|
106
|
+
const sheets = [];
|
|
107
|
+
const included = /* @__PURE__ */ new Set();
|
|
108
|
+
for (const [position, sheet] of model.sheets.entries()) {
|
|
109
|
+
const name = `worksheets/sheet${position + 1}.xml`;
|
|
110
|
+
sheetRelationships.push([
|
|
111
|
+
`rId${position + 1}`,
|
|
112
|
+
`${officeDocumentRelationshipsNamespace}/worksheet`,
|
|
113
|
+
name
|
|
114
|
+
]);
|
|
115
|
+
const relationship = sheet.relationshipId === void 0 ? void 0 : workbookRelationships.get(sheet.relationshipId);
|
|
116
|
+
const part = relationship === void 0 || relationship.external || !hasKind(relationship, "worksheet") ? void 0 : xmlPart(index, resolvePartPath(workbookDirectory, relationship.target));
|
|
117
|
+
if (part === void 0) {
|
|
118
|
+
sheets.push({
|
|
119
|
+
name: sheet.name,
|
|
120
|
+
part: void 0
|
|
121
|
+
});
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
if (included.has(part.name)) throw malformed();
|
|
125
|
+
included.add(part.name);
|
|
126
|
+
files[`xl/${name}`] = withoutCdata(part.bytes);
|
|
127
|
+
overrides.push([`xl/${name}`, contentType.worksheet]);
|
|
128
|
+
sheets.push({
|
|
129
|
+
name: sheet.name,
|
|
130
|
+
part: part.name
|
|
131
|
+
});
|
|
132
|
+
}
|
|
133
|
+
const sharedStrings = relatedXmlPart(index, workbookRelationships, workbookDirectory, (relationship) => hasKind(relationship, "sharedStrings"), "xl/sharedStrings.xml");
|
|
134
|
+
if (sharedStrings !== void 0) {
|
|
135
|
+
files["xl/sharedStrings.xml"] = withoutCdata(sharedStrings.bytes);
|
|
136
|
+
overrides.push(["xl/sharedStrings.xml", contentType.sharedStrings]);
|
|
137
|
+
}
|
|
138
|
+
const styles = relatedXmlPart(index, workbookRelationships, workbookDirectory, (relationship) => hasKind(relationship, "styles"), "xl/styles.xml");
|
|
139
|
+
if (styles !== void 0) {
|
|
140
|
+
files["xl/styles.xml"] = strToU8(stylesXml(readStyles(styles.bytes)));
|
|
141
|
+
overrides.push(["xl/styles.xml", contentType.styles]);
|
|
142
|
+
}
|
|
143
|
+
const core = relatedXmlPart(index, relationships(utf8Text(index.find("_rels/.rels")?.bytes)), "", (relationship) => relationship.type !== void 0 && corePropertiesTypes.has(relationship.type), "docProps/core.xml");
|
|
144
|
+
const archive = {
|
|
145
|
+
"[Content_Types].xml": strToU8(contentTypesXml([[workbookPartPath, contentType.workbook], ...overrides])),
|
|
146
|
+
"_rels/.rels": strToU8(relationshipsXml([[
|
|
147
|
+
"rId1",
|
|
148
|
+
`${officeDocumentRelationshipsNamespace}/officeDocument`,
|
|
149
|
+
workbookPartPath
|
|
150
|
+
]])),
|
|
151
|
+
[workbookPartPath]: strToU8(workbookXml(model)),
|
|
152
|
+
"xl/_rels/workbook.xml.rels": strToU8(relationshipsXml(sheetRelationships)),
|
|
153
|
+
...files
|
|
154
|
+
};
|
|
155
|
+
return {
|
|
156
|
+
archive: zipSync(archive, { level: 0 }),
|
|
157
|
+
names: Object.keys(archive),
|
|
158
|
+
sheets,
|
|
159
|
+
title: coreTitle(core?.bytes)
|
|
160
|
+
};
|
|
161
|
+
};
|
|
162
|
+
//#endregion
|
|
163
|
+
export { buildSheetJsInput, isSheetJsInputName, sheetJsFixedParts };
|
|
164
|
+
|
|
165
|
+
//# sourceMappingURL=xlsx-sheetjs-input.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-sheetjs-input.mjs","names":[],"sources":["../../src/node/xlsx-sheetjs-input.ts"],"sourcesContent":["import { strToU8, zipSync } from 'fflate'\nimport { FileExtractionError } from '../errors.ts'\nimport {\n coreTitle,\n officeDocumentRelationshipsNamespace,\n sheetJsCouldReadCdata,\n sheetJsXmlHeader\n} from './sheetjs-xml.ts'\nimport { readStyles, stylesXml } from './xlsx-styles.ts'\nimport { readWorkbookModel, workbookXml } from './xlsx-workbook.ts'\nimport {\n directoryOf,\n indexParts,\n relationships,\n relationshipsPathFor,\n resolvePartPath,\n utf8Text,\n workbookPartPath\n} from './xlsx-parts.ts'\nimport type { PartIndex, Relationship } from './xlsx-parts.ts'\n\n/**\n * The archive SheetJS reads is built here from scratch, never passed through. Every part SheetJS\n * reads is generated or checked:\n *\n * - generated: `[Content_Types].xml`, `_rels/.rels`, `xl/_rels/workbook.xml.rels`,\n * `xl/workbook.xml` (`xlsx-workbook.ts`), and `xl/styles.xml` (`xlsx-styles.ts`);\n * - copied after checks: worksheets (stored as `xl/worksheets/sheet<n>.xml`) and\n * `xl/sharedStrings.xml`, validated, hyperlink-stripped, and free of anything SheetJS could\n * turn into a CDATA marker (`sheetJsCouldReadCdata`).\n *\n * With these entries SheetJS 0.20.3 `parse_zip` can only take its XLSX path:\n *\n * - no `META-INF/manifest.xml`, `objectdata.xml`, or `Index/Document.iwa`, so it never reaches\n * `parse_ods` or `parse_numbers_iwa`; `[Content_Types].xml` exists, so `Index.zip` is never read;\n * - every entry ends in `.xml` or `.rels`, so no binary (XLSB) parser can receive data: SheetJS\n * dispatches on the requested path ending in `.bin`, and `safegetzipfile` only returns an entry\n * whose name equals that path ignoring case;\n * - the generated content types name the XML workbook, so `xlsb` stays false, and no attacker\n * `Override`, `PartName`, relationship `Type`, or `Target` reaches SheetJS;\n * - the generated workbook lists exactly the sheets the extractor counted, each with its own\n * relationship to its own `xl/worksheets/sheet<n>.xml`, so every part is parsed at most once.\n *\n * Comments, threaded comments, VML, drawings, worksheet relationships, external links, pivot\n * caches, calculation chains, metadata, themes, `customXml`, and `docProps/*` (the title is read\n * by the extractor) are left out; SheetJS reads none of them to produce cell values or display\n * text.\n */\n\nconst strictOfficeDocumentRelationships = 'http://purl.oclc.org/ooxml/officeDocument/relationships'\n\nconst corePropertiesTypes = new Set([\n 'http://schemas.openxmlformats.org/package/2006/relationships/metadata/core-properties',\n 'http://schemas.openxmlformats.org/officedocument/2006/relationships/metadata/core-properties'\n])\n\n/** A relationship of `kind` in the transitional or strict namespace. */\nconst hasKind = (relationship: Relationship, kind: string) =>\n relationship.type === `${officeDocumentRelationshipsNamespace}/${kind}` ||\n relationship.type === `${strictOfficeDocumentRelationships}/${kind}`\n\n/** Canonical names SheetJS can receive besides `xl/worksheets/sheet<n>.xml`. */\nexport const sheetJsFixedParts = [\n '[Content_Types].xml',\n '_rels/.rels',\n 'xl/workbook.xml',\n 'xl/_rels/workbook.xml.rels',\n 'xl/sharedStrings.xml',\n 'xl/styles.xml'\n] as const\n\nconst canonicalWorksheet = /^xl\\/worksheets\\/sheet[1-9]\\d*\\.xml$/\n\nconst fixedPartNames: ReadonlySet<string> = new Set(sheetJsFixedParts)\n\n/** Whether `name` is one of the canonical names `buildSheetJsInput` emits. */\nexport const isSheetJsInputName = (name: string) =>\n fixedPartNames.has(name) || canonicalWorksheet.test(name)\n\nconst contentType = {\n workbook: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml',\n worksheet: 'application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml',\n sharedStrings: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sharedStrings+xml',\n styles: 'application/vnd.openxmlformats-officedocument.spreadsheetml.styles+xml'\n} as const\n\nconst contentTypesXml = (overrides: ReadonlyArray<readonly [string, string]>) =>\n `${sheetJsXmlHeader}<Types xmlns=\"http://schemas.openxmlformats.org/package/2006/content-types\"><Default Extension=\"rels\" ContentType=\"application/vnd.openxmlformats-package.relationships+xml\"/><Default Extension=\"xml\" ContentType=\"application/xml\"/>${overrides\n .map(([path, type]) => `<Override PartName=\"/${path}\" ContentType=\"${type}\"/>`)\n .join('')}</Types>`\n\nconst relationshipsXml = (entries: ReadonlyArray<readonly [string, string, string]>) =>\n `${sheetJsXmlHeader}<Relationships xmlns=\"http://schemas.openxmlformats.org/package/2006/relationships\">${entries\n .map(([id, type, target]) => `<Relationship Id=\"${id}\" Type=\"${type}\" Target=\"${target}\"/>`)\n .join('')}</Relationships>`\n\n/** An XML part found by name ignoring case; `.xml` only, so no binary bytes are renamed. */\nconst xmlPart = (index: PartIndex, path: string | undefined) => {\n const part = path === undefined ? undefined : index.find(path)\n\n return part !== undefined && part.name.toLowerCase().endsWith('.xml') ? part : undefined\n}\n\n/** The part a relationship of `kind` points at, else the conventional path. */\nconst relatedXmlPart = (\n index: PartIndex,\n related: ReadonlyMap<string, Relationship>,\n baseDirectory: string,\n matches: (relationship: Relationship) => boolean,\n conventionalPath: string\n) => {\n for (const relationship of related.values()) {\n if (!relationship.external && matches(relationship))\n return xmlPart(index, resolvePartPath(baseDirectory, relationship.target))\n }\n\n return xmlPart(index, conventionalPath)\n}\n\nconst malformed = () =>\n new FileExtractionError({ format: 'xlsx', message: 'XLSX workbook is malformed.' })\n\nconst cdataRejected = () =>\n new FileExtractionError({\n format: 'xlsx',\n message:\n 'XLSX contains unsupported markup (CDATA, comments or declarations) in worksheet or shared-strings parts.'\n })\n\n/** A copied part, unless SheetJS could meet a CDATA marker in it (decoded or tag-removed). */\nconst withoutCdata = (bytes: Uint8Array) => {\n if (sheetJsCouldReadCdata(bytes)) throw cdataRejected()\n\n return bytes\n}\n\nexport type SheetJsInputSheet = {\n /** The sheet name exactly as SheetJS reads it from the generated workbook. */\n readonly name: string\n /** The validated source part handed over as this sheet's worksheet, if any. */\n readonly part: string | undefined\n}\n\nexport type SheetJsInput = {\n /** The stored-entry ZIP handed to SheetJS. */\n readonly archive: Uint8Array\n /** Entry names in `archive`, all canonical (see `isSheetJsInputName`). */\n readonly names: ReadonlyArray<string>\n /** Every sheet of the generated workbook, in order. */\n readonly sheets: ReadonlyArray<SheetJsInputSheet>\n /** The workbook title from its core properties, read without SheetJS. */\n readonly title: string | undefined\n}\n\n/**\n * Build SheetJS's input from validated parts. Sheet `n` of the workbook gets relationship\n * `rId<n>` to `xl/worksheets/sheet<n>.xml`. That entry holds the sheet's worksheet when its\n * workbook relationship is an internal worksheet resolving to an `.xml` part; other sheets (chart\n * sheets, missing parts) keep their name and order but have no entry, so SheetJS skips them.\n *\n * Throws `FileExtractionError` when the workbook declares more than `maxSheets` sheets, when it\n * is malformed (see `readWorkbookModel`; two sheets on one worksheet part count too), or when a\n * copied part holds CDATA, all before SheetJS runs.\n */\nexport const buildSheetJsInput = (\n parts: Readonly<Record<string, Uint8Array>>,\n maxSheets: number\n): SheetJsInput => {\n const index = indexParts(parts)\n const workbook = index.find(workbookPartPath)\n\n if (workbook === undefined)\n throw new FileExtractionError({ format: 'xlsx', message: 'Invalid Office archive.' })\n\n const workbookDirectory = directoryOf(workbookPartPath)\n\n const workbookRelationships = relationships(\n utf8Text(index.find(relationshipsPathFor(workbookPartPath))?.bytes)\n )\n\n const model = readWorkbookModel(workbook.bytes, maxSheets)\n const files: Record<string, Uint8Array> = Object.create(null)\n const overrides: Array<readonly [string, string]> = []\n const sheetRelationships: Array<readonly [string, string, string]> = []\n const sheets: Array<SheetJsInputSheet> = []\n const included = new Set<string>()\n\n for (const [position, sheet] of model.sheets.entries()) {\n const name = `worksheets/sheet${position + 1}.xml`\n\n sheetRelationships.push([\n `rId${position + 1}`,\n `${officeDocumentRelationshipsNamespace}/worksheet`,\n name\n ])\n\n const relationship =\n sheet.relationshipId === undefined\n ? undefined\n : workbookRelationships.get(sheet.relationshipId)\n\n const part =\n relationship === undefined || relationship.external || !hasKind(relationship, 'worksheet')\n ? undefined\n : xmlPart(index, resolvePartPath(workbookDirectory, relationship.target))\n\n if (part === undefined) {\n sheets.push({ name: sheet.name, part: undefined })\n continue\n }\n\n // One part per sheet: SheetJS would parse and keep a shared part once per declaration.\n if (included.has(part.name)) throw malformed()\n\n included.add(part.name)\n files[`xl/${name}`] = withoutCdata(part.bytes)\n overrides.push([`xl/${name}`, contentType.worksheet])\n sheets.push({ name: sheet.name, part: part.name })\n }\n\n const sharedStrings = relatedXmlPart(\n index,\n workbookRelationships,\n workbookDirectory,\n relationship => hasKind(relationship, 'sharedStrings'),\n 'xl/sharedStrings.xml'\n )\n\n if (sharedStrings !== undefined) {\n files['xl/sharedStrings.xml'] = withoutCdata(sharedStrings.bytes)\n overrides.push(['xl/sharedStrings.xml', contentType.sharedStrings])\n }\n\n const styles = relatedXmlPart(\n index,\n workbookRelationships,\n workbookDirectory,\n relationship => hasKind(relationship, 'styles'),\n 'xl/styles.xml'\n )\n\n if (styles !== undefined) {\n files['xl/styles.xml'] = strToU8(stylesXml(readStyles(styles.bytes)))\n overrides.push(['xl/styles.xml', contentType.styles])\n }\n\n const core = relatedXmlPart(\n index,\n relationships(utf8Text(index.find('_rels/.rels')?.bytes)),\n '',\n relationship => relationship.type !== undefined && corePropertiesTypes.has(relationship.type),\n 'docProps/core.xml'\n )\n\n const archive = {\n '[Content_Types].xml': strToU8(\n contentTypesXml([[workbookPartPath, contentType.workbook], ...overrides])\n ),\n '_rels/.rels': strToU8(\n relationshipsXml([\n ['rId1', `${officeDocumentRelationshipsNamespace}/officeDocument`, workbookPartPath]\n ])\n ),\n [workbookPartPath]: strToU8(workbookXml(model)),\n 'xl/_rels/workbook.xml.rels': strToU8(relationshipsXml(sheetRelationships)),\n ...files\n }\n\n return {\n archive: zipSync(archive, { level: 0 }),\n names: Object.keys(archive),\n sheets,\n title: coreTitle(core?.bytes)\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiDA,MAAM,oCAAoC;AAE1C,MAAM,sBAAsB,IAAI,IAAI,CAClC,yFACA,8FACF,CAAC;;AAGD,MAAM,WAAW,cAA4B,SAC3C,aAAa,SAAS,uEAA2C,UACjE,aAAa,SAAS,GAAG,kCAAkC,GAAG;;AAGhE,MAAa,oBAAoB;CAC/B;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,MAAM,qBAAqB;AAE3B,MAAM,iBAAsC,IAAI,IAAI,iBAAiB;;AAGrE,MAAa,sBAAsB,SACjC,eAAe,IAAI,IAAI,KAAK,mBAAmB,KAAK,IAAI;AAE1D,MAAM,cAAc;CAClB,UAAU;CACV,WAAW;CACX,eAAe;CACf,QAAQ;AACV;AAEA,MAAM,mBAAmB,cACvB,GAAG,iBAAiB,wOAAwO,UACzP,KAAK,CAAC,MAAM,UAAU,wBAAwB,KAAK,iBAAiB,KAAK,IAAI,EAC7E,KAAK,EAAE,EAAE;AAEd,MAAM,oBAAoB,YACxB,GAAG,iBAAiB,sFAAsF,QACvG,KAAK,CAAC,IAAI,MAAM,YAAY,qBAAqB,GAAG,UAAU,KAAK,YAAY,OAAO,IAAI,EAC1F,KAAK,EAAE,EAAE;;AAGd,MAAM,WAAW,OAAkB,SAA6B;CAC9D,MAAM,OAAO,SAAS,KAAA,IAAY,KAAA,IAAY,MAAM,KAAK,IAAI;CAE7D,OAAO,SAAS,KAAA,KAAa,KAAK,KAAK,YAAY,EAAE,SAAS,MAAM,IAAI,OAAO,KAAA;AACjF;;AAGA,MAAM,kBACJ,OACA,SACA,eACA,SACA,qBACG;CACH,KAAK,MAAM,gBAAgB,QAAQ,OAAO,GACxC,IAAI,CAAC,aAAa,YAAY,QAAQ,YAAY,GAChD,OAAO,QAAQ,OAAO,gBAAgB,eAAe,aAAa,MAAM,CAAC;CAG7E,OAAO,QAAQ,OAAO,gBAAgB;AACxC;AAEA,MAAM,kBACJ,IAAI,oBAAoB;CAAE,QAAQ;CAAQ,SAAS;AAA8B,CAAC;AAEpF,MAAM,sBACJ,IAAI,oBAAoB;CACtB,QAAQ;CACR,SACE;AACJ,CAAC;;AAGH,MAAM,gBAAgB,UAAsB;CAC1C,IAAI,sBAAsB,KAAK,GAAG,MAAM,cAAc;CAEtD,OAAO;AACT;;;;;;;;;;;AA8BA,MAAa,qBACX,OACA,cACiB;CACjB,MAAM,QAAQ,WAAW,KAAK;CAC9B,MAAM,WAAW,MAAM,KAAK,gBAAgB;CAE5C,IAAI,aAAa,KAAA,GACf,MAAM,IAAI,oBAAoB;EAAE,QAAQ;EAAQ,SAAS;CAA0B,CAAC;CAEtF,MAAM,oBAAoB,YAAY,gBAAgB;CAEtD,MAAM,wBAAwB,cAC5B,SAAS,MAAM,KAAK,qBAAqB,gBAAgB,CAAC,GAAG,KAAK,CACpE;CAEA,MAAM,QAAQ,kBAAkB,SAAS,OAAO,SAAS;CACzD,MAAM,QAAoC,OAAO,OAAO,IAAI;CAC5D,MAAM,YAA8C,CAAC;CACrD,MAAM,qBAA+D,CAAC;CACtE,MAAM,SAAmC,CAAC;CAC1C,MAAM,2BAAW,IAAI,IAAY;CAEjC,KAAK,MAAM,CAAC,UAAU,UAAU,MAAM,OAAO,QAAQ,GAAG;EACtD,MAAM,OAAO,mBAAmB,WAAW,EAAE;EAE7C,mBAAmB,KAAK;GACtB,MAAM,WAAW;GACjB,GAAG,qCAAqC;GACxC;EACF,CAAC;EAED,MAAM,eACJ,MAAM,mBAAmB,KAAA,IACrB,KAAA,IACA,sBAAsB,IAAI,MAAM,cAAc;EAEpD,MAAM,OACJ,iBAAiB,KAAA,KAAa,aAAa,YAAY,CAAC,QAAQ,cAAc,WAAW,IACrF,KAAA,IACA,QAAQ,OAAO,gBAAgB,mBAAmB,aAAa,MAAM,CAAC;EAE5E,IAAI,SAAS,KAAA,GAAW;GACtB,OAAO,KAAK;IAAE,MAAM,MAAM;IAAM,MAAM,KAAA;GAAU,CAAC;GACjD;EACF;EAGA,IAAI,SAAS,IAAI,KAAK,IAAI,GAAG,MAAM,UAAU;EAE7C,SAAS,IAAI,KAAK,IAAI;EACtB,MAAM,MAAM,UAAU,aAAa,KAAK,KAAK;EAC7C,UAAU,KAAK,CAAC,MAAM,QAAQ,YAAY,SAAS,CAAC;EACpD,OAAO,KAAK;GAAE,MAAM,MAAM;GAAM,MAAM,KAAK;EAAK,CAAC;CACnD;CAEA,MAAM,gBAAgB,eACpB,OACA,uBACA,oBACA,iBAAgB,QAAQ,cAAc,eAAe,GACrD,sBACF;CAEA,IAAI,kBAAkB,KAAA,GAAW;EAC/B,MAAM,0BAA0B,aAAa,cAAc,KAAK;EAChE,UAAU,KAAK,CAAC,wBAAwB,YAAY,aAAa,CAAC;CACpE;CAEA,MAAM,SAAS,eACb,OACA,uBACA,oBACA,iBAAgB,QAAQ,cAAc,QAAQ,GAC9C,eACF;CAEA,IAAI,WAAW,KAAA,GAAW;EACxB,MAAM,mBAAmB,QAAQ,UAAU,WAAW,OAAO,KAAK,CAAC,CAAC;EACpE,UAAU,KAAK,CAAC,iBAAiB,YAAY,MAAM,CAAC;CACtD;CAEA,MAAM,OAAO,eACX,OACA,cAAc,SAAS,MAAM,KAAK,aAAa,GAAG,KAAK,CAAC,GACxD,KACA,iBAAgB,aAAa,SAAS,KAAA,KAAa,oBAAoB,IAAI,aAAa,IAAI,GAC5F,mBACF;CAEA,MAAM,UAAU;EACd,uBAAuB,QACrB,gBAAgB,CAAC,CAAC,kBAAkB,YAAY,QAAQ,GAAG,GAAG,SAAS,CAAC,CAC1E;EACA,eAAe,QACb,iBAAiB,CACf;GAAC;GAAQ,GAAG,qCAAqC;GAAkB;EAAgB,CACrF,CAAC,CACH;GACC,mBAAmB,QAAQ,YAAY,KAAK,CAAC;EAC9C,8BAA8B,QAAQ,iBAAiB,kBAAkB,CAAC;EAC1E,GAAG;CACL;CAEA,OAAO;EACL,SAAS,QAAQ,SAAS,EAAE,OAAO,EAAE,CAAC;EACtC,OAAO,OAAO,KAAK,OAAO;EAC1B;EACA,OAAO,UAAU,MAAM,KAAK;CAC9B;AACF"}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
//#region src/node/xlsx-styles.d.ts
|
|
2
|
+
/**
|
|
3
|
+
* `xl/styles.xml` is never handed to SheetJS as uploaded. With `cellText: true`, SheetJS formats
|
|
4
|
+
* every styled cell by re-parsing its number format (`SSF_format`, work proportional to the
|
|
5
|
+
* format's length, a quoted literal built one character at a time), so one huge format reused by
|
|
6
|
+
* many cells amplifies before any budget runs. The extractor writes a minimal stylesheet with
|
|
7
|
+
* only what display text needs: custom number formats (`numFmtId`, `formatCode`) of at most
|
|
8
|
+
* `maxNumberFormatCharacters`, and one `<xf numFmtId>` per source cell format, in order, so cell
|
|
9
|
+
* `s` indexes keep their meaning. Fonts, fills, borders, cell styles, and dxfs are left out:
|
|
10
|
+
* SheetJS parses fonts, fills, and borders whatever the options but uses them only with
|
|
11
|
+
* `cellStyles`, so leaving them out also removes their parsers from the input.
|
|
12
|
+
*/
|
|
13
|
+
/** Excel's own limit on a number format code, counted after unescaping. */
|
|
14
|
+
declare const maxNumberFormatCharacters = 255;
|
|
15
|
+
/** Custom number formats kept; later ones are dropped (their cells show General). */
|
|
16
|
+
declare const maxCustomNumberFormats = 1000;
|
|
17
|
+
/** Excel's limit on cell formats (`cellXfs`); later ones are dropped (General). */
|
|
18
|
+
declare const maxCellFormats = 64000;
|
|
19
|
+
type StylesSummary = {
|
|
20
|
+
/** Custom number formats in document order, as `[numFmtId, formatCode]`. */readonly numberFormats: ReadonlyArray<readonly [number, string]>; /** The `numFmtId` of each cell format in order, or `undefined` without a `cellXfs` element. */
|
|
21
|
+
readonly cellFormats: ReadonlyArray<number> | undefined;
|
|
22
|
+
};
|
|
23
|
+
/**
|
|
24
|
+
* Read the number formats and cell formats SheetJS would read from a stylesheet (same comment and
|
|
25
|
+
* doctype removal, same regions, same tag grammar), bounded by the caps above. A format longer
|
|
26
|
+
* than `maxNumberFormatCharacters` after unescaping, holding CDATA, or with an invalid id is
|
|
27
|
+
* dropped; a cell format without a valid `numFmtId` becomes General (0).
|
|
28
|
+
*/
|
|
29
|
+
declare const readStyles: (content: Uint8Array) => StylesSummary;
|
|
30
|
+
/** The generated stylesheet SheetJS reads instead of the uploaded one. */
|
|
31
|
+
declare const stylesXml: ({
|
|
32
|
+
numberFormats,
|
|
33
|
+
cellFormats
|
|
34
|
+
}: StylesSummary) => string;
|
|
35
|
+
//#endregion
|
|
36
|
+
export { StylesSummary, maxCellFormats, maxCustomNumberFormats, maxNumberFormatCharacters, readStyles, stylesXml };
|
|
37
|
+
//# sourceMappingURL=xlsx-styles.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-styles.d.mts","names":[],"sources":["../../src/node/xlsx-styles.ts"],"mappings":";;AAuBA;;;;AAAsC;AAGtC;;;;AAAmC;AAGnC;AAAA,cANa,yBAAA;;cAGA,sBAAA;AAGc;AAAA,cAAd,cAAA;AAAA,KAyDD,aAAA;uFAED,aAAA,EAAe,aAAA,6BAAf;EAAA,SAEA,WAAA,EAAa,aAAa;AAAA;;;AAAA;AASrC;;;cAAa,UAAA,GAAc,OAAA,EAAS,UAAA,KAAa,aAyChD;;cAGY,SAAA;EAAa,aAAA;EAAA;AAAA,GAAgC,aAAA"}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
import { sheetJsAttributeEscape, sheetJsAttributeText, sheetJsTags, sheetJsXmlHeader, spreadsheetMainNamespace, stripSheetJsNamespace } from "./sheetjs-xml.mjs";
|
|
2
|
+
import { Buffer } from "node:buffer";
|
|
3
|
+
//#region src/node/xlsx-styles.ts
|
|
4
|
+
/**
|
|
5
|
+
* `xl/styles.xml` is never handed to SheetJS as uploaded. With `cellText: true`, SheetJS formats
|
|
6
|
+
* every styled cell by re-parsing its number format (`SSF_format`, work proportional to the
|
|
7
|
+
* format's length, a quoted literal built one character at a time), so one huge format reused by
|
|
8
|
+
* many cells amplifies before any budget runs. The extractor writes a minimal stylesheet with
|
|
9
|
+
* only what display text needs: custom number formats (`numFmtId`, `formatCode`) of at most
|
|
10
|
+
* `maxNumberFormatCharacters`, and one `<xf numFmtId>` per source cell format, in order, so cell
|
|
11
|
+
* `s` indexes keep their meaning. Fonts, fills, borders, cell styles, and dxfs are left out:
|
|
12
|
+
* SheetJS parses fonts, fills, and borders whatever the options but uses them only with
|
|
13
|
+
* `cellStyles`, so leaving them out also removes their parsers from the input.
|
|
14
|
+
*/
|
|
15
|
+
/** Excel's own limit on a number format code, counted after unescaping. */
|
|
16
|
+
const maxNumberFormatCharacters = 255;
|
|
17
|
+
/** Custom number formats kept; later ones are dropped (their cells show General). */
|
|
18
|
+
const maxCustomNumberFormats = 1e3;
|
|
19
|
+
/** Excel's limit on cell formats (`cellXfs`); later ones are dropped (General). */
|
|
20
|
+
const maxCellFormats = 64e3;
|
|
21
|
+
/** SheetJS `str_remove_ng(text, '<!--', '-->')`, including its unterminated-comment behavior. */
|
|
22
|
+
const removeComments = (text) => {
|
|
23
|
+
let start = text.indexOf("<!--");
|
|
24
|
+
if (start === -1) return text;
|
|
25
|
+
const output = [];
|
|
26
|
+
let last = 0;
|
|
27
|
+
while (start > -1) {
|
|
28
|
+
output.push(text.slice(last, start));
|
|
29
|
+
const end = text.indexOf("-->", start + 4);
|
|
30
|
+
if (end === -1) break;
|
|
31
|
+
last = end + 3;
|
|
32
|
+
start = text.indexOf("<!--", last);
|
|
33
|
+
if (start === -1) output.push(text.slice(last));
|
|
34
|
+
}
|
|
35
|
+
return output.join("");
|
|
36
|
+
};
|
|
37
|
+
/** SheetJS `remove_doctype`. */
|
|
38
|
+
const removeDoctype = (text) => {
|
|
39
|
+
const doctype = text.slice(0, 1024).indexOf("<!DOCTYPE");
|
|
40
|
+
if (doctype === -1) return text;
|
|
41
|
+
const element = /<\w/.exec(text);
|
|
42
|
+
return element === null ? text : text.slice(0, doctype) + text.slice(element.index);
|
|
43
|
+
};
|
|
44
|
+
/** SheetJS `str_match_xml_ns`: the first `<tag>` start tag through the next `</tag>` (any prefix). */
|
|
45
|
+
const sheetJsRegion = (text, tag) => {
|
|
46
|
+
const start = new RegExp(`<(?:\\w+:)?${tag}\\b[^<>]*>`, "g");
|
|
47
|
+
const end = new RegExp(`</(?:\\w+:)?${tag}>`, "g");
|
|
48
|
+
const open = start.exec(text);
|
|
49
|
+
if (open === null) return void 0;
|
|
50
|
+
end.lastIndex = start.lastIndex;
|
|
51
|
+
return end.exec(text) === null ? void 0 : text.slice(open.index, end.lastIndex);
|
|
52
|
+
};
|
|
53
|
+
const numberFormatId = (raw) => {
|
|
54
|
+
const id = Number.parseInt(raw ?? "", 10);
|
|
55
|
+
return Number.isSafeInteger(id) && id >= 0 ? id : void 0;
|
|
56
|
+
};
|
|
57
|
+
/**
|
|
58
|
+
* Read the number formats and cell formats SheetJS would read from a stylesheet (same comment and
|
|
59
|
+
* doctype removal, same regions, same tag grammar), bounded by the caps above. A format longer
|
|
60
|
+
* than `maxNumberFormatCharacters` after unescaping, holding CDATA, or with an invalid id is
|
|
61
|
+
* dropped; a cell format without a valid `numFmtId` becomes General (0).
|
|
62
|
+
*/
|
|
63
|
+
const readStyles = (content) => {
|
|
64
|
+
const text = removeDoctype(removeComments(Buffer.from(content.buffer, content.byteOffset, content.byteLength).toString("latin1")));
|
|
65
|
+
const numberFormats = [];
|
|
66
|
+
const formatsRegion = sheetJsRegion(text, "numFmts");
|
|
67
|
+
if (formatsRegion !== void 0) for (const tag of sheetJsTags(formatsRegion)) {
|
|
68
|
+
if (numberFormats.length >= 1e3) break;
|
|
69
|
+
if (stripSheetJsNamespace(tag.head) !== "<numFmt") continue;
|
|
70
|
+
const id = numberFormatId(tag.attributes.get("numFmtId"));
|
|
71
|
+
const raw = tag.attributes.get("formatCode");
|
|
72
|
+
const code = raw === void 0 ? void 0 : sheetJsAttributeText(raw);
|
|
73
|
+
if (id !== void 0 && code !== void 0 && code.length <= 255) numberFormats.push([id, code]);
|
|
74
|
+
}
|
|
75
|
+
const cellFormatsRegion = sheetJsRegion(text, "cellXfs");
|
|
76
|
+
if (cellFormatsRegion === void 0) return {
|
|
77
|
+
numberFormats,
|
|
78
|
+
cellFormats: void 0
|
|
79
|
+
};
|
|
80
|
+
const cellFormats = [];
|
|
81
|
+
for (const tag of sheetJsTags(cellFormatsRegion)) {
|
|
82
|
+
if (cellFormats.length >= 64e3) break;
|
|
83
|
+
const head = stripSheetJsNamespace(tag.head);
|
|
84
|
+
if (head === "<xf" || head === "<xf/>" || head === "<xf>") cellFormats.push(numberFormatId(tag.attributes.get("numFmtId")) ?? 0);
|
|
85
|
+
}
|
|
86
|
+
return {
|
|
87
|
+
numberFormats,
|
|
88
|
+
cellFormats
|
|
89
|
+
};
|
|
90
|
+
};
|
|
91
|
+
/** The generated stylesheet SheetJS reads instead of the uploaded one. */
|
|
92
|
+
const stylesXml = ({ numberFormats, cellFormats }) => `${sheetJsXmlHeader}<styleSheet xmlns="${spreadsheetMainNamespace}">${numberFormats.length === 0 ? "" : `<numFmts count="${numberFormats.length}">${numberFormats.map(([id, code]) => `<numFmt numFmtId="${id}" formatCode="${sheetJsAttributeEscape(code)}"/>`).join("")}</numFmts>`}${cellFormats === void 0 ? "" : `<cellXfs count="${cellFormats.length}">${cellFormats.map((id) => `<xf numFmtId="${id}"/>`).join("")}</cellXfs>`}</styleSheet>`;
|
|
93
|
+
//#endregion
|
|
94
|
+
export { maxCellFormats, maxCustomNumberFormats, maxNumberFormatCharacters, readStyles, stylesXml };
|
|
95
|
+
|
|
96
|
+
//# sourceMappingURL=xlsx-styles.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-styles.mjs","names":[],"sources":["../../src/node/xlsx-styles.ts"],"sourcesContent":["import { Buffer } from 'node:buffer'\nimport {\n sheetJsAttributeEscape,\n sheetJsAttributeText,\n sheetJsTags,\n sheetJsXmlHeader,\n spreadsheetMainNamespace,\n stripSheetJsNamespace\n} from './sheetjs-xml.ts'\n\n/**\n * `xl/styles.xml` is never handed to SheetJS as uploaded. With `cellText: true`, SheetJS formats\n * every styled cell by re-parsing its number format (`SSF_format`, work proportional to the\n * format's length, a quoted literal built one character at a time), so one huge format reused by\n * many cells amplifies before any budget runs. The extractor writes a minimal stylesheet with\n * only what display text needs: custom number formats (`numFmtId`, `formatCode`) of at most\n * `maxNumberFormatCharacters`, and one `<xf numFmtId>` per source cell format, in order, so cell\n * `s` indexes keep their meaning. Fonts, fills, borders, cell styles, and dxfs are left out:\n * SheetJS parses fonts, fills, and borders whatever the options but uses them only with\n * `cellStyles`, so leaving them out also removes their parsers from the input.\n */\n\n/** Excel's own limit on a number format code, counted after unescaping. */\nexport const maxNumberFormatCharacters = 255\n\n/** Custom number formats kept; later ones are dropped (their cells show General). */\nexport const maxCustomNumberFormats = 1000\n\n/** Excel's limit on cell formats (`cellXfs`); later ones are dropped (General). */\nexport const maxCellFormats = 64_000\n\n/** SheetJS `str_remove_ng(text, '<!--', '-->')`, including its unterminated-comment behavior. */\nconst removeComments = (text: string) => {\n let start = text.indexOf('<!--')\n\n if (start === -1) return text\n\n const output: Array<string> = []\n let last = 0\n\n while (start > -1) {\n output.push(text.slice(last, start))\n\n const end = text.indexOf('-->', start + 4)\n\n if (end === -1) break\n\n last = end + 3\n start = text.indexOf('<!--', last)\n\n if (start === -1) output.push(text.slice(last))\n }\n\n return output.join('')\n}\n\n/** SheetJS `remove_doctype`. */\nconst removeDoctype = (text: string) => {\n const doctype = text.slice(0, 1024).indexOf('<!DOCTYPE')\n\n if (doctype === -1) return text\n\n const element = /<\\w/.exec(text)\n\n return element === null ? text : text.slice(0, doctype) + text.slice(element.index)\n}\n\n/** SheetJS `str_match_xml_ns`: the first `<tag>` start tag through the next `</tag>` (any prefix). */\nconst sheetJsRegion = (text: string, tag: string) => {\n const start = new RegExp(`<(?:\\\\w+:)?${tag}\\\\b[^<>]*>`, 'g')\n const end = new RegExp(`</(?:\\\\w+:)?${tag}>`, 'g')\n const open = start.exec(text)\n\n if (open === null) return undefined\n\n end.lastIndex = start.lastIndex\n\n return end.exec(text) === null ? undefined : text.slice(open.index, end.lastIndex)\n}\n\nconst numberFormatId = (raw: string | undefined) => {\n const id = Number.parseInt(raw ?? '', 10)\n\n return Number.isSafeInteger(id) && id >= 0 ? id : undefined\n}\n\nexport type StylesSummary = {\n /** Custom number formats in document order, as `[numFmtId, formatCode]`. */\n readonly numberFormats: ReadonlyArray<readonly [number, string]>\n /** The `numFmtId` of each cell format in order, or `undefined` without a `cellXfs` element. */\n readonly cellFormats: ReadonlyArray<number> | undefined\n}\n\n/**\n * Read the number formats and cell formats SheetJS would read from a stylesheet (same comment and\n * doctype removal, same regions, same tag grammar), bounded by the caps above. A format longer\n * than `maxNumberFormatCharacters` after unescaping, holding CDATA, or with an invalid id is\n * dropped; a cell format without a valid `numFmtId` becomes General (0).\n */\nexport const readStyles = (content: Uint8Array): StylesSummary => {\n const text = removeDoctype(\n removeComments(\n Buffer.from(content.buffer, content.byteOffset, content.byteLength).toString('latin1')\n )\n )\n\n const numberFormats: Array<readonly [number, string]> = []\n const formatsRegion = sheetJsRegion(text, 'numFmts')\n\n if (formatsRegion !== undefined) {\n for (const tag of sheetJsTags(formatsRegion)) {\n if (numberFormats.length >= maxCustomNumberFormats) break\n\n if (stripSheetJsNamespace(tag.head) !== '<numFmt') continue\n\n const id = numberFormatId(tag.attributes.get('numFmtId'))\n const raw = tag.attributes.get('formatCode')\n const code = raw === undefined ? undefined : sheetJsAttributeText(raw)\n\n if (id !== undefined && code !== undefined && code.length <= maxNumberFormatCharacters)\n numberFormats.push([id, code])\n }\n }\n\n const cellFormatsRegion = sheetJsRegion(text, 'cellXfs')\n\n if (cellFormatsRegion === undefined) return { numberFormats, cellFormats: undefined }\n\n const cellFormats: Array<number> = []\n\n for (const tag of sheetJsTags(cellFormatsRegion)) {\n if (cellFormats.length >= maxCellFormats) break\n\n const head = stripSheetJsNamespace(tag.head)\n\n if (head === '<xf' || head === '<xf/>' || head === '<xf>')\n cellFormats.push(numberFormatId(tag.attributes.get('numFmtId')) ?? 0)\n }\n\n return { numberFormats, cellFormats }\n}\n\n/** The generated stylesheet SheetJS reads instead of the uploaded one. */\nexport const stylesXml = ({ numberFormats, cellFormats }: StylesSummary) =>\n `${sheetJsXmlHeader}<styleSheet xmlns=\"${spreadsheetMainNamespace}\">${\n numberFormats.length === 0\n ? ''\n : `<numFmts count=\"${numberFormats.length}\">${numberFormats\n .map(\n ([id, code]) =>\n `<numFmt numFmtId=\"${id}\" formatCode=\"${sheetJsAttributeEscape(code)}\"/>`\n )\n .join('')}</numFmts>`\n }${\n cellFormats === undefined\n ? ''\n : `<cellXfs count=\"${cellFormats.length}\">${cellFormats\n .map(id => `<xf numFmtId=\"${id}\"/>`)\n .join('')}</cellXfs>`\n }</styleSheet>`\n"],"mappings":";;;;;;;;;;;;;;;AAuBA,MAAa,4BAA4B;;AAGzC,MAAa,yBAAyB;;AAGtC,MAAa,iBAAiB;;AAG9B,MAAM,kBAAkB,SAAiB;CACvC,IAAI,QAAQ,KAAK,QAAQ,MAAM;CAE/B,IAAI,UAAU,IAAI,OAAO;CAEzB,MAAM,SAAwB,CAAC;CAC/B,IAAI,OAAO;CAEX,OAAO,QAAQ,IAAI;EACjB,OAAO,KAAK,KAAK,MAAM,MAAM,KAAK,CAAC;EAEnC,MAAM,MAAM,KAAK,QAAQ,OAAO,QAAQ,CAAC;EAEzC,IAAI,QAAQ,IAAI;EAEhB,OAAO,MAAM;EACb,QAAQ,KAAK,QAAQ,QAAQ,IAAI;EAEjC,IAAI,UAAU,IAAI,OAAO,KAAK,KAAK,MAAM,IAAI,CAAC;CAChD;CAEA,OAAO,OAAO,KAAK,EAAE;AACvB;;AAGA,MAAM,iBAAiB,SAAiB;CACtC,MAAM,UAAU,KAAK,MAAM,GAAG,IAAI,EAAE,QAAQ,WAAW;CAEvD,IAAI,YAAY,IAAI,OAAO;CAE3B,MAAM,UAAU,MAAM,KAAK,IAAI;CAE/B,OAAO,YAAY,OAAO,OAAO,KAAK,MAAM,GAAG,OAAO,IAAI,KAAK,MAAM,QAAQ,KAAK;AACpF;;AAGA,MAAM,iBAAiB,MAAc,QAAgB;CACnD,MAAM,QAAQ,IAAI,OAAO,cAAc,IAAI,aAAa,GAAG;CAC3D,MAAM,MAAM,IAAI,OAAO,eAAe,IAAI,IAAI,GAAG;CACjD,MAAM,OAAO,MAAM,KAAK,IAAI;CAE5B,IAAI,SAAS,MAAM,OAAO,KAAA;CAE1B,IAAI,YAAY,MAAM;CAEtB,OAAO,IAAI,KAAK,IAAI,MAAM,OAAO,KAAA,IAAY,KAAK,MAAM,KAAK,OAAO,IAAI,SAAS;AACnF;AAEA,MAAM,kBAAkB,QAA4B;CAClD,MAAM,KAAK,OAAO,SAAS,OAAO,IAAI,EAAE;CAExC,OAAO,OAAO,cAAc,EAAE,KAAK,MAAM,IAAI,KAAK,KAAA;AACpD;;;;;;;AAeA,MAAa,cAAc,YAAuC;CAChE,MAAM,OAAO,cACX,eACE,OAAO,KAAK,QAAQ,QAAQ,QAAQ,YAAY,QAAQ,UAAU,EAAE,SAAS,QAAQ,CACvF,CACF;CAEA,MAAM,gBAAkD,CAAC;CACzD,MAAM,gBAAgB,cAAc,MAAM,SAAS;CAEnD,IAAI,kBAAkB,KAAA,GACpB,KAAK,MAAM,OAAO,YAAY,aAAa,GAAG;EAC5C,IAAI,cAAc,UAAA,KAAkC;EAEpD,IAAI,sBAAsB,IAAI,IAAI,MAAM,WAAW;EAEnD,MAAM,KAAK,eAAe,IAAI,WAAW,IAAI,UAAU,CAAC;EACxD,MAAM,MAAM,IAAI,WAAW,IAAI,YAAY;EAC3C,MAAM,OAAO,QAAQ,KAAA,IAAY,KAAA,IAAY,qBAAqB,GAAG;EAErE,IAAI,OAAO,KAAA,KAAa,SAAS,KAAA,KAAa,KAAK,UAAA,KACjD,cAAc,KAAK,CAAC,IAAI,IAAI,CAAC;CACjC;CAGF,MAAM,oBAAoB,cAAc,MAAM,SAAS;CAEvD,IAAI,sBAAsB,KAAA,GAAW,OAAO;EAAE;EAAe,aAAa,KAAA;CAAU;CAEpF,MAAM,cAA6B,CAAC;CAEpC,KAAK,MAAM,OAAO,YAAY,iBAAiB,GAAG;EAChD,IAAI,YAAY,UAAA,MAA0B;EAE1C,MAAM,OAAO,sBAAsB,IAAI,IAAI;EAE3C,IAAI,SAAS,SAAS,SAAS,WAAW,SAAS,QACjD,YAAY,KAAK,eAAe,IAAI,WAAW,IAAI,UAAU,CAAC,KAAK,CAAC;CACxE;CAEA,OAAO;EAAE;EAAe;CAAY;AACtC;;AAGA,MAAa,aAAa,EAAE,eAAe,kBACzC,GAAG,iBAAiB,qBAAqB,yBAAyB,IAChE,cAAc,WAAW,IACrB,KACA,mBAAmB,cAAc,OAAO,IAAI,cACzC,KACE,CAAC,IAAI,UACJ,qBAAqB,GAAG,gBAAgB,uBAAuB,IAAI,EAAE,IACzE,EACC,KAAK,EAAE,EAAE,cAEhB,gBAAgB,KAAA,IACZ,KACA,mBAAmB,YAAY,OAAO,IAAI,YACvC,KAAI,OAAM,iBAAiB,GAAG,IAAI,EAClC,KAAK,EAAE,EAAE,YACjB"}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { FileExtractorLimits } from "../limits.mjs";
|
|
2
|
+
import { XlsxHyperlinks } from "./xlsx-hyperlinks.mjs";
|
|
3
|
+
|
|
4
|
+
//#region src/node/xlsx-text.d.ts
|
|
5
|
+
type XlsxTextLimits = Pick<FileExtractorLimits, 'maxXlsxCellVisits' | 'maxXlsxSheets' | 'maxXlsxTextCharacters'>;
|
|
6
|
+
/** The parsed workbook as SheetJS returns it; sheets and cells are read as own properties. */
|
|
7
|
+
type XlsxWorkbook = {
|
|
8
|
+
readonly SheetNames: ReadonlyArray<string>;
|
|
9
|
+
readonly Sheets: object;
|
|
10
|
+
};
|
|
11
|
+
type XlsxTextOptions = {
|
|
12
|
+
readonly hyperlinks?: XlsxHyperlinks; /** Some hyperlinks were never read because the workbook exceeded `maxXlsxHyperlinks`. */
|
|
13
|
+
readonly hyperlinksTruncated?: boolean;
|
|
14
|
+
};
|
|
15
|
+
/**
|
|
16
|
+
* Space reserved for the marker whenever a workbook has hyperlinks, so it always fits.
|
|
17
|
+
* `maxXlsxTextCharacters` must exceed it (`minimumXlsxTextCharacters`).
|
|
18
|
+
*/
|
|
19
|
+
declare const omittedHyperlinksMarkerReserve: number;
|
|
20
|
+
/**
|
|
21
|
+
* Validate every sheet before touching any cell, then generate bounded CSV incrementally, never
|
|
22
|
+
* with `sheet_to_csv` (which can walk billions of absent cells or allocate an unbounded quoted
|
|
23
|
+
* string).
|
|
24
|
+
*
|
|
25
|
+
* Existing cells inside an external hyperlink are written as `text <url>`. Plain text is
|
|
26
|
+
* rendered first, so links only use budget the plain text leaves over: in document order, an
|
|
27
|
+
* annotation that no longer fits leaves its cell plain and one marker ends the output.
|
|
28
|
+
*/
|
|
29
|
+
declare const extractBoundedXlsxText: (workbook: XlsxWorkbook, limits: XlsxTextLimits, options?: XlsxTextOptions) => string;
|
|
30
|
+
//#endregion
|
|
31
|
+
export { XlsxTextLimits, XlsxTextOptions, XlsxWorkbook, extractBoundedXlsxText, omittedHyperlinksMarkerReserve };
|
|
32
|
+
//# sourceMappingURL=xlsx-text.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"xlsx-text.d.mts","names":[],"sources":["../../src/node/xlsx-text.ts"],"mappings":";;;;KAQY,cAAA,GAAiB,IAAI,CAC/B,mBAAA;;KAKU,YAAA;EAAA,SACD,UAAA,EAAY,aAAa;EAAA,SACzB,MAAA;AAAA;AAAA,KAGC,eAAA;EAAA,SACD,UAAA,GAAa,cAAc,EANd;EAAA,SAQb,mBAAA;AAAA;;;;;cAcE,8BAAA;AAjBb;;;;;;;;AAG8B;AAH9B,cAmLa,sBAAA,GACX,QAAA,EAAU,YAAA,EACV,MAAA,EAAQ,cAAA,EACR,OAAA,GAAS,eAAA"}
|