@yolk-sdk/extractors 0.1.0-canary.98
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +318 -0
- package/dist/errors.d.mts +60 -0
- package/dist/errors.d.mts.map +1 -0
- package/dist/errors.mjs +69 -0
- package/dist/errors.mjs.map +1 -0
- package/dist/format.d.mts +32 -0
- package/dist/format.d.mts.map +1 -0
- package/dist/format.mjs +52 -0
- package/dist/format.mjs.map +1 -0
- package/dist/index.d.mts +6 -0
- package/dist/index.mjs +6 -0
- package/dist/knowledge.d.mts +17 -0
- package/dist/knowledge.d.mts.map +1 -0
- package/dist/knowledge.mjs +77 -0
- package/dist/knowledge.mjs.map +1 -0
- package/dist/limits.d.mts +28 -0
- package/dist/limits.d.mts.map +1 -0
- package/dist/limits.mjs +43 -0
- package/dist/limits.mjs.map +1 -0
- package/dist/node/extract-file.d.mts +31 -0
- package/dist/node/extract-file.d.mts.map +1 -0
- package/dist/node/extract-file.mjs +183 -0
- package/dist/node/extract-file.mjs.map +1 -0
- package/dist/node/extraction-isolation.d.mts +94 -0
- package/dist/node/extraction-isolation.d.mts.map +1 -0
- package/dist/node/extraction-isolation.mjs +155 -0
- package/dist/node/extraction-isolation.mjs.map +1 -0
- package/dist/node/extraction-worker-protocol.d.mts +60 -0
- package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
- package/dist/node/extraction-worker-protocol.mjs +105 -0
- package/dist/node/extraction-worker-protocol.mjs.map +1 -0
- package/dist/node/extraction-worker.d.mts +1 -0
- package/dist/node/extraction-worker.mjs +114729 -0
- package/dist/node/index.d.mts +6 -0
- package/dist/node/index.mjs +5 -0
- package/dist/node/live-layer.d.mts +36 -0
- package/dist/node/live-layer.d.mts.map +1 -0
- package/dist/node/live-layer.mjs +70 -0
- package/dist/node/live-layer.mjs.map +1 -0
- package/dist/node/office-archive.d.mts +51 -0
- package/dist/node/office-archive.d.mts.map +1 -0
- package/dist/node/office-archive.mjs +193 -0
- package/dist/node/office-archive.mjs.map +1 -0
- package/dist/node/pptx-text.d.mts +6 -0
- package/dist/node/pptx-text.d.mts.map +1 -0
- package/dist/node/pptx-text.mjs +63 -0
- package/dist/node/pptx-text.mjs.map +1 -0
- package/dist/node/sheetjs-xml.d.mts +89 -0
- package/dist/node/sheetjs-xml.d.mts.map +1 -0
- package/dist/node/sheetjs-xml.mjs +253 -0
- package/dist/node/sheetjs-xml.mjs.map +1 -0
- package/dist/node/sheetjs.d.mts +62 -0
- package/dist/node/sheetjs.d.mts.map +1 -0
- package/dist/node/sheetjs.mjs +122 -0
- package/dist/node/sheetjs.mjs.map +1 -0
- package/dist/node/worker-admission.d.mts +58 -0
- package/dist/node/worker-admission.d.mts.map +1 -0
- package/dist/node/worker-admission.mjs +107 -0
- package/dist/node/worker-admission.mjs.map +1 -0
- package/dist/node/xlsx-hyperlinks.d.mts +34 -0
- package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
- package/dist/node/xlsx-hyperlinks.mjs +159 -0
- package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
- package/dist/node/xlsx-parts.d.mts +29 -0
- package/dist/node/xlsx-parts.d.mts.map +1 -0
- package/dist/node/xlsx-parts.mjs +49 -0
- package/dist/node/xlsx-parts.mjs.map +1 -0
- package/dist/node/xlsx-range.d.mts +21 -0
- package/dist/node/xlsx-range.d.mts.map +1 -0
- package/dist/node/xlsx-range.mjs +49 -0
- package/dist/node/xlsx-range.mjs.map +1 -0
- package/dist/node/xlsx-routing.d.mts +36 -0
- package/dist/node/xlsx-routing.d.mts.map +1 -0
- package/dist/node/xlsx-routing.mjs +115 -0
- package/dist/node/xlsx-routing.mjs.map +1 -0
- package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
- package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
- package/dist/node/xlsx-sheetjs-input.mjs +165 -0
- package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
- package/dist/node/xlsx-styles.d.mts +37 -0
- package/dist/node/xlsx-styles.d.mts.map +1 -0
- package/dist/node/xlsx-styles.mjs +96 -0
- package/dist/node/xlsx-styles.mjs.map +1 -0
- package/dist/node/xlsx-text.d.mts +32 -0
- package/dist/node/xlsx-text.d.mts.map +1 -0
- package/dist/node/xlsx-text.mjs +181 -0
- package/dist/node/xlsx-text.mjs.map +1 -0
- package/dist/node/xlsx-workbook.d.mts +32 -0
- package/dist/node/xlsx-workbook.d.mts.map +1 -0
- package/dist/node/xlsx-workbook.mjs +70 -0
- package/dist/node/xlsx-workbook.mjs.map +1 -0
- package/dist/node/xml-text.d.mts +12 -0
- package/dist/node/xml-text.d.mts.map +1 -0
- package/dist/node/xml-text.mjs +51 -0
- package/dist/node/xml-text.mjs.map +1 -0
- package/dist/sanitize.d.mts +6 -0
- package/dist/sanitize.d.mts.map +1 -0
- package/dist/sanitize.mjs +11 -0
- package/dist/sanitize.mjs.map +1 -0
- package/dist/service.d.mts +22 -0
- package/dist/service.d.mts.map +1 -0
- package/dist/service.mjs +11 -0
- package/dist/service.mjs.map +1 -0
- package/package.json +87 -0
- package/src/errors.ts +96 -0
- package/src/format.ts +84 -0
- package/src/index.ts +32 -0
- package/src/knowledge.ts +101 -0
- package/src/limits.ts +49 -0
- package/src/node/extract-file.ts +269 -0
- package/src/node/extraction-isolation.ts +289 -0
- package/src/node/extraction-worker-protocol.ts +130 -0
- package/src/node/extraction-worker.ts +56 -0
- package/src/node/index.ts +21 -0
- package/src/node/live-layer.ts +136 -0
- package/src/node/office-archive.ts +368 -0
- package/src/node/pptx-text.ts +125 -0
- package/src/node/sheetjs-xml.ts +356 -0
- package/src/node/sheetjs.ts +177 -0
- package/src/node/worker-admission.ts +162 -0
- package/src/node/xlsx-hyperlinks.ts +260 -0
- package/src/node/xlsx-parts.ts +83 -0
- package/src/node/xlsx-range.ts +70 -0
- package/src/node/xlsx-routing.ts +171 -0
- package/src/node/xlsx-sheetjs-input.ts +275 -0
- package/src/node/xlsx-styles.ts +160 -0
- package/src/node/xlsx-text.ts +288 -0
- package/src/node/xlsx-workbook.ts +133 -0
- package/src/node/xml-text.ts +77 -0
- package/src/sanitize.ts +18 -0
- package/src/service.ts +21 -0
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { FileExtractor, FileExtractorApi } from "../service.mjs";
|
|
2
|
+
import { OfficeArchiveLimits, normalizeOfficeArchive } from "./office-archive.mjs";
|
|
3
|
+
import { SheetJsLoader } from "./sheetjs.mjs";
|
|
4
|
+
import { FileExtractorIsolation, WorkerIsolationOptions, defaultWorkerIsolation } from "./extraction-isolation.mjs";
|
|
5
|
+
import { FileExtractorLayer, FileExtractorOptions, makeFileExtractorLayer } from "./live-layer.mjs";
|
|
6
|
+
export { FileExtractor, type FileExtractorApi, type FileExtractorIsolation, FileExtractorLayer, type FileExtractorOptions, type OfficeArchiveLimits, type SheetJsLoader, type WorkerIsolationOptions, defaultWorkerIsolation, makeFileExtractorLayer, normalizeOfficeArchive };
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import { FileExtractor } from "../service.mjs";
|
|
2
|
+
import { normalizeOfficeArchive } from "./office-archive.mjs";
|
|
3
|
+
import { defaultWorkerIsolation } from "./extraction-isolation.mjs";
|
|
4
|
+
import { FileExtractorLayer, makeFileExtractorLayer } from "./live-layer.mjs";
|
|
5
|
+
export { FileExtractor, FileExtractorLayer, defaultWorkerIsolation, makeFileExtractorLayer, normalizeOfficeArchive };
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { FileExtractorLimits } from "../limits.mjs";
|
|
2
|
+
import { FileExtractor } from "../service.mjs";
|
|
3
|
+
import { SheetJsLoader } from "./sheetjs.mjs";
|
|
4
|
+
import { FileExtractorIsolation } from "./extraction-isolation.mjs";
|
|
5
|
+
import { Layer } from "effect";
|
|
6
|
+
|
|
7
|
+
//#region src/node/live-layer.d.ts
|
|
8
|
+
type FileExtractorOptions = {
|
|
9
|
+
/** Override any default limit; see `defaultFileExtractorLimits`. */readonly limits?: Partial<FileExtractorLimits>;
|
|
10
|
+
/**
|
|
11
|
+
* Where parsers run (default `'worker'`): each PDF, DOCX, XLSX, and PPTX extraction runs in a
|
|
12
|
+
* fresh worker thread with V8 heap, stack, and time limits, admitted through a pool of 4 workers
|
|
13
|
+
* per JavaScript realm; pass an object to change them. `'none'` parses in the calling thread and is
|
|
14
|
+
* unsafe for untrusted input.
|
|
15
|
+
*/
|
|
16
|
+
readonly isolation?: FileExtractorIsolation;
|
|
17
|
+
/**
|
|
18
|
+
* Load SheetJS in the calling thread. Only with `isolation: 'none'` (a worker cannot receive a
|
|
19
|
+
* function and always imports the installed `xlsx` itself); any other isolation is a defect
|
|
20
|
+
* when the layer is built. Defaults to a lazy `import('xlsx')`. Missing, non-SheetJS, or
|
|
21
|
+
* pre-0.20.3 modules fail with `SheetJsUnavailableError`.
|
|
22
|
+
*/
|
|
23
|
+
readonly loadSheetJs?: SheetJsLoader;
|
|
24
|
+
};
|
|
25
|
+
/**
|
|
26
|
+
* Node `FileExtractor` layer with custom limits, isolation, or (in-process) SheetJS loader.
|
|
27
|
+
* Invalid limits or isolation settings are defects.
|
|
28
|
+
*/
|
|
29
|
+
declare const makeFileExtractorLayer: (options?: FileExtractorOptions) => Layer.Layer<FileExtractor, never, never>;
|
|
30
|
+
/**
|
|
31
|
+
* Node `FileExtractor` layer with the default limits, worker isolation, and lazy SheetJS loading.
|
|
32
|
+
*/
|
|
33
|
+
declare const FileExtractorLayer: Layer.Layer<FileExtractor, never, never>;
|
|
34
|
+
//#endregion
|
|
35
|
+
export { FileExtractorLayer, FileExtractorOptions, makeFileExtractorLayer };
|
|
36
|
+
//# sourceMappingURL=live-layer.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"live-layer.d.mts","names":[],"sources":["../../src/node/live-layer.ts"],"mappings":";;;;;;;KAmBY,oBAAA;+EAED,MAAA,GAAS,OAAA,CAAQ,mBAAA;EAFI;;;;;;EAAA,SASrB,SAAA,GAAY,sBAAA;EAOe;;;;;;EAAA,SAA3B,WAAA,GAAc,aAAA;AAAA;;AAAa;AAkFtC;;cAAa,sBAAA,GAA0B,OAAA,GAAS,oBAAA,KAAyB,KAAA,CAAA,KAAA,CAAA,aAAA;;;;cAkB5D,kBAAA,EAAkB,KAAA,CAAA,KAAA,CAAA,aAAA"}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { FileExtractionError, UnsupportedFileFormatError } from "../errors.mjs";
|
|
2
|
+
import { fileFormatFor } from "../format.mjs";
|
|
3
|
+
import { FileExtractorLimits, defaultFileExtractorLimits } from "../limits.mjs";
|
|
4
|
+
import { FileExtractor } from "../service.mjs";
|
|
5
|
+
import { defaultSheetJsLoader } from "./sheetjs.mjs";
|
|
6
|
+
import { extractParsedFile, isParsedFileFormat, makeExtractedFile } from "./extract-file.mjs";
|
|
7
|
+
import { WorkerIsolationSettings, defaultExtractionWorkerUrl, defaultWorkerIsolation, makeWorkerExtractor } from "./extraction-isolation.mjs";
|
|
8
|
+
import { Effect, Layer } from "effect";
|
|
9
|
+
import * as Schema from "effect/Schema";
|
|
10
|
+
//#region src/node/live-layer.ts
|
|
11
|
+
const decodeText = (bytes) => new TextDecoder("utf-8", { fatal: false }).decode(bytes);
|
|
12
|
+
/** Build the Node `FileExtractor` from validated limits and the parser runner. */
|
|
13
|
+
const makeFileExtractor = (limits, extractParsed) => ({ extract: (input) => Effect.gen(function* () {
|
|
14
|
+
const format = fileFormatFor(input);
|
|
15
|
+
if (format === void 0) return yield* Effect.fail(new UnsupportedFileFormatError({
|
|
16
|
+
filename: input.filename,
|
|
17
|
+
mediaType: input.mediaType
|
|
18
|
+
}));
|
|
19
|
+
yield* Effect.annotateCurrentSpan({
|
|
20
|
+
"file_extractor.format": format,
|
|
21
|
+
"file_extractor.file_size": input.bytes.byteLength
|
|
22
|
+
});
|
|
23
|
+
if (input.bytes.byteLength > limits.maxInputBytes) return yield* Effect.fail(new FileExtractionError({
|
|
24
|
+
message: "File exceeds the extraction size limit",
|
|
25
|
+
format
|
|
26
|
+
}));
|
|
27
|
+
if (isParsedFileFormat(format)) return yield* extractParsed(input, format, limits);
|
|
28
|
+
return yield* makeExtractedFile(decodeText(input.bytes), {
|
|
29
|
+
format,
|
|
30
|
+
title: input.filename
|
|
31
|
+
});
|
|
32
|
+
}).pipe(Effect.withSpan("FileExtractor.extract")) });
|
|
33
|
+
const parsedFileExtractor = (options) => {
|
|
34
|
+
const isolation = options.isolation ?? "worker";
|
|
35
|
+
if (isolation === "none") {
|
|
36
|
+
const loader = options.loadSheetJs ?? defaultSheetJsLoader;
|
|
37
|
+
return Effect.succeed((input, format, limits) => extractParsedFile(input, format, limits, loader));
|
|
38
|
+
}
|
|
39
|
+
if (options.loadSheetJs !== void 0) return Effect.die(/* @__PURE__ */ new Error("FileExtractorOptions.loadSheetJs runs SheetJS in the calling thread and needs isolation: \"none\""));
|
|
40
|
+
const overrides = isolation === "worker" ? {} : isolation;
|
|
41
|
+
const timeoutMs = overrides.timeoutMs ?? defaultWorkerIsolation.timeoutMs;
|
|
42
|
+
return Schema.decodeUnknownEffect(WorkerIsolationSettings)({
|
|
43
|
+
maxOldGenerationSizeMb: overrides.maxOldGenerationSizeMb ?? defaultWorkerIsolation.maxOldGenerationSizeMb,
|
|
44
|
+
maxYoungGenerationSizeMb: overrides.maxYoungGenerationSizeMb ?? defaultWorkerIsolation.maxYoungGenerationSizeMb,
|
|
45
|
+
stackSizeMb: overrides.stackSizeMb ?? defaultWorkerIsolation.stackSizeMb,
|
|
46
|
+
timeoutMs,
|
|
47
|
+
maxConcurrentWorkers: overrides.maxConcurrentWorkers ?? defaultWorkerIsolation.maxConcurrentWorkers,
|
|
48
|
+
maxQueueWaitMs: overrides.maxQueueWaitMs ?? timeoutMs
|
|
49
|
+
}).pipe(Effect.flatMap((settings) => makeWorkerExtractor(settings, overrides.workerUrl ?? defaultExtractionWorkerUrl())));
|
|
50
|
+
};
|
|
51
|
+
/**
|
|
52
|
+
* Node `FileExtractor` layer with custom limits, isolation, or (in-process) SheetJS loader.
|
|
53
|
+
* Invalid limits or isolation settings are defects.
|
|
54
|
+
*/
|
|
55
|
+
const makeFileExtractorLayer = (options = {}) => Layer.effect(FileExtractor, Effect.gen(function* () {
|
|
56
|
+
const limits = yield* Schema.decodeUnknownEffect(FileExtractorLimits)({
|
|
57
|
+
...defaultFileExtractorLimits,
|
|
58
|
+
...options.limits
|
|
59
|
+
});
|
|
60
|
+
const extractParsed = yield* parsedFileExtractor(options);
|
|
61
|
+
return FileExtractor.of(makeFileExtractor(limits, extractParsed));
|
|
62
|
+
}).pipe(Effect.orDie));
|
|
63
|
+
/**
|
|
64
|
+
* Node `FileExtractor` layer with the default limits, worker isolation, and lazy SheetJS loading.
|
|
65
|
+
*/
|
|
66
|
+
const FileExtractorLayer = makeFileExtractorLayer();
|
|
67
|
+
//#endregion
|
|
68
|
+
export { FileExtractorLayer, makeFileExtractorLayer };
|
|
69
|
+
|
|
70
|
+
//# sourceMappingURL=live-layer.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"live-layer.mjs","names":[],"sources":["../../src/node/live-layer.ts"],"sourcesContent":["import { Effect, Layer } from 'effect'\nimport * as Schema from 'effect/Schema'\nimport { FileExtractionError, UnsupportedFileFormatError } from '../errors.ts'\nimport { fileFormatFor } from '../format.ts'\nimport { defaultFileExtractorLimits, FileExtractorLimits } from '../limits.ts'\nimport { FileExtractor } from '../service.ts'\nimport type { FileExtractorApi } from '../service.ts'\nimport { extractParsedFile, isParsedFileFormat, makeExtractedFile } from './extract-file.ts'\nimport type { ParsedFileFormat } from './extract-file.ts'\nimport {\n defaultExtractionWorkerUrl,\n defaultWorkerIsolation,\n makeWorkerExtractor,\n WorkerIsolationSettings\n} from './extraction-isolation.ts'\nimport type { FileExtractorIsolation, WorkerExtractor } from './extraction-isolation.ts'\nimport { defaultSheetJsLoader } from './sheetjs.ts'\nimport type { SheetJsLoader } from './sheetjs.ts'\n\nexport type FileExtractorOptions = {\n /** Override any default limit; see `defaultFileExtractorLimits`. */\n readonly limits?: Partial<FileExtractorLimits>\n /**\n * Where parsers run (default `'worker'`): each PDF, DOCX, XLSX, and PPTX extraction runs in a\n * fresh worker thread with V8 heap, stack, and time limits, admitted through a pool of 4 workers\n * per JavaScript realm; pass an object to change them. `'none'` parses in the calling thread and is\n * unsafe for untrusted input.\n */\n readonly isolation?: FileExtractorIsolation\n /**\n * Load SheetJS in the calling thread. Only with `isolation: 'none'` (a worker cannot receive a\n * function and always imports the installed `xlsx` itself); any other isolation is a defect\n * when the layer is built. Defaults to a lazy `import('xlsx')`. Missing, non-SheetJS, or\n * pre-0.20.3 modules fail with `SheetJsUnavailableError`.\n */\n readonly loadSheetJs?: SheetJsLoader\n}\n\nconst decodeText = (bytes: Uint8Array) => new TextDecoder('utf-8', { fatal: false }).decode(bytes)\n\ntype ParsedFileExtractor = WorkerExtractor\n\n/** Build the Node `FileExtractor` from validated limits and the parser runner. */\nconst makeFileExtractor = (\n limits: FileExtractorLimits,\n extractParsed: ParsedFileExtractor\n): FileExtractorApi => ({\n extract: input =>\n Effect.gen(function* () {\n const format = fileFormatFor(input)\n\n if (format === undefined)\n return yield* Effect.fail(\n new UnsupportedFileFormatError({ filename: input.filename, mediaType: input.mediaType })\n )\n\n yield* Effect.annotateCurrentSpan({\n 'file_extractor.format': format,\n 'file_extractor.file_size': input.bytes.byteLength\n })\n\n if (input.bytes.byteLength > limits.maxInputBytes)\n return yield* Effect.fail(\n new FileExtractionError({ message: 'File exceeds the extraction size limit', format })\n )\n\n if (isParsedFileFormat(format)) return yield* extractParsed(input, format, limits)\n\n // Text formats are only decoded and sanitized; no parser reads them.\n return yield* makeExtractedFile(decodeText(input.bytes), { format, title: input.filename })\n }).pipe(Effect.withSpan('FileExtractor.extract'))\n})\n\nconst parsedFileExtractor = (\n options: FileExtractorOptions\n): Effect.Effect<ParsedFileExtractor, Schema.SchemaError> => {\n const isolation = options.isolation ?? 'worker'\n\n if (isolation === 'none') {\n const loader = options.loadSheetJs ?? defaultSheetJsLoader\n\n return Effect.succeed((input, format: ParsedFileFormat, limits) =>\n extractParsedFile(input, format, limits, loader)\n )\n }\n\n if (options.loadSheetJs !== undefined)\n return Effect.die(\n new Error(\n 'FileExtractorOptions.loadSheetJs runs SheetJS in the calling thread and needs isolation: \"none\"'\n )\n )\n\n const overrides = isolation === 'worker' ? {} : isolation\n const timeoutMs = overrides.timeoutMs ?? defaultWorkerIsolation.timeoutMs\n\n return Schema.decodeUnknownEffect(WorkerIsolationSettings)({\n maxOldGenerationSizeMb:\n overrides.maxOldGenerationSizeMb ?? defaultWorkerIsolation.maxOldGenerationSizeMb,\n maxYoungGenerationSizeMb:\n overrides.maxYoungGenerationSizeMb ?? defaultWorkerIsolation.maxYoungGenerationSizeMb,\n stackSizeMb: overrides.stackSizeMb ?? defaultWorkerIsolation.stackSizeMb,\n timeoutMs,\n maxConcurrentWorkers:\n overrides.maxConcurrentWorkers ?? defaultWorkerIsolation.maxConcurrentWorkers,\n maxQueueWaitMs: overrides.maxQueueWaitMs ?? timeoutMs\n }).pipe(\n Effect.flatMap(settings =>\n makeWorkerExtractor(settings, overrides.workerUrl ?? defaultExtractionWorkerUrl())\n )\n )\n}\n\n/**\n * Node `FileExtractor` layer with custom limits, isolation, or (in-process) SheetJS loader.\n * Invalid limits or isolation settings are defects.\n */\nexport const makeFileExtractorLayer = (options: FileExtractorOptions = {}) =>\n Layer.effect(\n FileExtractor,\n Effect.gen(function* () {\n const limits = yield* Schema.decodeUnknownEffect(FileExtractorLimits)({\n ...defaultFileExtractorLimits,\n ...options.limits\n })\n\n const extractParsed = yield* parsedFileExtractor(options)\n\n return FileExtractor.of(makeFileExtractor(limits, extractParsed))\n }).pipe(Effect.orDie)\n )\n\n/**\n * Node `FileExtractor` layer with the default limits, worker isolation, and lazy SheetJS loading.\n */\nexport const FileExtractorLayer = makeFileExtractorLayer()\n"],"mappings":";;;;;;;;;;AAsCA,MAAM,cAAc,UAAsB,IAAI,YAAY,SAAS,EAAE,OAAO,MAAM,CAAC,EAAE,OAAO,KAAK;;AAKjG,MAAM,qBACJ,QACA,mBACsB,EACtB,UAAS,UACP,OAAO,IAAI,aAAa;CACtB,MAAM,SAAS,cAAc,KAAK;CAElC,IAAI,WAAW,KAAA,GACb,OAAO,OAAO,OAAO,KACnB,IAAI,2BAA2B;EAAE,UAAU,MAAM;EAAU,WAAW,MAAM;CAAU,CAAC,CACzF;CAEF,OAAO,OAAO,oBAAoB;EAChC,yBAAyB;EACzB,4BAA4B,MAAM,MAAM;CAC1C,CAAC;CAED,IAAI,MAAM,MAAM,aAAa,OAAO,eAClC,OAAO,OAAO,OAAO,KACnB,IAAI,oBAAoB;EAAE,SAAS;EAA0C;CAAO,CAAC,CACvF;CAEF,IAAI,mBAAmB,MAAM,GAAG,OAAO,OAAO,cAAc,OAAO,QAAQ,MAAM;CAGjF,OAAO,OAAO,kBAAkB,WAAW,MAAM,KAAK,GAAG;EAAE;EAAQ,OAAO,MAAM;CAAS,CAAC;AAC5F,CAAC,EAAE,KAAK,OAAO,SAAS,uBAAuB,CAAC,EACpD;AAEA,MAAM,uBACJ,YAC2D;CAC3D,MAAM,YAAY,QAAQ,aAAa;CAEvC,IAAI,cAAc,QAAQ;EACxB,MAAM,SAAS,QAAQ,eAAe;EAEtC,OAAO,OAAO,SAAS,OAAO,QAA0B,WACtD,kBAAkB,OAAO,QAAQ,QAAQ,MAAM,CACjD;CACF;CAEA,IAAI,QAAQ,gBAAgB,KAAA,GAC1B,OAAO,OAAO,oBACZ,IAAI,MACF,mGACF,CACF;CAEF,MAAM,YAAY,cAAc,WAAW,CAAC,IAAI;CAChD,MAAM,YAAY,UAAU,aAAa,uBAAuB;CAEhE,OAAO,OAAO,oBAAoB,uBAAuB,EAAE;EACzD,wBACE,UAAU,0BAA0B,uBAAuB;EAC7D,0BACE,UAAU,4BAA4B,uBAAuB;EAC/D,aAAa,UAAU,eAAe,uBAAuB;EAC7D;EACA,sBACE,UAAU,wBAAwB,uBAAuB;EAC3D,gBAAgB,UAAU,kBAAkB;CAC9C,CAAC,EAAE,KACD,OAAO,SAAQ,aACb,oBAAoB,UAAU,UAAU,aAAa,2BAA2B,CAAC,CACnF,CACF;AACF;;;;;AAMA,MAAa,0BAA0B,UAAgC,CAAC,MACtE,MAAM,OACJ,eACA,OAAO,IAAI,aAAa;CACtB,MAAM,SAAS,OAAO,OAAO,oBAAoB,mBAAmB,EAAE;EACpE,GAAG;EACH,GAAG,QAAQ;CACb,CAAC;CAED,MAAM,gBAAgB,OAAO,oBAAoB,OAAO;CAExD,OAAO,cAAc,GAAG,kBAAkB,QAAQ,aAAa,CAAC;AAClE,CAAC,EAAE,KAAK,OAAO,KAAK,CACtB;;;;AAKF,MAAa,qBAAqB,uBAAuB"}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { OfficeArchiveError } from "../errors.mjs";
|
|
2
|
+
import { OfficeFileFormat } from "../format.mjs";
|
|
3
|
+
import { FileExtractorLimits } from "../limits.mjs";
|
|
4
|
+
import { Effect } from "effect";
|
|
5
|
+
|
|
6
|
+
//#region src/node/office-archive.d.ts
|
|
7
|
+
type OfficeArchiveLimits = Pick<FileExtractorLimits, 'maxArchiveEntries' | 'maxExpandedBytes' | 'maxInputBytes'>;
|
|
8
|
+
/** Inflated output is counted in slices of at most this many bytes. */
|
|
9
|
+
declare const officeInflateChunkBytes: number;
|
|
10
|
+
/**
|
|
11
|
+
* Whether a part SheetJS would decode as BOM-marked UTF-16 contains a hyperlink tag. Covers
|
|
12
|
+
* SheetJS `cc2str` UTF-16 BOM decoding (little- and big-endian from byte 2, including its
|
|
13
|
+
* `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
|
|
14
|
+
*/
|
|
15
|
+
declare const utf16PartHasHyperlink: (content: Uint8Array) => boolean;
|
|
16
|
+
/** Hyperlink start tags (Latin-1 text of the original bytes) found while stripping, per part. */
|
|
17
|
+
type StrippedHyperlinkTags = ReadonlyMap<string, ReadonlyArray<string>>;
|
|
18
|
+
type NormalizedOfficeArchive = {
|
|
19
|
+
/** Every validated (and, for XLSX, hyperlink-stripped) part by entry name. */readonly parts: Readonly<Record<string, Uint8Array>>; /** XLSX only: removed hyperlink tags, at most `maxHyperlinkTags` across the workbook. */
|
|
20
|
+
readonly hyperlinkTags: StrippedHyperlinkTags;
|
|
21
|
+
};
|
|
22
|
+
type ReadOfficeArchiveOptions = {
|
|
23
|
+
/** XLSX: hyperlink start tags to capture across the workbook (default 0). */readonly maxHyperlinkTags?: number;
|
|
24
|
+
};
|
|
25
|
+
/** A fresh stored-entry ZIP of validated parts; the attacker's ZIP index is never reused. */
|
|
26
|
+
declare const storedArchive: (parts: Readonly<Record<string, Uint8Array>>) => Uint8Array<ArrayBuffer>;
|
|
27
|
+
/**
|
|
28
|
+
* Inflate bounded input chunks, count actual output, and return the validated parts. The
|
|
29
|
+
* attacker's ZIP index is discarded: parsers only get archives rebuilt from these parts.
|
|
30
|
+
*
|
|
31
|
+
* For XLSX, every part loses its `<hyperlink>` tags: SheetJS expands each hyperlink range into
|
|
32
|
+
* per-cell objects before any budget runs, so one `ref="A1:XFD1048576"` exhausts memory. The
|
|
33
|
+
* removed tags are returned so links can still be shown. XLSX input that SheetJS would route to
|
|
34
|
+
* its binary (XLSB), ODS, or Numbers parsers is rejected early with a clear error; the guarantee
|
|
35
|
+
* is that SheetJS only receives `buildSheetJsInput`'s allowlisted archive.
|
|
36
|
+
*/
|
|
37
|
+
declare const readOfficeArchive: (input: Uint8Array, format: OfficeFileFormat, limits: OfficeArchiveLimits, options?: ReadOfficeArchiveOptions) => Effect.Effect<NormalizedOfficeArchive, OfficeArchiveError, never>;
|
|
38
|
+
/**
|
|
39
|
+
* Validate a DOCX, XLSX, or PPTX archive with bounded inflation and return a rebuilt stored-entry
|
|
40
|
+
* ZIP of every validated part, for storage or for other parsers. XLSX parts lose their hyperlink
|
|
41
|
+
* tags, and XLSX input with ODS or Numbers marker entries or XLSB parts is rejected; `.bin` parts
|
|
42
|
+
* SheetJS never parses (printer settings, OLE objects) are kept so stored files still open.
|
|
43
|
+
*
|
|
44
|
+
* The output is not SheetJS input. Never run SheetJS on it directly: extract XLSX text through
|
|
45
|
+
* `FileExtractor`, which hands SheetJS only an allowlisted archive it builds itself (worksheet,
|
|
46
|
+
* shared-string, style, and core-property parts with generated content types and relationships).
|
|
47
|
+
*/
|
|
48
|
+
declare const normalizeOfficeArchive: (bytes: Uint8Array, format: OfficeFileFormat, limits?: Partial<OfficeArchiveLimits>) => Effect.Effect<Uint8Array<ArrayBuffer>, OfficeArchiveError, never>;
|
|
49
|
+
//#endregion
|
|
50
|
+
export { NormalizedOfficeArchive, OfficeArchiveLimits, ReadOfficeArchiveOptions, StrippedHyperlinkTags, normalizeOfficeArchive, officeInflateChunkBytes, readOfficeArchive, storedArchive, utf16PartHasHyperlink };
|
|
51
|
+
//# sourceMappingURL=office-archive.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"office-archive.d.mts","names":[],"sources":["../../src/node/office-archive.ts"],"mappings":";;;;;;KAgBY,mBAAA,GAAsB,IAAI,CACpC,mBAAA;;cAgBW,uBAAA;;;;AAhBQ;AAgBrB;cA2Ia,qBAAA,GAAyB,OAAmB,EAAV,UAAU;;KAW7C,qBAAA,GAAwB,WAAW,SAAS,aAAA;AAAA,KAE5C,uBAAA;EAbC,uFAeF,KAAA,EAAO,QAAA,CAAS,MAAA,SAAe,UAAA;WAE/B,aAAA,EAAe,qBAAA;AAAA;AAAA,KAwCd,wBAAA;EA9CqB,sFAgDtB,gBAAgB;AAAA;AAhD0C;AAAA,cAoDxD,aAAA,GAAiB,KAAA,EAAO,QAAA,CAAS,MAAA,SAAe,UAAA,OAAY,UAAA,CAAA,WAAA;;;;;;;;;;;cAa5D,iBAAA,GACX,KAAA,EAAO,UAAA,EACP,MAAA,EAAQ,gBAAA,EACR,MAAA,EAAQ,mBAAA,EACR,OAAA,GAAS,wBAAA,KAA6B,MAAA,CAAA,MAAA,CAAA,uBAAA,EAAA,kBAAA;;;;;AA/DO;AAwC/C;;;;AAE2B;cA+Hd,sBAAA,GACX,KAAA,EAAO,UAAA,EACP,MAAA,EAAQ,gBAAA,EACR,MAAA,GAAQ,OAAA,CAAQ,mBAAA,MAAyB,MAAA,CAAA,MAAA,CAAA,UAAA,CAAA,WAAA,GAAA,kBAAA"}
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
import { OfficeArchiveError } from "../errors.mjs";
|
|
2
|
+
import { defaultFileExtractorLimits } from "../limits.mjs";
|
|
3
|
+
import { sheetJsTextViews } from "./sheetjs-xml.mjs";
|
|
4
|
+
import { contentTypesRouteToBinary, isAlternateFormatEntry, relationshipsRouteToBinary } from "./xlsx-routing.mjs";
|
|
5
|
+
import { Effect } from "effect";
|
|
6
|
+
import { Buffer } from "node:buffer";
|
|
7
|
+
import { Readable } from "node:stream";
|
|
8
|
+
import { createInflateRaw } from "node:zlib";
|
|
9
|
+
import { zipSync } from "fflate";
|
|
10
|
+
//#region src/node/office-archive.ts
|
|
11
|
+
const invalid = () => new OfficeArchiveError({ message: "Invalid Office archive." });
|
|
12
|
+
const tooLargeMessage = "Office archive expansion exceeds its declared size or limit.";
|
|
13
|
+
const tooLarge = (expandedBytes) => expandedBytes === void 0 ? new OfficeArchiveError({ message: tooLargeMessage }) : new OfficeArchiveError({
|
|
14
|
+
message: tooLargeMessage,
|
|
15
|
+
expandedBytes
|
|
16
|
+
});
|
|
17
|
+
const compressedChunkBytes = 1024;
|
|
18
|
+
/** Inflated output is counted in slices of at most this many bytes. */
|
|
19
|
+
const officeInflateChunkBytes = 16 * 1024;
|
|
20
|
+
const mainParts = {
|
|
21
|
+
docx: "word/document.xml",
|
|
22
|
+
pptx: "ppt/presentation.xml",
|
|
23
|
+
xlsx: "xl/workbook.xml"
|
|
24
|
+
};
|
|
25
|
+
/** Read the directory only as a bounded index. Its sizes are never trusted for allocation. */
|
|
26
|
+
const archiveEntries = (bytes, limits) => {
|
|
27
|
+
let end = bytes.length - 22;
|
|
28
|
+
const earliest = Math.max(0, end - 65535);
|
|
29
|
+
while (end >= earliest && bytes.readUInt32LE(end) !== 101010256) end -= 1;
|
|
30
|
+
if (end < earliest || end + 22 + bytes.readUInt16LE(end + 20) !== bytes.length) throw invalid();
|
|
31
|
+
const count = bytes.readUInt16LE(end + 10);
|
|
32
|
+
const directorySize = bytes.readUInt32LE(end + 12);
|
|
33
|
+
const directoryOffset = bytes.readUInt32LE(end + 16);
|
|
34
|
+
if (bytes.readUInt32LE(end + 4) !== 0 || bytes.readUInt16LE(end + 8) !== count || count === 0 || count > limits.maxArchiveEntries || directoryOffset + directorySize !== end) throw invalid();
|
|
35
|
+
const entries = [];
|
|
36
|
+
const names = /* @__PURE__ */ new Set();
|
|
37
|
+
let offset = directoryOffset;
|
|
38
|
+
let declaredTotal = 0;
|
|
39
|
+
for (let index = 0; index < count; index += 1) {
|
|
40
|
+
if (offset + 46 > end || bytes.readUInt32LE(offset) !== 33639248) throw invalid();
|
|
41
|
+
const flags = bytes.readUInt16LE(offset + 8);
|
|
42
|
+
const method = bytes.readUInt16LE(offset + 10);
|
|
43
|
+
const compressedSize = bytes.readUInt32LE(offset + 20);
|
|
44
|
+
const originalSize = bytes.readUInt32LE(offset + 24);
|
|
45
|
+
const nameSize = bytes.readUInt16LE(offset + 28);
|
|
46
|
+
const extraSize = bytes.readUInt16LE(offset + 30);
|
|
47
|
+
const commentSize = bytes.readUInt16LE(offset + 32);
|
|
48
|
+
const local = bytes.readUInt32LE(offset + 42);
|
|
49
|
+
const next = offset + 46 + nameSize + extraSize + commentSize;
|
|
50
|
+
if (next > end || nameSize === 0 || nameSize > 1024 || (flags & -2063) !== 0 || method !== 0 && method !== 8 || bytes.readUInt16LE(offset + 34) !== 0 || compressedSize === 4294967295 || originalSize === 4294967295 || local + 30 > directoryOffset) throw invalid();
|
|
51
|
+
const nameBytes = bytes.subarray(offset + 46, offset + 46 + nameSize);
|
|
52
|
+
const name = new TextDecoder("utf-8", { fatal: true }).decode(nameBytes);
|
|
53
|
+
if (/[\\\u0000-\u001f]/.test(name) || name.startsWith("/") || name.includes("//") || name.split("/").some((part) => part === ".." || part === ".") || names.has(name.toLowerCase()) || /vbaProject\.bin$/i.test(name)) throw invalid();
|
|
54
|
+
names.add(name.toLowerCase());
|
|
55
|
+
if (bytes.readUInt32LE(local) !== 67324752 || bytes.readUInt16LE(local + 6) !== flags || bytes.readUInt16LE(local + 8) !== method || bytes.readUInt16LE(local + 26) !== nameSize) throw invalid();
|
|
56
|
+
const start = local + 30 + nameSize + bytes.readUInt16LE(local + 28);
|
|
57
|
+
if (start + compressedSize > directoryOffset || !bytes.subarray(local + 30, local + 30 + nameSize).equals(nameBytes)) throw invalid();
|
|
58
|
+
for (const [position, expected] of [[18, compressedSize], [22, originalSize]]) {
|
|
59
|
+
const value = bytes.readUInt32LE(local + position);
|
|
60
|
+
if (value !== expected && !((flags & 8) !== 0 && value === 0)) throw invalid();
|
|
61
|
+
}
|
|
62
|
+
declaredTotal += originalSize;
|
|
63
|
+
if (declaredTotal > limits.maxExpandedBytes) throw tooLarge();
|
|
64
|
+
entries.push({
|
|
65
|
+
name,
|
|
66
|
+
start,
|
|
67
|
+
compressedSize,
|
|
68
|
+
originalSize,
|
|
69
|
+
method
|
|
70
|
+
});
|
|
71
|
+
offset = next;
|
|
72
|
+
}
|
|
73
|
+
if (offset !== end) throw invalid();
|
|
74
|
+
return entries;
|
|
75
|
+
};
|
|
76
|
+
/**
|
|
77
|
+
* A superset of SheetJS's `hlinkregex` (`/<(?:\w+:)?hyperlink [^<>]*>/`). A candidate never spans
|
|
78
|
+
* a `<`, so each scan stops at the next tag and the strip stays linear in the part size.
|
|
79
|
+
*/
|
|
80
|
+
const hyperlinkTag = /<\/?(?:[\w.-]+:)?hyperlink\b[^<>]*>/gi;
|
|
81
|
+
const hyperlinkTagStart = /<\/?(?:[\w.-]+:)?hyperlink\b/i;
|
|
82
|
+
/**
|
|
83
|
+
* Whether a part SheetJS would decode as BOM-marked UTF-16 contains a hyperlink tag. Covers
|
|
84
|
+
* SheetJS `cc2str` UTF-16 BOM decoding (little- and big-endian from byte 2, including its
|
|
85
|
+
* `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
|
|
86
|
+
*/
|
|
87
|
+
const utf16PartHasHyperlink = (content) => sheetJsTextViews(content).slice(1).some((text) => hyperlinkTagStart.test(text));
|
|
88
|
+
const unsupportedXlsxParts = () => new OfficeArchiveError({ message: "XLSX archive contains binary (XLSB), ODS, or Numbers parts." });
|
|
89
|
+
const inflateEntry = async (compressed, record) => {
|
|
90
|
+
function* inputChunks() {
|
|
91
|
+
for (let offset = 0; offset < compressed.length; offset += compressedChunkBytes) yield compressed.subarray(offset, offset + compressedChunkBytes);
|
|
92
|
+
}
|
|
93
|
+
const source = Readable.from(inputChunks(), { highWaterMark: 1 });
|
|
94
|
+
const inflater = createInflateRaw({
|
|
95
|
+
chunkSize: officeInflateChunkBytes,
|
|
96
|
+
readableHighWaterMark: officeInflateChunkBytes,
|
|
97
|
+
writableHighWaterMark: compressedChunkBytes
|
|
98
|
+
});
|
|
99
|
+
source.pipe(inflater);
|
|
100
|
+
try {
|
|
101
|
+
for await (const value of inflater) {
|
|
102
|
+
if (!Buffer.isBuffer(value)) throw invalid();
|
|
103
|
+
for (let offset = 0; offset < value.length; offset += officeInflateChunkBytes) record(value.subarray(offset, offset + officeInflateChunkBytes));
|
|
104
|
+
}
|
|
105
|
+
if (inflater.bytesWritten !== compressed.length) throw invalid();
|
|
106
|
+
} finally {
|
|
107
|
+
source.destroy();
|
|
108
|
+
inflater.destroy();
|
|
109
|
+
}
|
|
110
|
+
};
|
|
111
|
+
/** A fresh stored-entry ZIP of validated parts; the attacker's ZIP index is never reused. */
|
|
112
|
+
const storedArchive = (parts) => zipSync({ ...parts }, { level: 0 });
|
|
113
|
+
/**
|
|
114
|
+
* Inflate bounded input chunks, count actual output, and return the validated parts. The
|
|
115
|
+
* attacker's ZIP index is discarded: parsers only get archives rebuilt from these parts.
|
|
116
|
+
*
|
|
117
|
+
* For XLSX, every part loses its `<hyperlink>` tags: SheetJS expands each hyperlink range into
|
|
118
|
+
* per-cell objects before any budget runs, so one `ref="A1:XFD1048576"` exhausts memory. The
|
|
119
|
+
* removed tags are returned so links can still be shown. XLSX input that SheetJS would route to
|
|
120
|
+
* its binary (XLSB), ODS, or Numbers parsers is rejected early with a clear error; the guarantee
|
|
121
|
+
* is that SheetJS only receives `buildSheetJsInput`'s allowlisted archive.
|
|
122
|
+
*/
|
|
123
|
+
const readOfficeArchive = (input, format, limits, options = {}) => Effect.tryPromise({
|
|
124
|
+
try: async () => {
|
|
125
|
+
const bytes = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
|
|
126
|
+
if (bytes.length < 22 || bytes.length > limits.maxInputBytes) throw invalid();
|
|
127
|
+
const entries = archiveEntries(bytes, limits);
|
|
128
|
+
const mainPart = mainParts[format];
|
|
129
|
+
const maxHyperlinkTags = options.maxHyperlinkTags ?? 0;
|
|
130
|
+
if (format === "xlsx" && entries.some((entry) => isAlternateFormatEntry(entry.name))) throw unsupportedXlsxParts();
|
|
131
|
+
if (!entries.some((entry) => entry.name === "[Content_Types].xml") || !entries.some((entry) => entry.name === mainPart)) throw invalid();
|
|
132
|
+
const validated = Object.create(null);
|
|
133
|
+
const hyperlinkTags = /* @__PURE__ */ new Map();
|
|
134
|
+
let capturedTags = 0;
|
|
135
|
+
let expandedBytes = 0;
|
|
136
|
+
for (const entry of entries) {
|
|
137
|
+
const chunks = [];
|
|
138
|
+
let entryBytes = 0;
|
|
139
|
+
const record = (chunk) => {
|
|
140
|
+
entryBytes += chunk.length;
|
|
141
|
+
expandedBytes += chunk.length;
|
|
142
|
+
if (expandedBytes > limits.maxExpandedBytes || entryBytes > entry.originalSize) throw tooLarge(expandedBytes);
|
|
143
|
+
chunks.push(chunk);
|
|
144
|
+
};
|
|
145
|
+
const compressed = bytes.subarray(entry.start, entry.start + entry.compressedSize);
|
|
146
|
+
if (entry.method === 0) record(compressed);
|
|
147
|
+
else await inflateEntry(compressed, record);
|
|
148
|
+
if (entryBytes !== entry.originalSize) throw invalid();
|
|
149
|
+
const content = Buffer.concat(chunks, entryBytes);
|
|
150
|
+
if (format !== "xlsx") {
|
|
151
|
+
validated[entry.name] = content;
|
|
152
|
+
continue;
|
|
153
|
+
}
|
|
154
|
+
const tags = [];
|
|
155
|
+
const rewritten = Buffer.from(content.toString("latin1").replace(hyperlinkTag, (tag) => {
|
|
156
|
+
if (!tag.startsWith("</") && capturedTags < maxHyperlinkTags) {
|
|
157
|
+
capturedTags += 1;
|
|
158
|
+
tags.push(tag);
|
|
159
|
+
}
|
|
160
|
+
return " ";
|
|
161
|
+
}), "latin1");
|
|
162
|
+
if (utf16PartHasHyperlink(rewritten)) throw invalid();
|
|
163
|
+
const lowerName = entry.name.toLowerCase();
|
|
164
|
+
if (lowerName === "[content_types].xml" && contentTypesRouteToBinary(rewritten) || lowerName.endsWith(".rels") && relationshipsRouteToBinary(rewritten)) throw unsupportedXlsxParts();
|
|
165
|
+
if (tags.length > 0) hyperlinkTags.set(entry.name, tags);
|
|
166
|
+
validated[entry.name] = rewritten;
|
|
167
|
+
}
|
|
168
|
+
return {
|
|
169
|
+
parts: validated,
|
|
170
|
+
hyperlinkTags
|
|
171
|
+
};
|
|
172
|
+
},
|
|
173
|
+
catch: (error) => error instanceof OfficeArchiveError ? error : invalid()
|
|
174
|
+
});
|
|
175
|
+
/**
|
|
176
|
+
* Validate a DOCX, XLSX, or PPTX archive with bounded inflation and return a rebuilt stored-entry
|
|
177
|
+
* ZIP of every validated part, for storage or for other parsers. XLSX parts lose their hyperlink
|
|
178
|
+
* tags, and XLSX input with ODS or Numbers marker entries or XLSB parts is rejected; `.bin` parts
|
|
179
|
+
* SheetJS never parses (printer settings, OLE objects) are kept so stored files still open.
|
|
180
|
+
*
|
|
181
|
+
* The output is not SheetJS input. Never run SheetJS on it directly: extract XLSX text through
|
|
182
|
+
* `FileExtractor`, which hands SheetJS only an allowlisted archive it builds itself (worksheet,
|
|
183
|
+
* shared-string, style, and core-property parts with generated content types and relationships).
|
|
184
|
+
*/
|
|
185
|
+
const normalizeOfficeArchive = (bytes, format, limits = {}) => readOfficeArchive(bytes, format, {
|
|
186
|
+
maxArchiveEntries: limits.maxArchiveEntries ?? defaultFileExtractorLimits.maxArchiveEntries,
|
|
187
|
+
maxExpandedBytes: limits.maxExpandedBytes ?? defaultFileExtractorLimits.maxExpandedBytes,
|
|
188
|
+
maxInputBytes: limits.maxInputBytes ?? defaultFileExtractorLimits.maxInputBytes
|
|
189
|
+
}).pipe(Effect.map((normalized) => storedArchive(normalized.parts)));
|
|
190
|
+
//#endregion
|
|
191
|
+
export { normalizeOfficeArchive, officeInflateChunkBytes, readOfficeArchive, storedArchive, utf16PartHasHyperlink };
|
|
192
|
+
|
|
193
|
+
//# sourceMappingURL=office-archive.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"office-archive.mjs","names":[],"sources":["../../src/node/office-archive.ts"],"sourcesContent":["import { Buffer } from 'node:buffer'\nimport { Readable } from 'node:stream'\nimport { createInflateRaw } from 'node:zlib'\nimport { Effect } from 'effect'\nimport { zipSync } from 'fflate'\nimport { OfficeArchiveError } from '../errors.ts'\nimport type { OfficeFileFormat } from '../format.ts'\nimport { defaultFileExtractorLimits } from '../limits.ts'\nimport type { FileExtractorLimits } from '../limits.ts'\nimport { sheetJsTextViews } from './sheetjs-xml.ts'\nimport {\n contentTypesRouteToBinary,\n isAlternateFormatEntry,\n relationshipsRouteToBinary\n} from './xlsx-routing.ts'\n\nexport type OfficeArchiveLimits = Pick<\n FileExtractorLimits,\n 'maxArchiveEntries' | 'maxExpandedBytes' | 'maxInputBytes'\n>\n\nconst invalid = () => new OfficeArchiveError({ message: 'Invalid Office archive.' })\n\nconst tooLargeMessage = 'Office archive expansion exceeds its declared size or limit.'\n\nconst tooLarge = (expandedBytes?: number) =>\n expandedBytes === undefined\n ? new OfficeArchiveError({ message: tooLargeMessage })\n : new OfficeArchiveError({ message: tooLargeMessage, expandedBytes })\n\nconst compressedChunkBytes = 1024\n\n/** Inflated output is counted in slices of at most this many bytes. */\nexport const officeInflateChunkBytes = 16 * 1024\n\nconst mainParts: Readonly<Record<OfficeFileFormat, string>> = {\n docx: 'word/document.xml',\n pptx: 'ppt/presentation.xml',\n xlsx: 'xl/workbook.xml'\n}\n\ntype ArchiveEntry = {\n readonly name: string\n readonly start: number\n readonly compressedSize: number\n readonly originalSize: number\n readonly method: number\n}\n\n/** Read the directory only as a bounded index. Its sizes are never trusted for allocation. */\nconst archiveEntries = (bytes: Buffer, limits: OfficeArchiveLimits) => {\n let end = bytes.length - 22\n const earliest = Math.max(0, end - 65535)\n\n while (end >= earliest && bytes.readUInt32LE(end) !== 0x06054b50) end -= 1\n\n if (end < earliest || end + 22 + bytes.readUInt16LE(end + 20) !== bytes.length) throw invalid()\n\n const count = bytes.readUInt16LE(end + 10)\n const directorySize = bytes.readUInt32LE(end + 12)\n const directoryOffset = bytes.readUInt32LE(end + 16)\n\n if (\n bytes.readUInt32LE(end + 4) !== 0 ||\n bytes.readUInt16LE(end + 8) !== count ||\n count === 0 ||\n count > limits.maxArchiveEntries ||\n directoryOffset + directorySize !== end\n )\n throw invalid()\n\n const entries: Array<ArchiveEntry> = []\n // OPC part names are case-insensitive, and SheetJS looks entries up ignoring case.\n const names = new Set<string>()\n let offset = directoryOffset\n let declaredTotal = 0\n\n for (let index = 0; index < count; index += 1) {\n if (offset + 46 > end || bytes.readUInt32LE(offset) !== 0x02014b50) throw invalid()\n\n const flags = bytes.readUInt16LE(offset + 8)\n const method = bytes.readUInt16LE(offset + 10)\n const compressedSize = bytes.readUInt32LE(offset + 20)\n const originalSize = bytes.readUInt32LE(offset + 24)\n const nameSize = bytes.readUInt16LE(offset + 28)\n const extraSize = bytes.readUInt16LE(offset + 30)\n const commentSize = bytes.readUInt16LE(offset + 32)\n const local = bytes.readUInt32LE(offset + 42)\n const next = offset + 46 + nameSize + extraSize + commentSize\n\n // Reject encryption, unsupported methods, split/ZIP64 archives, and ambiguous paths.\n if (\n next > end ||\n nameSize === 0 ||\n nameSize > 1024 ||\n (flags & ~0x080e) !== 0 ||\n (method !== 0 && method !== 8) ||\n bytes.readUInt16LE(offset + 34) !== 0 ||\n compressedSize === 0xffffffff ||\n originalSize === 0xffffffff ||\n local + 30 > directoryOffset\n )\n throw invalid()\n\n const nameBytes = bytes.subarray(offset + 46, offset + 46 + nameSize)\n const name = new TextDecoder('utf-8', { fatal: true }).decode(nameBytes)\n\n // `//` is rejected (SheetJS collapses the first one), but directory entries ending in `/` stay.\n if (\n /[\\\\\\u0000-\\u001f]/.test(name) ||\n name.startsWith('/') ||\n name.includes('//') ||\n name.split('/').some(part => part === '..' || part === '.') ||\n names.has(name.toLowerCase()) ||\n /vbaProject\\.bin$/i.test(name)\n )\n throw invalid()\n\n names.add(name.toLowerCase())\n\n if (\n bytes.readUInt32LE(local) !== 0x04034b50 ||\n bytes.readUInt16LE(local + 6) !== flags ||\n bytes.readUInt16LE(local + 8) !== method ||\n bytes.readUInt16LE(local + 26) !== nameSize\n )\n throw invalid()\n\n const start = local + 30 + nameSize + bytes.readUInt16LE(local + 28)\n\n if (\n start + compressedSize > directoryOffset ||\n !bytes.subarray(local + 30, local + 30 + nameSize).equals(nameBytes)\n )\n throw invalid()\n\n // Data descriptors may leave local sizes zero; all nonzero declarations must agree.\n for (const [position, expected] of [\n [18, compressedSize],\n [22, originalSize]\n ] as const) {\n const value = bytes.readUInt32LE(local + position)\n\n if (value !== expected && !((flags & 8) !== 0 && value === 0)) throw invalid()\n }\n\n declaredTotal += originalSize\n\n if (declaredTotal > limits.maxExpandedBytes) throw tooLarge()\n\n entries.push({ name, start, compressedSize, originalSize, method })\n offset = next\n }\n\n if (offset !== end) throw invalid()\n\n return entries\n}\n\n/**\n * A superset of SheetJS's `hlinkregex` (`/<(?:\\w+:)?hyperlink [^<>]*>/`). A candidate never spans\n * a `<`, so each scan stops at the next tag and the strip stays linear in the part size.\n */\nconst hyperlinkTag = /<\\/?(?:[\\w.-]+:)?hyperlink\\b[^<>]*>/gi\n\nconst hyperlinkTagStart = /<\\/?(?:[\\w.-]+:)?hyperlink\\b/i\n\n/**\n * Whether a part SheetJS would decode as BOM-marked UTF-16 contains a hyperlink tag. Covers\n * SheetJS `cc2str` UTF-16 BOM decoding (little- and big-endian from byte 2, including its\n * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.\n */\nexport const utf16PartHasHyperlink = (content: Uint8Array) =>\n sheetJsTextViews(content)\n .slice(1)\n .some(text => hyperlinkTagStart.test(text))\n\nconst unsupportedXlsxParts = () =>\n new OfficeArchiveError({\n message: 'XLSX archive contains binary (XLSB), ODS, or Numbers parts.'\n })\n\n/** Hyperlink start tags (Latin-1 text of the original bytes) found while stripping, per part. */\nexport type StrippedHyperlinkTags = ReadonlyMap<string, ReadonlyArray<string>>\n\nexport type NormalizedOfficeArchive = {\n /** Every validated (and, for XLSX, hyperlink-stripped) part by entry name. */\n readonly parts: Readonly<Record<string, Uint8Array>>\n /** XLSX only: removed hyperlink tags, at most `maxHyperlinkTags` across the workbook. */\n readonly hyperlinkTags: StrippedHyperlinkTags\n}\n\nconst inflateEntry = async (compressed: Buffer, record: (chunk: Buffer) => void): Promise<void> => {\n function* inputChunks() {\n for (let offset = 0; offset < compressed.length; offset += compressedChunkBytes) {\n yield compressed.subarray(offset, offset + compressedChunkBytes)\n }\n }\n\n const source = Readable.from(inputChunks(), { highWaterMark: 1 })\n\n // Stream high-water marks are valid Transform options that `ZlibOptions` does not declare.\n const inflateOptions = {\n chunkSize: officeInflateChunkBytes,\n readableHighWaterMark: officeInflateChunkBytes,\n writableHighWaterMark: compressedChunkBytes\n }\n\n const inflater = createInflateRaw(inflateOptions)\n\n source.pipe(inflater)\n\n try {\n // Async iteration may combine buffered chunks; bound accounting slices explicitly.\n for await (const value of inflater) {\n if (!Buffer.isBuffer(value)) throw invalid()\n\n for (let offset = 0; offset < value.length; offset += officeInflateChunkBytes) {\n record(value.subarray(offset, offset + officeInflateChunkBytes))\n }\n }\n\n if (inflater.bytesWritten !== compressed.length) throw invalid()\n } finally {\n source.destroy()\n inflater.destroy()\n }\n}\n\nexport type ReadOfficeArchiveOptions = {\n /** XLSX: hyperlink start tags to capture across the workbook (default 0). */\n readonly maxHyperlinkTags?: number\n}\n\n/** A fresh stored-entry ZIP of validated parts; the attacker's ZIP index is never reused. */\nexport const storedArchive = (parts: Readonly<Record<string, Uint8Array>>) =>\n zipSync({ ...parts }, { level: 0 })\n\n/**\n * Inflate bounded input chunks, count actual output, and return the validated parts. The\n * attacker's ZIP index is discarded: parsers only get archives rebuilt from these parts.\n *\n * For XLSX, every part loses its `<hyperlink>` tags: SheetJS expands each hyperlink range into\n * per-cell objects before any budget runs, so one `ref=\"A1:XFD1048576\"` exhausts memory. The\n * removed tags are returned so links can still be shown. XLSX input that SheetJS would route to\n * its binary (XLSB), ODS, or Numbers parsers is rejected early with a clear error; the guarantee\n * is that SheetJS only receives `buildSheetJsInput`'s allowlisted archive.\n */\nexport const readOfficeArchive = (\n input: Uint8Array,\n format: OfficeFileFormat,\n limits: OfficeArchiveLimits,\n options: ReadOfficeArchiveOptions = {}\n) =>\n Effect.tryPromise({\n try: async (): Promise<NormalizedOfficeArchive> => {\n const bytes = Buffer.from(input.buffer, input.byteOffset, input.byteLength)\n\n if (bytes.length < 22 || bytes.length > limits.maxInputBytes) throw invalid()\n\n const entries = archiveEntries(bytes, limits)\n const mainPart = mainParts[format]\n const maxHyperlinkTags = options.maxHyperlinkTags ?? 0\n\n if (format === 'xlsx' && entries.some(entry => isAlternateFormatEntry(entry.name)))\n throw unsupportedXlsxParts()\n\n if (\n !entries.some(entry => entry.name === '[Content_Types].xml') ||\n !entries.some(entry => entry.name === mainPart)\n )\n throw invalid()\n\n const validated: Record<string, Uint8Array> = Object.create(null)\n const hyperlinkTags = new Map<string, Array<string>>()\n let capturedTags = 0\n let expandedBytes = 0\n\n for (const entry of entries) {\n const chunks: Array<Buffer> = []\n let entryBytes = 0\n\n const record = (chunk: Buffer) => {\n entryBytes += chunk.length\n expandedBytes += chunk.length\n\n if (expandedBytes > limits.maxExpandedBytes || entryBytes > entry.originalSize)\n throw tooLarge(expandedBytes)\n\n chunks.push(chunk)\n }\n\n const compressed = bytes.subarray(entry.start, entry.start + entry.compressedSize)\n\n if (entry.method === 0) record(compressed)\n else await inflateEntry(compressed, record)\n\n if (entryBytes !== entry.originalSize) throw invalid()\n\n const content = Buffer.concat(chunks, entryBytes)\n\n if (format !== 'xlsx') {\n validated[entry.name] = content\n continue\n }\n\n // Relationship targets need not end in .xml, so scan every entry without changing other\n // bytes (including UTF-8 and binary parts). A space prevents removal from joining\n // attacker-controlled fragments into a new parser-accepted hyperlink tag.\n const tags: Array<string> = []\n\n const rewritten = Buffer.from(\n content.toString('latin1').replace(hyperlinkTag, tag => {\n if (!tag.startsWith('</') && capturedTags < maxHyperlinkTags) {\n capturedTags += 1\n tags.push(tag)\n }\n\n return ' '\n }),\n 'latin1'\n )\n\n // SheetJS also decodes BOM-marked UTF-16 parts, which the Latin-1 strip cannot see\n // through, and the strip itself can shift byte alignment or create a BOM. Check the exact\n // bytes SheetJS will parse; Excel never writes UTF-16 parts, so reject rather than rewrite.\n if (utf16PartHasHyperlink(rewritten)) throw invalid()\n\n // Content types and relationships decide which parser SheetJS runs on each part.\n const lowerName = entry.name.toLowerCase()\n\n if (\n (lowerName === '[content_types].xml' && contentTypesRouteToBinary(rewritten)) ||\n (lowerName.endsWith('.rels') && relationshipsRouteToBinary(rewritten))\n )\n throw unsupportedXlsxParts()\n\n if (tags.length > 0) hyperlinkTags.set(entry.name, tags)\n\n validated[entry.name] = rewritten\n }\n\n return { parts: validated, hyperlinkTags }\n },\n // Out-of-range header reads (RangeError) and inflate failures are malformed archives too.\n catch: error => (error instanceof OfficeArchiveError ? error : invalid())\n })\n\n/**\n * Validate a DOCX, XLSX, or PPTX archive with bounded inflation and return a rebuilt stored-entry\n * ZIP of every validated part, for storage or for other parsers. XLSX parts lose their hyperlink\n * tags, and XLSX input with ODS or Numbers marker entries or XLSB parts is rejected; `.bin` parts\n * SheetJS never parses (printer settings, OLE objects) are kept so stored files still open.\n *\n * The output is not SheetJS input. Never run SheetJS on it directly: extract XLSX text through\n * `FileExtractor`, which hands SheetJS only an allowlisted archive it builds itself (worksheet,\n * shared-string, style, and core-property parts with generated content types and relationships).\n */\nexport const normalizeOfficeArchive = (\n bytes: Uint8Array,\n format: OfficeFileFormat,\n limits: Partial<OfficeArchiveLimits> = {}\n) =>\n readOfficeArchive(bytes, format, {\n maxArchiveEntries: limits.maxArchiveEntries ?? defaultFileExtractorLimits.maxArchiveEntries,\n maxExpandedBytes: limits.maxExpandedBytes ?? defaultFileExtractorLimits.maxExpandedBytes,\n maxInputBytes: limits.maxInputBytes ?? defaultFileExtractorLimits.maxInputBytes\n }).pipe(Effect.map(normalized => storedArchive(normalized.parts)))\n"],"mappings":";;;;;;;;;;AAqBA,MAAM,gBAAgB,IAAI,mBAAmB,EAAE,SAAS,0BAA0B,CAAC;AAEnF,MAAM,kBAAkB;AAExB,MAAM,YAAY,kBAChB,kBAAkB,KAAA,IACd,IAAI,mBAAmB,EAAE,SAAS,gBAAgB,CAAC,IACnD,IAAI,mBAAmB;CAAE,SAAS;CAAiB;AAAc,CAAC;AAExE,MAAM,uBAAuB;;AAG7B,MAAa,0BAA0B,KAAK;AAE5C,MAAM,YAAwD;CAC5D,MAAM;CACN,MAAM;CACN,MAAM;AACR;;AAWA,MAAM,kBAAkB,OAAe,WAAgC;CACrE,IAAI,MAAM,MAAM,SAAS;CACzB,MAAM,WAAW,KAAK,IAAI,GAAG,MAAM,KAAK;CAExC,OAAO,OAAO,YAAY,MAAM,aAAa,GAAG,MAAM,WAAY,OAAO;CAEzE,IAAI,MAAM,YAAY,MAAM,KAAK,MAAM,aAAa,MAAM,EAAE,MAAM,MAAM,QAAQ,MAAM,QAAQ;CAE9F,MAAM,QAAQ,MAAM,aAAa,MAAM,EAAE;CACzC,MAAM,gBAAgB,MAAM,aAAa,MAAM,EAAE;CACjD,MAAM,kBAAkB,MAAM,aAAa,MAAM,EAAE;CAEnD,IACE,MAAM,aAAa,MAAM,CAAC,MAAM,KAChC,MAAM,aAAa,MAAM,CAAC,MAAM,SAChC,UAAU,KACV,QAAQ,OAAO,qBACf,kBAAkB,kBAAkB,KAEpC,MAAM,QAAQ;CAEhB,MAAM,UAA+B,CAAC;CAEtC,MAAM,wBAAQ,IAAI,IAAY;CAC9B,IAAI,SAAS;CACb,IAAI,gBAAgB;CAEpB,KAAK,IAAI,QAAQ,GAAG,QAAQ,OAAO,SAAS,GAAG;EAC7C,IAAI,SAAS,KAAK,OAAO,MAAM,aAAa,MAAM,MAAM,UAAY,MAAM,QAAQ;EAElF,MAAM,QAAQ,MAAM,aAAa,SAAS,CAAC;EAC3C,MAAM,SAAS,MAAM,aAAa,SAAS,EAAE;EAC7C,MAAM,iBAAiB,MAAM,aAAa,SAAS,EAAE;EACrD,MAAM,eAAe,MAAM,aAAa,SAAS,EAAE;EACnD,MAAM,WAAW,MAAM,aAAa,SAAS,EAAE;EAC/C,MAAM,YAAY,MAAM,aAAa,SAAS,EAAE;EAChD,MAAM,cAAc,MAAM,aAAa,SAAS,EAAE;EAClD,MAAM,QAAQ,MAAM,aAAa,SAAS,EAAE;EAC5C,MAAM,OAAO,SAAS,KAAK,WAAW,YAAY;EAGlD,IACE,OAAO,OACP,aAAa,KACb,WAAW,SACV,QAAQ,WAAa,KACrB,WAAW,KAAK,WAAW,KAC5B,MAAM,aAAa,SAAS,EAAE,MAAM,KACpC,mBAAmB,cACnB,iBAAiB,cACjB,QAAQ,KAAK,iBAEb,MAAM,QAAQ;EAEhB,MAAM,YAAY,MAAM,SAAS,SAAS,IAAI,SAAS,KAAK,QAAQ;EACpE,MAAM,OAAO,IAAI,YAAY,SAAS,EAAE,OAAO,KAAK,CAAC,EAAE,OAAO,SAAS;EAGvE,IACE,oBAAoB,KAAK,IAAI,KAC7B,KAAK,WAAW,GAAG,KACnB,KAAK,SAAS,IAAI,KAClB,KAAK,MAAM,GAAG,EAAE,MAAK,SAAQ,SAAS,QAAQ,SAAS,GAAG,KAC1D,MAAM,IAAI,KAAK,YAAY,CAAC,KAC5B,oBAAoB,KAAK,IAAI,GAE7B,MAAM,QAAQ;EAEhB,MAAM,IAAI,KAAK,YAAY,CAAC;EAE5B,IACE,MAAM,aAAa,KAAK,MAAM,YAC9B,MAAM,aAAa,QAAQ,CAAC,MAAM,SAClC,MAAM,aAAa,QAAQ,CAAC,MAAM,UAClC,MAAM,aAAa,QAAQ,EAAE,MAAM,UAEnC,MAAM,QAAQ;EAEhB,MAAM,QAAQ,QAAQ,KAAK,WAAW,MAAM,aAAa,QAAQ,EAAE;EAEnE,IACE,QAAQ,iBAAiB,mBACzB,CAAC,MAAM,SAAS,QAAQ,IAAI,QAAQ,KAAK,QAAQ,EAAE,OAAO,SAAS,GAEnE,MAAM,QAAQ;EAGhB,KAAK,MAAM,CAAC,UAAU,aAAa,CACjC,CAAC,IAAI,cAAc,GACnB,CAAC,IAAI,YAAY,CACnB,GAAY;GACV,MAAM,QAAQ,MAAM,aAAa,QAAQ,QAAQ;GAEjD,IAAI,UAAU,YAAY,GAAG,QAAQ,OAAO,KAAK,UAAU,IAAI,MAAM,QAAQ;EAC/E;EAEA,iBAAiB;EAEjB,IAAI,gBAAgB,OAAO,kBAAkB,MAAM,SAAS;EAE5D,QAAQ,KAAK;GAAE;GAAM;GAAO;GAAgB;GAAc;EAAO,CAAC;EAClE,SAAS;CACX;CAEA,IAAI,WAAW,KAAK,MAAM,QAAQ;CAElC,OAAO;AACT;;;;;AAMA,MAAM,eAAe;AAErB,MAAM,oBAAoB;;;;;;AAO1B,MAAa,yBAAyB,YACpC,iBAAiB,OAAO,EACrB,MAAM,CAAC,EACP,MAAK,SAAQ,kBAAkB,KAAK,IAAI,CAAC;AAE9C,MAAM,6BACJ,IAAI,mBAAmB,EACrB,SAAS,8DACX,CAAC;AAYH,MAAM,eAAe,OAAO,YAAoB,WAAmD;CACjG,UAAU,cAAc;EACtB,KAAK,IAAI,SAAS,GAAG,SAAS,WAAW,QAAQ,UAAU,sBACzD,MAAM,WAAW,SAAS,QAAQ,SAAS,oBAAoB;CAEnE;CAEA,MAAM,SAAS,SAAS,KAAK,YAAY,GAAG,EAAE,eAAe,EAAE,CAAC;CAShE,MAAM,WAAW,iBAAiB;EALhC,WAAW;EACX,uBAAuB;EACvB,uBAAuB;CAGsB,CAAC;CAEhD,OAAO,KAAK,QAAQ;CAEpB,IAAI;EAEF,WAAW,MAAM,SAAS,UAAU;GAClC,IAAI,CAAC,OAAO,SAAS,KAAK,GAAG,MAAM,QAAQ;GAE3C,KAAK,IAAI,SAAS,GAAG,SAAS,MAAM,QAAQ,UAAU,yBACpD,OAAO,MAAM,SAAS,QAAQ,SAAS,uBAAuB,CAAC;EAEnE;EAEA,IAAI,SAAS,iBAAiB,WAAW,QAAQ,MAAM,QAAQ;CACjE,UAAU;EACR,OAAO,QAAQ;EACf,SAAS,QAAQ;CACnB;AACF;;AAQA,MAAa,iBAAiB,UAC5B,QAAQ,EAAE,GAAG,MAAM,GAAG,EAAE,OAAO,EAAE,CAAC;;;;;;;;;;;AAYpC,MAAa,qBACX,OACA,QACA,QACA,UAAoC,CAAC,MAErC,OAAO,WAAW;CAChB,KAAK,YAA8C;EACjD,MAAM,QAAQ,OAAO,KAAK,MAAM,QAAQ,MAAM,YAAY,MAAM,UAAU;EAE1E,IAAI,MAAM,SAAS,MAAM,MAAM,SAAS,OAAO,eAAe,MAAM,QAAQ;EAE5E,MAAM,UAAU,eAAe,OAAO,MAAM;EAC5C,MAAM,WAAW,UAAU;EAC3B,MAAM,mBAAmB,QAAQ,oBAAoB;EAErD,IAAI,WAAW,UAAU,QAAQ,MAAK,UAAS,uBAAuB,MAAM,IAAI,CAAC,GAC/E,MAAM,qBAAqB;EAE7B,IACE,CAAC,QAAQ,MAAK,UAAS,MAAM,SAAS,qBAAqB,KAC3D,CAAC,QAAQ,MAAK,UAAS,MAAM,SAAS,QAAQ,GAE9C,MAAM,QAAQ;EAEhB,MAAM,YAAwC,OAAO,OAAO,IAAI;EAChE,MAAM,gCAAgB,IAAI,IAA2B;EACrD,IAAI,eAAe;EACnB,IAAI,gBAAgB;EAEpB,KAAK,MAAM,SAAS,SAAS;GAC3B,MAAM,SAAwB,CAAC;GAC/B,IAAI,aAAa;GAEjB,MAAM,UAAU,UAAkB;IAChC,cAAc,MAAM;IACpB,iBAAiB,MAAM;IAEvB,IAAI,gBAAgB,OAAO,oBAAoB,aAAa,MAAM,cAChE,MAAM,SAAS,aAAa;IAE9B,OAAO,KAAK,KAAK;GACnB;GAEA,MAAM,aAAa,MAAM,SAAS,MAAM,OAAO,MAAM,QAAQ,MAAM,cAAc;GAEjF,IAAI,MAAM,WAAW,GAAG,OAAO,UAAU;QACpC,MAAM,aAAa,YAAY,MAAM;GAE1C,IAAI,eAAe,MAAM,cAAc,MAAM,QAAQ;GAErD,MAAM,UAAU,OAAO,OAAO,QAAQ,UAAU;GAEhD,IAAI,WAAW,QAAQ;IACrB,UAAU,MAAM,QAAQ;IACxB;GACF;GAKA,MAAM,OAAsB,CAAC;GAE7B,MAAM,YAAY,OAAO,KACvB,QAAQ,SAAS,QAAQ,EAAE,QAAQ,eAAc,QAAO;IACtD,IAAI,CAAC,IAAI,WAAW,IAAI,KAAK,eAAe,kBAAkB;KAC5D,gBAAgB;KAChB,KAAK,KAAK,GAAG;IACf;IAEA,OAAO;GACT,CAAC,GACD,QACF;GAKA,IAAI,sBAAsB,SAAS,GAAG,MAAM,QAAQ;GAGpD,MAAM,YAAY,MAAM,KAAK,YAAY;GAEzC,IACG,cAAc,yBAAyB,0BAA0B,SAAS,KAC1E,UAAU,SAAS,OAAO,KAAK,2BAA2B,SAAS,GAEpE,MAAM,qBAAqB;GAE7B,IAAI,KAAK,SAAS,GAAG,cAAc,IAAI,MAAM,MAAM,IAAI;GAEvD,UAAU,MAAM,QAAQ;EAC1B;EAEA,OAAO;GAAE,OAAO;GAAW;EAAc;CAC3C;CAEA,QAAO,UAAU,iBAAiB,qBAAqB,QAAQ,QAAQ;AACzE,CAAC;;;;;;;;;;;AAYH,MAAa,0BACX,OACA,QACA,SAAuC,CAAC,MAExC,kBAAkB,OAAO,QAAQ;CAC/B,mBAAmB,OAAO,qBAAqB,2BAA2B;CAC1E,kBAAkB,OAAO,oBAAoB,2BAA2B;CACxE,eAAe,OAAO,iBAAiB,2BAA2B;AACpE,CAAC,EAAE,KAAK,OAAO,KAAI,eAAc,cAAc,WAAW,KAAK,CAAC,CAAC"}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
//#region src/node/pptx-text.d.ts
|
|
2
|
+
/** Slide text in slide order, then speaker notes, from the parts of a validated archive. */
|
|
3
|
+
declare const extractPptxText: (parts: Readonly<Record<string, Uint8Array>>) => string;
|
|
4
|
+
//#endregion
|
|
5
|
+
export { extractPptxText };
|
|
6
|
+
//# sourceMappingURL=pptx-text.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pptx-text.d.mts","names":[],"sources":["../../src/node/pptx-text.ts"],"mappings":";;cAkHa,eAAA,GAAmB,KAAA,EAAO,QAAA,CAAS,MAAA,SAAe,UAAA"}
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import { decodeXmlEntities } from "./xml-text.mjs";
|
|
2
|
+
import { strFromU8 } from "fflate";
|
|
3
|
+
//#region src/node/pptx-text.ts
|
|
4
|
+
const slideXmlFile = /^ppt\/slides\/slide(\d+)\.xml$/;
|
|
5
|
+
const notesXmlFile = /^ppt\/notesSlides\/notesSlide(\d+)\.xml$/;
|
|
6
|
+
const optionalXmlPrefix = `(?:[A-Za-z_][\\w.-]*:)?`;
|
|
7
|
+
const xmlElement = (localName) => ({
|
|
8
|
+
start: new RegExp(`<${optionalXmlPrefix}${localName}\\b[^<>]*>`, "g"),
|
|
9
|
+
end: new RegExp(`</${optionalXmlPrefix}${localName}>`, "g")
|
|
10
|
+
});
|
|
11
|
+
const paragraphXml = xmlElement("p");
|
|
12
|
+
const textXml = xmlElement("t");
|
|
13
|
+
const lineBreakXml = new RegExp(`<${optionalXmlPrefix}br\\b[^<>]*/>`, "g");
|
|
14
|
+
const tabXml = new RegExp(`<${optionalXmlPrefix}tab\\b[^<>]*/>`, "g");
|
|
15
|
+
/** Each start tag paired with the next end tag after it (the old lazy `[\s\S]*?` match). */
|
|
16
|
+
const elementMatches = (xml, element) => {
|
|
17
|
+
const matches = [];
|
|
18
|
+
let position = 0;
|
|
19
|
+
for (;;) {
|
|
20
|
+
element.start.lastIndex = position;
|
|
21
|
+
const start = element.start.exec(xml);
|
|
22
|
+
if (start === null) return matches;
|
|
23
|
+
const contentStart = start.index + start[0].length;
|
|
24
|
+
element.end.lastIndex = contentStart;
|
|
25
|
+
const end = element.end.exec(xml);
|
|
26
|
+
if (end === null) return matches;
|
|
27
|
+
position = end.index + end[0].length;
|
|
28
|
+
matches.push({
|
|
29
|
+
outer: xml.slice(start.index, position),
|
|
30
|
+
inner: xml.slice(contentStart, end.index)
|
|
31
|
+
});
|
|
32
|
+
}
|
|
33
|
+
};
|
|
34
|
+
const indexedXmlFile = (fileName, bytes, pattern, group) => {
|
|
35
|
+
const indexText = pattern.exec(fileName)?.[1];
|
|
36
|
+
if (indexText === void 0) return void 0;
|
|
37
|
+
const index = Number.parseInt(indexText, 10);
|
|
38
|
+
if (!Number.isInteger(index)) return void 0;
|
|
39
|
+
return {
|
|
40
|
+
fileName,
|
|
41
|
+
bytes,
|
|
42
|
+
group,
|
|
43
|
+
index
|
|
44
|
+
};
|
|
45
|
+
};
|
|
46
|
+
const pptxXmlFile = (fileName, bytes) => indexedXmlFile(fileName, bytes, slideXmlFile, 0) ?? indexedXmlFile(fileName, bytes, notesXmlFile, 1);
|
|
47
|
+
const comparePptxXmlFiles = (left, right) => left.group - right.group || left.index - right.index || left.fileName.localeCompare(right.fileName);
|
|
48
|
+
const extractParagraphText = (paragraph) => {
|
|
49
|
+
return elementMatches(paragraph.replace(lineBreakXml, "<a:t>\n</a:t>").replace(tabXml, "<a:t> </a:t>"), textXml).map((match) => decodeXmlEntities(match.inner)).join("").trim();
|
|
50
|
+
};
|
|
51
|
+
const extractXmlText = (xml) => {
|
|
52
|
+
const paragraphs = elementMatches(xml, paragraphXml).map((match) => match.outer);
|
|
53
|
+
return (paragraphs.length > 0 ? paragraphs : [xml]).map(extractParagraphText).filter((text) => text.length > 0).join("\n");
|
|
54
|
+
};
|
|
55
|
+
/** Slide text in slide order, then speaker notes, from the parts of a validated archive. */
|
|
56
|
+
const extractPptxText = (parts) => Object.entries(parts).flatMap(([fileName, fileBytes]) => {
|
|
57
|
+
const xmlFile = pptxXmlFile(fileName, fileBytes);
|
|
58
|
+
return xmlFile === void 0 ? [] : [xmlFile];
|
|
59
|
+
}).sort(comparePptxXmlFiles).map((file) => extractXmlText(strFromU8(file.bytes))).filter((text) => text.length > 0).join("\n\n");
|
|
60
|
+
//#endregion
|
|
61
|
+
export { extractPptxText };
|
|
62
|
+
|
|
63
|
+
//# sourceMappingURL=pptx-text.mjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"pptx-text.mjs","names":[],"sources":["../../src/node/pptx-text.ts"],"sourcesContent":["import { strFromU8 } from 'fflate'\nimport { decodeXmlEntities } from './xml-text.ts'\n\ntype PptxXmlFile = {\n readonly fileName: string\n readonly bytes: Uint8Array\n readonly group: number\n readonly index: number\n}\n\nconst slideXmlFile = /^ppt\\/slides\\/slide(\\d+)\\.xml$/\n\nconst notesXmlFile = /^ppt\\/notesSlides\\/notesSlide(\\d+)\\.xml$/\n\nconst xmlName = '[A-Za-z_][\\\\w.-]*'\n\nconst optionalXmlPrefix = `(?:${xmlName}:)?`\n\n// Every pattern stops at the next `<` (`[^<>]`), and elements are paired in one forward pass, so\n// extraction stays linear in the part size even for unterminated or unbalanced markup.\ntype XmlElement = { readonly start: RegExp; readonly end: RegExp }\n\nconst xmlElement = (localName: string): XmlElement => ({\n start: new RegExp(`<${optionalXmlPrefix}${localName}\\\\b[^<>]*>`, 'g'),\n end: new RegExp(`</${optionalXmlPrefix}${localName}>`, 'g')\n})\n\nconst paragraphXml = xmlElement('p')\n\nconst textXml = xmlElement('t')\n\nconst lineBreakXml = new RegExp(`<${optionalXmlPrefix}br\\\\b[^<>]*/>`, 'g')\n\nconst tabXml = new RegExp(`<${optionalXmlPrefix}tab\\\\b[^<>]*/>`, 'g')\n\ntype ElementMatch = {\n /** Start tag through end tag. */\n readonly outer: string\n /** Text between the start and end tags. */\n readonly inner: string\n}\n\n/** Each start tag paired with the next end tag after it (the old lazy `[\\s\\S]*?` match). */\nconst elementMatches = (xml: string, element: XmlElement): ReadonlyArray<ElementMatch> => {\n const matches: Array<ElementMatch> = []\n let position = 0\n\n for (;;) {\n element.start.lastIndex = position\n const start = element.start.exec(xml)\n\n if (start === null) return matches\n\n const contentStart = start.index + start[0].length\n element.end.lastIndex = contentStart\n const end = element.end.exec(xml)\n\n // No end tag after this start means none after any later start either.\n if (end === null) return matches\n\n position = end.index + end[0].length\n matches.push({\n outer: xml.slice(start.index, position),\n inner: xml.slice(contentStart, end.index)\n })\n }\n}\n\nconst indexedXmlFile = (\n fileName: string,\n bytes: Uint8Array,\n pattern: RegExp,\n group: number\n): PptxXmlFile | undefined => {\n const indexText = pattern.exec(fileName)?.[1]\n\n if (indexText === undefined) return undefined\n\n const index = Number.parseInt(indexText, 10)\n\n if (!Number.isInteger(index)) return undefined\n\n return { fileName, bytes, group, index }\n}\n\nconst pptxXmlFile = (fileName: string, bytes: Uint8Array): PptxXmlFile | undefined =>\n indexedXmlFile(fileName, bytes, slideXmlFile, 0) ??\n indexedXmlFile(fileName, bytes, notesXmlFile, 1)\n\nconst comparePptxXmlFiles = (left: PptxXmlFile, right: PptxXmlFile) =>\n left.group - right.group ||\n left.index - right.index ||\n left.fileName.localeCompare(right.fileName)\n\nconst extractParagraphText = (paragraph: string) => {\n const xml = paragraph.replace(lineBreakXml, '<a:t>\\n</a:t>').replace(tabXml, '<a:t>\\t</a:t>')\n\n return elementMatches(xml, textXml)\n .map(match => decodeXmlEntities(match.inner))\n .join('')\n .trim()\n}\n\nconst extractXmlText = (xml: string) => {\n const paragraphs = elementMatches(xml, paragraphXml).map(match => match.outer)\n const textSources = paragraphs.length > 0 ? paragraphs : [xml]\n\n return textSources\n .map(extractParagraphText)\n .filter(text => text.length > 0)\n .join('\\n')\n}\n\n/** Slide text in slide order, then speaker notes, from the parts of a validated archive. */\nexport const extractPptxText = (parts: Readonly<Record<string, Uint8Array>>) =>\n Object.entries(parts)\n .flatMap(([fileName, fileBytes]) => {\n const xmlFile = pptxXmlFile(fileName, fileBytes)\n\n return xmlFile === undefined ? [] : [xmlFile]\n })\n .sort(comparePptxXmlFiles)\n .map(file => extractXmlText(strFromU8(file.bytes)))\n .filter(text => text.length > 0)\n .join('\\n\\n')\n"],"mappings":";;;AAUA,MAAM,eAAe;AAErB,MAAM,eAAe;AAIrB,MAAM,oBAAoB;AAM1B,MAAM,cAAc,eAAmC;CACrD,OAAO,IAAI,OAAO,IAAI,oBAAoB,UAAU,aAAa,GAAG;CACpE,KAAK,IAAI,OAAO,KAAK,oBAAoB,UAAU,IAAI,GAAG;AAC5D;AAEA,MAAM,eAAe,WAAW,GAAG;AAEnC,MAAM,UAAU,WAAW,GAAG;AAE9B,MAAM,eAAe,IAAI,OAAO,IAAI,kBAAkB,gBAAgB,GAAG;AAEzE,MAAM,SAAS,IAAI,OAAO,IAAI,kBAAkB,iBAAiB,GAAG;;AAUpE,MAAM,kBAAkB,KAAa,YAAqD;CACxF,MAAM,UAA+B,CAAC;CACtC,IAAI,WAAW;CAEf,SAAS;EACP,QAAQ,MAAM,YAAY;EAC1B,MAAM,QAAQ,QAAQ,MAAM,KAAK,GAAG;EAEpC,IAAI,UAAU,MAAM,OAAO;EAE3B,MAAM,eAAe,MAAM,QAAQ,MAAM,GAAG;EAC5C,QAAQ,IAAI,YAAY;EACxB,MAAM,MAAM,QAAQ,IAAI,KAAK,GAAG;EAGhC,IAAI,QAAQ,MAAM,OAAO;EAEzB,WAAW,IAAI,QAAQ,IAAI,GAAG;EAC9B,QAAQ,KAAK;GACX,OAAO,IAAI,MAAM,MAAM,OAAO,QAAQ;GACtC,OAAO,IAAI,MAAM,cAAc,IAAI,KAAK;EAC1C,CAAC;CACH;AACF;AAEA,MAAM,kBACJ,UACA,OACA,SACA,UAC4B;CAC5B,MAAM,YAAY,QAAQ,KAAK,QAAQ,IAAI;CAE3C,IAAI,cAAc,KAAA,GAAW,OAAO,KAAA;CAEpC,MAAM,QAAQ,OAAO,SAAS,WAAW,EAAE;CAE3C,IAAI,CAAC,OAAO,UAAU,KAAK,GAAG,OAAO,KAAA;CAErC,OAAO;EAAE;EAAU;EAAO;EAAO;CAAM;AACzC;AAEA,MAAM,eAAe,UAAkB,UACrC,eAAe,UAAU,OAAO,cAAc,CAAC,KAC/C,eAAe,UAAU,OAAO,cAAc,CAAC;AAEjD,MAAM,uBAAuB,MAAmB,UAC9C,KAAK,QAAQ,MAAM,SACnB,KAAK,QAAQ,MAAM,SACnB,KAAK,SAAS,cAAc,MAAM,QAAQ;AAE5C,MAAM,wBAAwB,cAAsB;CAGlD,OAAO,eAFK,UAAU,QAAQ,cAAc,eAAe,EAAE,QAAQ,QAAQ,cAErD,GAAG,OAAO,EAC/B,KAAI,UAAS,kBAAkB,MAAM,KAAK,CAAC,EAC3C,KAAK,EAAE,EACP,KAAK;AACV;AAEA,MAAM,kBAAkB,QAAgB;CACtC,MAAM,aAAa,eAAe,KAAK,YAAY,EAAE,KAAI,UAAS,MAAM,KAAK;CAG7E,QAFoB,WAAW,SAAS,IAAI,aAAa,CAAC,GAAG,GAG1D,IAAI,oBAAoB,EACxB,QAAO,SAAQ,KAAK,SAAS,CAAC,EAC9B,KAAK,IAAI;AACd;;AAGA,MAAa,mBAAmB,UAC9B,OAAO,QAAQ,KAAK,EACjB,SAAS,CAAC,UAAU,eAAe;CAClC,MAAM,UAAU,YAAY,UAAU,SAAS;CAE/C,OAAO,YAAY,KAAA,IAAY,CAAC,IAAI,CAAC,OAAO;AAC9C,CAAC,EACA,KAAK,mBAAmB,EACxB,KAAI,SAAQ,eAAe,UAAU,KAAK,KAAK,CAAC,CAAC,EACjD,QAAO,SAAQ,KAAK,SAAS,CAAC,EAC9B,KAAK,MAAM"}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
//#region src/node/sheetjs-xml.d.ts
|
|
2
|
+
/**
|
|
3
|
+
* The texts SheetJS can read from a part: its Latin-1 ("binary") view and, for BOM-marked parts,
|
|
4
|
+
* the UTF-16 decodings of `cc2str` (little- and big-endian from byte 2, including its
|
|
5
|
+
* `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
|
|
6
|
+
*/
|
|
7
|
+
declare const sheetJsTextViews: (content: Uint8Array) => ReadonlyArray<string>;
|
|
8
|
+
type SheetJsTag = {
|
|
9
|
+
/** The tag up to its first space, line feed, or carriage return (SheetJS `y[0]`). */readonly head: string; /** Raw (still escaped) attribute values by SheetJS key, plus lower-cased copies. */
|
|
10
|
+
readonly attributes: ReadonlyMap<string, string>;
|
|
11
|
+
};
|
|
12
|
+
/**
|
|
13
|
+
* Port of SheetJS `parsexmltag`: exact-case keys (plus lower-cased copies), a namespace prefix
|
|
14
|
+
* dropped, an unprefixed name cut at its first `_`, the last value winning. Values are raw.
|
|
15
|
+
*/
|
|
16
|
+
declare const parseSheetJsTag: (tag: string) => SheetJsTag;
|
|
17
|
+
/** Every tag SheetJS's pattern finds in `text`, in document order. */
|
|
18
|
+
declare function sheetJsTags(text: string): Generator<SheetJsTag>;
|
|
19
|
+
/** SheetJS `strip_ns`: the first `<prefix:` (or `</prefix:`) loses its prefix. */
|
|
20
|
+
declare const stripSheetJsNamespace: (head: string) => string;
|
|
21
|
+
/** SheetJS `utf8read` in Node: the Latin-1 ("binary") string read back as UTF-8. */
|
|
22
|
+
declare const sheetJsUtf8Read: (binary: string) => string;
|
|
23
|
+
/**
|
|
24
|
+
* SheetJS `unescapexml` for text without CDATA, quirks included: entity names match ignoring
|
|
25
|
+
* case but only lower-case ones map (`"` becomes U+0000), `A` is read as decimal, and
|
|
26
|
+
* numeric references wrap at U+FFFF. Returns `undefined` for text with a CDATA marker, which
|
|
27
|
+
* SheetJS splits recursively (and never ends for an unterminated one).
|
|
28
|
+
*/
|
|
29
|
+
declare const sheetJsUnescapeXml: (text: string) => string | undefined;
|
|
30
|
+
/** An attribute value as SheetJS reads text attributes: `unescapexml(utf8read(raw))`. */
|
|
31
|
+
declare const sheetJsAttributeText: (raw: string) => string | undefined;
|
|
32
|
+
/**
|
|
33
|
+
* Escape `text` for a double-quoted attribute of a generated UTF-8 part so that SheetJS's
|
|
34
|
+
* `unescapexml(utf8read(…))` returns `text` exactly: markup characters become entities, an `_`
|
|
35
|
+
* that would start an `_xHHHH_` code becomes `_x005F_`, and control characters, U+FFFE, U+FFFF,
|
|
36
|
+
* and lone surrogates become `_xHHHH_` codes.
|
|
37
|
+
*/
|
|
38
|
+
declare const sheetJsAttributeEscape: (text: string) => string;
|
|
39
|
+
/** SheetJS's own `XML_HEADER`, used for every generated part. */
|
|
40
|
+
declare const sheetJsXmlHeader = "<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"yes\"?>\r\n";
|
|
41
|
+
declare const spreadsheetMainNamespace = "http://schemas.openxmlformats.org/spreadsheetml/2006/main";
|
|
42
|
+
declare const officeDocumentRelationshipsNamespace = "http://schemas.openxmlformats.org/officeDocument/2006/relationships";
|
|
43
|
+
/**
|
|
44
|
+
* `text` without its simple opening tags, `<(?:[\w.-]+:)?[\w.-]+>`: a superset of the tags SheetJS
|
|
45
|
+
* removes before decoding (`<(?:\w+:)?(?:si|sstItem)>` in `parse_sst_xml`, `<(?:\w+:)?r>` in
|
|
46
|
+
* `parse_rs`). One forward scan: a `<` that does not start such a tag is kept and the scan resumes
|
|
47
|
+
* at the next `<`, so every character is read at most twice.
|
|
48
|
+
*/
|
|
49
|
+
declare const withoutSimpleTags: (text: string) => string;
|
|
50
|
+
/**
|
|
51
|
+
* Whether SheetJS could meet a CDATA marker in this part. SheetJS's `unescapexml` handles CDATA by
|
|
52
|
+
* recursing on a string two characters shorter and copying the whole tail at every level, so an
|
|
53
|
+
* unterminated marker costs quadratic time and memory.
|
|
54
|
+
*
|
|
55
|
+
* What SheetJS hands to `unescapexml` in a worksheet or shared-strings part comes from the part
|
|
56
|
+
* text through two kinds of transformation:
|
|
57
|
+
*
|
|
58
|
+
* - Decodes: raw (`<v>` of every cell), `utf8read(raw)` (shared and inline strings), and
|
|
59
|
+
* `utf8read(unescapexml(raw))` (cells of type `str`, decoded again after `utf8read`).
|
|
60
|
+
* `utf8read` keeps only the low byte of each character, so U+013C from `_x013C_`, `ļ`, or
|
|
61
|
+
* `ļ` becomes `<`.
|
|
62
|
+
* - Tag removal before decoding: `parse_sst_xml` removes every `<si>`/`<sstItem>` opening tag
|
|
63
|
+
* from the whole shared-strings table, and `parse_rs` removes every `<r>` opening tag from rich
|
|
64
|
+
* text (after `utf8read`). Inline strings (`t="inlineStr"`) call `parse_si` without options,
|
|
65
|
+
* so their rich text is processed even with `cellHTML: false`. `A<<r>![CDATA[B` thus reaches
|
|
66
|
+
* `unescapexml` as `A<![CDATA[B`.
|
|
67
|
+
*
|
|
68
|
+
* So the check rejects a text view (`sheetJsTextViews`) when:
|
|
69
|
+
*
|
|
70
|
+
* - the view, or `utf8read` of it, contains `<<` or `<!`. A marker assembled by removing tags
|
|
71
|
+
* needs a literal `<` (in the view, or from `utf8read`) followed by a removed tag, or by `!`,
|
|
72
|
+
* and every removed tag starts with `<`;
|
|
73
|
+
* - the view, or the view without any simple opening tag (`withoutSimpleTags`, a superset of
|
|
74
|
+
* SheetJS's removals), meets the marker raw or after any chain of up to two steps of
|
|
75
|
+
* `unescapexml` and `utf8read`, in any order (a superset of SheetJS's decode sequences).
|
|
76
|
+
*
|
|
77
|
+
* Every step is a linear pass, at most a few per view. The first rule deliberately fails closed:
|
|
78
|
+
* it also rejects XML comments, `<!DOCTYPE`, and any other `<!…` declaration, which Excel,
|
|
79
|
+
* LibreOffice, and Google Sheets never write in worksheets or shared strings (nor a literal `<<`).
|
|
80
|
+
*/
|
|
81
|
+
declare const sheetJsCouldReadCdata: (content: Uint8Array) => boolean;
|
|
82
|
+
/**
|
|
83
|
+
* The `dc:title` of a core-properties part, read as SheetJS `parse_core_props` reads it, without
|
|
84
|
+
* handing the part to SheetJS. A title holding CDATA is ignored.
|
|
85
|
+
*/
|
|
86
|
+
declare const coreTitle: (content: Uint8Array | undefined) => string | undefined;
|
|
87
|
+
//#endregion
|
|
88
|
+
export { SheetJsTag, coreTitle, officeDocumentRelationshipsNamespace, parseSheetJsTag, sheetJsAttributeEscape, sheetJsAttributeText, sheetJsCouldReadCdata, sheetJsTags, sheetJsTextViews, sheetJsUnescapeXml, sheetJsUtf8Read, sheetJsXmlHeader, spreadsheetMainNamespace, stripSheetJsNamespace, withoutSimpleTags };
|
|
89
|
+
//# sourceMappingURL=sheetjs-xml.d.mts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sheetjs-xml.d.mts","names":[],"sources":["../../src/node/sheetjs-xml.ts"],"mappings":";;AAgBA;;;;cAAa,gBAAA,GAAoB,OAAA,EAAS,UAAA,KAAa,aAAa;AAAA,KA2BxD,UAAA;EA3B2C,8FA6B5C,IAAA,UA7ByD;EAAA,SA+BzD,UAAA,EAAY,WAAW;AAAA;;;;;cAOrB,eAAA,GAAmB,GAAA,aAAc,UAwC7C;;iBAGgB,WAAA,CAAY,IAAA,WAAe,SAAS,CAAC,UAAA;AA3CtD;AAAA,cAgDa,qBAAA,GAAyB,IAAY;;cAGrC,eAAA,GAAmB,MAAc;AAX7C;AAGD;;;;;AAHC,cA2BY,kBAAA,GAAsB,IAAY;;cAgBlC,oBAAA,GAAwB,GAAW;AAxCgB;AAKhE;;;;AAAkD;AALc,cAgEnD,sBAAA,GAA0B,IAAY;;cAuCtC,gBAAA;AAAA,cAEA,wBAAA;AAAA,cAEA,oCAAA;AAnFb;;;;AAA+C;AAgB/C;AAhBA,cAiIa,iBAAA,GAAqB,IAAY;;;AAjHE;AAwBhD;;;;AAAmD;AAuCnD;;;;AAA6B;AAE7B;;;;AAAqC;AAErC;;;;AAAiD;AA8CjD;;;;AAA8C;AA4D9C;;;cAAa,qBAAA,GAAyB,OAAmB,EAAV,UAAU;AAAA;AAkCzD;;;AAlCyD,cAkC5C,SAAA,GAAa,OAA+B,EAAtB,UAAU"}
|