@yolk-sdk/extractors 0.1.0-canary.98

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +318 -0
  3. package/dist/errors.d.mts +60 -0
  4. package/dist/errors.d.mts.map +1 -0
  5. package/dist/errors.mjs +69 -0
  6. package/dist/errors.mjs.map +1 -0
  7. package/dist/format.d.mts +32 -0
  8. package/dist/format.d.mts.map +1 -0
  9. package/dist/format.mjs +52 -0
  10. package/dist/format.mjs.map +1 -0
  11. package/dist/index.d.mts +6 -0
  12. package/dist/index.mjs +6 -0
  13. package/dist/knowledge.d.mts +17 -0
  14. package/dist/knowledge.d.mts.map +1 -0
  15. package/dist/knowledge.mjs +77 -0
  16. package/dist/knowledge.mjs.map +1 -0
  17. package/dist/limits.d.mts +28 -0
  18. package/dist/limits.d.mts.map +1 -0
  19. package/dist/limits.mjs +43 -0
  20. package/dist/limits.mjs.map +1 -0
  21. package/dist/node/extract-file.d.mts +31 -0
  22. package/dist/node/extract-file.d.mts.map +1 -0
  23. package/dist/node/extract-file.mjs +183 -0
  24. package/dist/node/extract-file.mjs.map +1 -0
  25. package/dist/node/extraction-isolation.d.mts +94 -0
  26. package/dist/node/extraction-isolation.d.mts.map +1 -0
  27. package/dist/node/extraction-isolation.mjs +155 -0
  28. package/dist/node/extraction-isolation.mjs.map +1 -0
  29. package/dist/node/extraction-worker-protocol.d.mts +60 -0
  30. package/dist/node/extraction-worker-protocol.d.mts.map +1 -0
  31. package/dist/node/extraction-worker-protocol.mjs +105 -0
  32. package/dist/node/extraction-worker-protocol.mjs.map +1 -0
  33. package/dist/node/extraction-worker.d.mts +1 -0
  34. package/dist/node/extraction-worker.mjs +114729 -0
  35. package/dist/node/index.d.mts +6 -0
  36. package/dist/node/index.mjs +5 -0
  37. package/dist/node/live-layer.d.mts +36 -0
  38. package/dist/node/live-layer.d.mts.map +1 -0
  39. package/dist/node/live-layer.mjs +70 -0
  40. package/dist/node/live-layer.mjs.map +1 -0
  41. package/dist/node/office-archive.d.mts +51 -0
  42. package/dist/node/office-archive.d.mts.map +1 -0
  43. package/dist/node/office-archive.mjs +193 -0
  44. package/dist/node/office-archive.mjs.map +1 -0
  45. package/dist/node/pptx-text.d.mts +6 -0
  46. package/dist/node/pptx-text.d.mts.map +1 -0
  47. package/dist/node/pptx-text.mjs +63 -0
  48. package/dist/node/pptx-text.mjs.map +1 -0
  49. package/dist/node/sheetjs-xml.d.mts +89 -0
  50. package/dist/node/sheetjs-xml.d.mts.map +1 -0
  51. package/dist/node/sheetjs-xml.mjs +253 -0
  52. package/dist/node/sheetjs-xml.mjs.map +1 -0
  53. package/dist/node/sheetjs.d.mts +62 -0
  54. package/dist/node/sheetjs.d.mts.map +1 -0
  55. package/dist/node/sheetjs.mjs +122 -0
  56. package/dist/node/sheetjs.mjs.map +1 -0
  57. package/dist/node/worker-admission.d.mts +58 -0
  58. package/dist/node/worker-admission.d.mts.map +1 -0
  59. package/dist/node/worker-admission.mjs +107 -0
  60. package/dist/node/worker-admission.mjs.map +1 -0
  61. package/dist/node/xlsx-hyperlinks.d.mts +34 -0
  62. package/dist/node/xlsx-hyperlinks.d.mts.map +1 -0
  63. package/dist/node/xlsx-hyperlinks.mjs +159 -0
  64. package/dist/node/xlsx-hyperlinks.mjs.map +1 -0
  65. package/dist/node/xlsx-parts.d.mts +29 -0
  66. package/dist/node/xlsx-parts.d.mts.map +1 -0
  67. package/dist/node/xlsx-parts.mjs +49 -0
  68. package/dist/node/xlsx-parts.mjs.map +1 -0
  69. package/dist/node/xlsx-range.d.mts +21 -0
  70. package/dist/node/xlsx-range.d.mts.map +1 -0
  71. package/dist/node/xlsx-range.mjs +49 -0
  72. package/dist/node/xlsx-range.mjs.map +1 -0
  73. package/dist/node/xlsx-routing.d.mts +36 -0
  74. package/dist/node/xlsx-routing.d.mts.map +1 -0
  75. package/dist/node/xlsx-routing.mjs +115 -0
  76. package/dist/node/xlsx-routing.mjs.map +1 -0
  77. package/dist/node/xlsx-sheetjs-input.d.mts +29 -0
  78. package/dist/node/xlsx-sheetjs-input.d.mts.map +1 -0
  79. package/dist/node/xlsx-sheetjs-input.mjs +165 -0
  80. package/dist/node/xlsx-sheetjs-input.mjs.map +1 -0
  81. package/dist/node/xlsx-styles.d.mts +37 -0
  82. package/dist/node/xlsx-styles.d.mts.map +1 -0
  83. package/dist/node/xlsx-styles.mjs +96 -0
  84. package/dist/node/xlsx-styles.mjs.map +1 -0
  85. package/dist/node/xlsx-text.d.mts +32 -0
  86. package/dist/node/xlsx-text.d.mts.map +1 -0
  87. package/dist/node/xlsx-text.mjs +181 -0
  88. package/dist/node/xlsx-text.mjs.map +1 -0
  89. package/dist/node/xlsx-workbook.d.mts +32 -0
  90. package/dist/node/xlsx-workbook.d.mts.map +1 -0
  91. package/dist/node/xlsx-workbook.mjs +70 -0
  92. package/dist/node/xlsx-workbook.mjs.map +1 -0
  93. package/dist/node/xml-text.d.mts +12 -0
  94. package/dist/node/xml-text.d.mts.map +1 -0
  95. package/dist/node/xml-text.mjs +51 -0
  96. package/dist/node/xml-text.mjs.map +1 -0
  97. package/dist/sanitize.d.mts +6 -0
  98. package/dist/sanitize.d.mts.map +1 -0
  99. package/dist/sanitize.mjs +11 -0
  100. package/dist/sanitize.mjs.map +1 -0
  101. package/dist/service.d.mts +22 -0
  102. package/dist/service.d.mts.map +1 -0
  103. package/dist/service.mjs +11 -0
  104. package/dist/service.mjs.map +1 -0
  105. package/package.json +87 -0
  106. package/src/errors.ts +96 -0
  107. package/src/format.ts +84 -0
  108. package/src/index.ts +32 -0
  109. package/src/knowledge.ts +101 -0
  110. package/src/limits.ts +49 -0
  111. package/src/node/extract-file.ts +269 -0
  112. package/src/node/extraction-isolation.ts +289 -0
  113. package/src/node/extraction-worker-protocol.ts +130 -0
  114. package/src/node/extraction-worker.ts +56 -0
  115. package/src/node/index.ts +21 -0
  116. package/src/node/live-layer.ts +136 -0
  117. package/src/node/office-archive.ts +368 -0
  118. package/src/node/pptx-text.ts +125 -0
  119. package/src/node/sheetjs-xml.ts +356 -0
  120. package/src/node/sheetjs.ts +177 -0
  121. package/src/node/worker-admission.ts +162 -0
  122. package/src/node/xlsx-hyperlinks.ts +260 -0
  123. package/src/node/xlsx-parts.ts +83 -0
  124. package/src/node/xlsx-range.ts +70 -0
  125. package/src/node/xlsx-routing.ts +171 -0
  126. package/src/node/xlsx-sheetjs-input.ts +275 -0
  127. package/src/node/xlsx-styles.ts +160 -0
  128. package/src/node/xlsx-text.ts +288 -0
  129. package/src/node/xlsx-workbook.ts +133 -0
  130. package/src/node/xml-text.ts +77 -0
  131. package/src/sanitize.ts +18 -0
  132. package/src/service.ts +21 -0
@@ -0,0 +1,6 @@
1
+ import { FileExtractor, FileExtractorApi } from "../service.mjs";
2
+ import { OfficeArchiveLimits, normalizeOfficeArchive } from "./office-archive.mjs";
3
+ import { SheetJsLoader } from "./sheetjs.mjs";
4
+ import { FileExtractorIsolation, WorkerIsolationOptions, defaultWorkerIsolation } from "./extraction-isolation.mjs";
5
+ import { FileExtractorLayer, FileExtractorOptions, makeFileExtractorLayer } from "./live-layer.mjs";
6
+ export { FileExtractor, type FileExtractorApi, type FileExtractorIsolation, FileExtractorLayer, type FileExtractorOptions, type OfficeArchiveLimits, type SheetJsLoader, type WorkerIsolationOptions, defaultWorkerIsolation, makeFileExtractorLayer, normalizeOfficeArchive };
@@ -0,0 +1,5 @@
1
+ import { FileExtractor } from "../service.mjs";
2
+ import { normalizeOfficeArchive } from "./office-archive.mjs";
3
+ import { defaultWorkerIsolation } from "./extraction-isolation.mjs";
4
+ import { FileExtractorLayer, makeFileExtractorLayer } from "./live-layer.mjs";
5
+ export { FileExtractor, FileExtractorLayer, defaultWorkerIsolation, makeFileExtractorLayer, normalizeOfficeArchive };
@@ -0,0 +1,36 @@
1
+ import { FileExtractorLimits } from "../limits.mjs";
2
+ import { FileExtractor } from "../service.mjs";
3
+ import { SheetJsLoader } from "./sheetjs.mjs";
4
+ import { FileExtractorIsolation } from "./extraction-isolation.mjs";
5
+ import { Layer } from "effect";
6
+
7
+ //#region src/node/live-layer.d.ts
8
+ type FileExtractorOptions = {
9
+ /** Override any default limit; see `defaultFileExtractorLimits`. */readonly limits?: Partial<FileExtractorLimits>;
10
+ /**
11
+ * Where parsers run (default `'worker'`): each PDF, DOCX, XLSX, and PPTX extraction runs in a
12
+ * fresh worker thread with V8 heap, stack, and time limits, admitted through a pool of 4 workers
13
+ * per JavaScript realm; pass an object to change them. `'none'` parses in the calling thread and is
14
+ * unsafe for untrusted input.
15
+ */
16
+ readonly isolation?: FileExtractorIsolation;
17
+ /**
18
+ * Load SheetJS in the calling thread. Only with `isolation: 'none'` (a worker cannot receive a
19
+ * function and always imports the installed `xlsx` itself); any other isolation is a defect
20
+ * when the layer is built. Defaults to a lazy `import('xlsx')`. Missing, non-SheetJS, or
21
+ * pre-0.20.3 modules fail with `SheetJsUnavailableError`.
22
+ */
23
+ readonly loadSheetJs?: SheetJsLoader;
24
+ };
25
+ /**
26
+ * Node `FileExtractor` layer with custom limits, isolation, or (in-process) SheetJS loader.
27
+ * Invalid limits or isolation settings are defects.
28
+ */
29
+ declare const makeFileExtractorLayer: (options?: FileExtractorOptions) => Layer.Layer<FileExtractor, never, never>;
30
+ /**
31
+ * Node `FileExtractor` layer with the default limits, worker isolation, and lazy SheetJS loading.
32
+ */
33
+ declare const FileExtractorLayer: Layer.Layer<FileExtractor, never, never>;
34
+ //#endregion
35
+ export { FileExtractorLayer, FileExtractorOptions, makeFileExtractorLayer };
36
+ //# sourceMappingURL=live-layer.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"live-layer.d.mts","names":[],"sources":["../../src/node/live-layer.ts"],"mappings":";;;;;;;KAmBY,oBAAA;+EAED,MAAA,GAAS,OAAA,CAAQ,mBAAA;EAFI;;;;;;EAAA,SASrB,SAAA,GAAY,sBAAA;EAOe;;;;;;EAAA,SAA3B,WAAA,GAAc,aAAA;AAAA;;AAAa;AAkFtC;;cAAa,sBAAA,GAA0B,OAAA,GAAS,oBAAA,KAAyB,KAAA,CAAA,KAAA,CAAA,aAAA;;;;cAkB5D,kBAAA,EAAkB,KAAA,CAAA,KAAA,CAAA,aAAA"}
@@ -0,0 +1,70 @@
1
+ import { FileExtractionError, UnsupportedFileFormatError } from "../errors.mjs";
2
+ import { fileFormatFor } from "../format.mjs";
3
+ import { FileExtractorLimits, defaultFileExtractorLimits } from "../limits.mjs";
4
+ import { FileExtractor } from "../service.mjs";
5
+ import { defaultSheetJsLoader } from "./sheetjs.mjs";
6
+ import { extractParsedFile, isParsedFileFormat, makeExtractedFile } from "./extract-file.mjs";
7
+ import { WorkerIsolationSettings, defaultExtractionWorkerUrl, defaultWorkerIsolation, makeWorkerExtractor } from "./extraction-isolation.mjs";
8
+ import { Effect, Layer } from "effect";
9
+ import * as Schema from "effect/Schema";
10
+ //#region src/node/live-layer.ts
11
+ const decodeText = (bytes) => new TextDecoder("utf-8", { fatal: false }).decode(bytes);
12
+ /** Build the Node `FileExtractor` from validated limits and the parser runner. */
13
+ const makeFileExtractor = (limits, extractParsed) => ({ extract: (input) => Effect.gen(function* () {
14
+ const format = fileFormatFor(input);
15
+ if (format === void 0) return yield* Effect.fail(new UnsupportedFileFormatError({
16
+ filename: input.filename,
17
+ mediaType: input.mediaType
18
+ }));
19
+ yield* Effect.annotateCurrentSpan({
20
+ "file_extractor.format": format,
21
+ "file_extractor.file_size": input.bytes.byteLength
22
+ });
23
+ if (input.bytes.byteLength > limits.maxInputBytes) return yield* Effect.fail(new FileExtractionError({
24
+ message: "File exceeds the extraction size limit",
25
+ format
26
+ }));
27
+ if (isParsedFileFormat(format)) return yield* extractParsed(input, format, limits);
28
+ return yield* makeExtractedFile(decodeText(input.bytes), {
29
+ format,
30
+ title: input.filename
31
+ });
32
+ }).pipe(Effect.withSpan("FileExtractor.extract")) });
33
+ const parsedFileExtractor = (options) => {
34
+ const isolation = options.isolation ?? "worker";
35
+ if (isolation === "none") {
36
+ const loader = options.loadSheetJs ?? defaultSheetJsLoader;
37
+ return Effect.succeed((input, format, limits) => extractParsedFile(input, format, limits, loader));
38
+ }
39
+ if (options.loadSheetJs !== void 0) return Effect.die(/* @__PURE__ */ new Error("FileExtractorOptions.loadSheetJs runs SheetJS in the calling thread and needs isolation: \"none\""));
40
+ const overrides = isolation === "worker" ? {} : isolation;
41
+ const timeoutMs = overrides.timeoutMs ?? defaultWorkerIsolation.timeoutMs;
42
+ return Schema.decodeUnknownEffect(WorkerIsolationSettings)({
43
+ maxOldGenerationSizeMb: overrides.maxOldGenerationSizeMb ?? defaultWorkerIsolation.maxOldGenerationSizeMb,
44
+ maxYoungGenerationSizeMb: overrides.maxYoungGenerationSizeMb ?? defaultWorkerIsolation.maxYoungGenerationSizeMb,
45
+ stackSizeMb: overrides.stackSizeMb ?? defaultWorkerIsolation.stackSizeMb,
46
+ timeoutMs,
47
+ maxConcurrentWorkers: overrides.maxConcurrentWorkers ?? defaultWorkerIsolation.maxConcurrentWorkers,
48
+ maxQueueWaitMs: overrides.maxQueueWaitMs ?? timeoutMs
49
+ }).pipe(Effect.flatMap((settings) => makeWorkerExtractor(settings, overrides.workerUrl ?? defaultExtractionWorkerUrl())));
50
+ };
51
+ /**
52
+ * Node `FileExtractor` layer with custom limits, isolation, or (in-process) SheetJS loader.
53
+ * Invalid limits or isolation settings are defects.
54
+ */
55
+ const makeFileExtractorLayer = (options = {}) => Layer.effect(FileExtractor, Effect.gen(function* () {
56
+ const limits = yield* Schema.decodeUnknownEffect(FileExtractorLimits)({
57
+ ...defaultFileExtractorLimits,
58
+ ...options.limits
59
+ });
60
+ const extractParsed = yield* parsedFileExtractor(options);
61
+ return FileExtractor.of(makeFileExtractor(limits, extractParsed));
62
+ }).pipe(Effect.orDie));
63
+ /**
64
+ * Node `FileExtractor` layer with the default limits, worker isolation, and lazy SheetJS loading.
65
+ */
66
+ const FileExtractorLayer = makeFileExtractorLayer();
67
+ //#endregion
68
+ export { FileExtractorLayer, makeFileExtractorLayer };
69
+
70
+ //# sourceMappingURL=live-layer.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"live-layer.mjs","names":[],"sources":["../../src/node/live-layer.ts"],"sourcesContent":["import { Effect, Layer } from 'effect'\nimport * as Schema from 'effect/Schema'\nimport { FileExtractionError, UnsupportedFileFormatError } from '../errors.ts'\nimport { fileFormatFor } from '../format.ts'\nimport { defaultFileExtractorLimits, FileExtractorLimits } from '../limits.ts'\nimport { FileExtractor } from '../service.ts'\nimport type { FileExtractorApi } from '../service.ts'\nimport { extractParsedFile, isParsedFileFormat, makeExtractedFile } from './extract-file.ts'\nimport type { ParsedFileFormat } from './extract-file.ts'\nimport {\n defaultExtractionWorkerUrl,\n defaultWorkerIsolation,\n makeWorkerExtractor,\n WorkerIsolationSettings\n} from './extraction-isolation.ts'\nimport type { FileExtractorIsolation, WorkerExtractor } from './extraction-isolation.ts'\nimport { defaultSheetJsLoader } from './sheetjs.ts'\nimport type { SheetJsLoader } from './sheetjs.ts'\n\nexport type FileExtractorOptions = {\n /** Override any default limit; see `defaultFileExtractorLimits`. */\n readonly limits?: Partial<FileExtractorLimits>\n /**\n * Where parsers run (default `'worker'`): each PDF, DOCX, XLSX, and PPTX extraction runs in a\n * fresh worker thread with V8 heap, stack, and time limits, admitted through a pool of 4 workers\n * per JavaScript realm; pass an object to change them. `'none'` parses in the calling thread and is\n * unsafe for untrusted input.\n */\n readonly isolation?: FileExtractorIsolation\n /**\n * Load SheetJS in the calling thread. Only with `isolation: 'none'` (a worker cannot receive a\n * function and always imports the installed `xlsx` itself); any other isolation is a defect\n * when the layer is built. Defaults to a lazy `import('xlsx')`. Missing, non-SheetJS, or\n * pre-0.20.3 modules fail with `SheetJsUnavailableError`.\n */\n readonly loadSheetJs?: SheetJsLoader\n}\n\nconst decodeText = (bytes: Uint8Array) => new TextDecoder('utf-8', { fatal: false }).decode(bytes)\n\ntype ParsedFileExtractor = WorkerExtractor\n\n/** Build the Node `FileExtractor` from validated limits and the parser runner. */\nconst makeFileExtractor = (\n limits: FileExtractorLimits,\n extractParsed: ParsedFileExtractor\n): FileExtractorApi => ({\n extract: input =>\n Effect.gen(function* () {\n const format = fileFormatFor(input)\n\n if (format === undefined)\n return yield* Effect.fail(\n new UnsupportedFileFormatError({ filename: input.filename, mediaType: input.mediaType })\n )\n\n yield* Effect.annotateCurrentSpan({\n 'file_extractor.format': format,\n 'file_extractor.file_size': input.bytes.byteLength\n })\n\n if (input.bytes.byteLength > limits.maxInputBytes)\n return yield* Effect.fail(\n new FileExtractionError({ message: 'File exceeds the extraction size limit', format })\n )\n\n if (isParsedFileFormat(format)) return yield* extractParsed(input, format, limits)\n\n // Text formats are only decoded and sanitized; no parser reads them.\n return yield* makeExtractedFile(decodeText(input.bytes), { format, title: input.filename })\n }).pipe(Effect.withSpan('FileExtractor.extract'))\n})\n\nconst parsedFileExtractor = (\n options: FileExtractorOptions\n): Effect.Effect<ParsedFileExtractor, Schema.SchemaError> => {\n const isolation = options.isolation ?? 'worker'\n\n if (isolation === 'none') {\n const loader = options.loadSheetJs ?? defaultSheetJsLoader\n\n return Effect.succeed((input, format: ParsedFileFormat, limits) =>\n extractParsedFile(input, format, limits, loader)\n )\n }\n\n if (options.loadSheetJs !== undefined)\n return Effect.die(\n new Error(\n 'FileExtractorOptions.loadSheetJs runs SheetJS in the calling thread and needs isolation: \"none\"'\n )\n )\n\n const overrides = isolation === 'worker' ? {} : isolation\n const timeoutMs = overrides.timeoutMs ?? defaultWorkerIsolation.timeoutMs\n\n return Schema.decodeUnknownEffect(WorkerIsolationSettings)({\n maxOldGenerationSizeMb:\n overrides.maxOldGenerationSizeMb ?? defaultWorkerIsolation.maxOldGenerationSizeMb,\n maxYoungGenerationSizeMb:\n overrides.maxYoungGenerationSizeMb ?? defaultWorkerIsolation.maxYoungGenerationSizeMb,\n stackSizeMb: overrides.stackSizeMb ?? defaultWorkerIsolation.stackSizeMb,\n timeoutMs,\n maxConcurrentWorkers:\n overrides.maxConcurrentWorkers ?? defaultWorkerIsolation.maxConcurrentWorkers,\n maxQueueWaitMs: overrides.maxQueueWaitMs ?? timeoutMs\n }).pipe(\n Effect.flatMap(settings =>\n makeWorkerExtractor(settings, overrides.workerUrl ?? defaultExtractionWorkerUrl())\n )\n )\n}\n\n/**\n * Node `FileExtractor` layer with custom limits, isolation, or (in-process) SheetJS loader.\n * Invalid limits or isolation settings are defects.\n */\nexport const makeFileExtractorLayer = (options: FileExtractorOptions = {}) =>\n Layer.effect(\n FileExtractor,\n Effect.gen(function* () {\n const limits = yield* Schema.decodeUnknownEffect(FileExtractorLimits)({\n ...defaultFileExtractorLimits,\n ...options.limits\n })\n\n const extractParsed = yield* parsedFileExtractor(options)\n\n return FileExtractor.of(makeFileExtractor(limits, extractParsed))\n }).pipe(Effect.orDie)\n )\n\n/**\n * Node `FileExtractor` layer with the default limits, worker isolation, and lazy SheetJS loading.\n */\nexport const FileExtractorLayer = makeFileExtractorLayer()\n"],"mappings":";;;;;;;;;;AAsCA,MAAM,cAAc,UAAsB,IAAI,YAAY,SAAS,EAAE,OAAO,MAAM,CAAC,EAAE,OAAO,KAAK;;AAKjG,MAAM,qBACJ,QACA,mBACsB,EACtB,UAAS,UACP,OAAO,IAAI,aAAa;CACtB,MAAM,SAAS,cAAc,KAAK;CAElC,IAAI,WAAW,KAAA,GACb,OAAO,OAAO,OAAO,KACnB,IAAI,2BAA2B;EAAE,UAAU,MAAM;EAAU,WAAW,MAAM;CAAU,CAAC,CACzF;CAEF,OAAO,OAAO,oBAAoB;EAChC,yBAAyB;EACzB,4BAA4B,MAAM,MAAM;CAC1C,CAAC;CAED,IAAI,MAAM,MAAM,aAAa,OAAO,eAClC,OAAO,OAAO,OAAO,KACnB,IAAI,oBAAoB;EAAE,SAAS;EAA0C;CAAO,CAAC,CACvF;CAEF,IAAI,mBAAmB,MAAM,GAAG,OAAO,OAAO,cAAc,OAAO,QAAQ,MAAM;CAGjF,OAAO,OAAO,kBAAkB,WAAW,MAAM,KAAK,GAAG;EAAE;EAAQ,OAAO,MAAM;CAAS,CAAC;AAC5F,CAAC,EAAE,KAAK,OAAO,SAAS,uBAAuB,CAAC,EACpD;AAEA,MAAM,uBACJ,YAC2D;CAC3D,MAAM,YAAY,QAAQ,aAAa;CAEvC,IAAI,cAAc,QAAQ;EACxB,MAAM,SAAS,QAAQ,eAAe;EAEtC,OAAO,OAAO,SAAS,OAAO,QAA0B,WACtD,kBAAkB,OAAO,QAAQ,QAAQ,MAAM,CACjD;CACF;CAEA,IAAI,QAAQ,gBAAgB,KAAA,GAC1B,OAAO,OAAO,oBACZ,IAAI,MACF,mGACF,CACF;CAEF,MAAM,YAAY,cAAc,WAAW,CAAC,IAAI;CAChD,MAAM,YAAY,UAAU,aAAa,uBAAuB;CAEhE,OAAO,OAAO,oBAAoB,uBAAuB,EAAE;EACzD,wBACE,UAAU,0BAA0B,uBAAuB;EAC7D,0BACE,UAAU,4BAA4B,uBAAuB;EAC/D,aAAa,UAAU,eAAe,uBAAuB;EAC7D;EACA,sBACE,UAAU,wBAAwB,uBAAuB;EAC3D,gBAAgB,UAAU,kBAAkB;CAC9C,CAAC,EAAE,KACD,OAAO,SAAQ,aACb,oBAAoB,UAAU,UAAU,aAAa,2BAA2B,CAAC,CACnF,CACF;AACF;;;;;AAMA,MAAa,0BAA0B,UAAgC,CAAC,MACtE,MAAM,OACJ,eACA,OAAO,IAAI,aAAa;CACtB,MAAM,SAAS,OAAO,OAAO,oBAAoB,mBAAmB,EAAE;EACpE,GAAG;EACH,GAAG,QAAQ;CACb,CAAC;CAED,MAAM,gBAAgB,OAAO,oBAAoB,OAAO;CAExD,OAAO,cAAc,GAAG,kBAAkB,QAAQ,aAAa,CAAC;AAClE,CAAC,EAAE,KAAK,OAAO,KAAK,CACtB;;;;AAKF,MAAa,qBAAqB,uBAAuB"}
@@ -0,0 +1,51 @@
1
+ import { OfficeArchiveError } from "../errors.mjs";
2
+ import { OfficeFileFormat } from "../format.mjs";
3
+ import { FileExtractorLimits } from "../limits.mjs";
4
+ import { Effect } from "effect";
5
+
6
+ //#region src/node/office-archive.d.ts
7
+ type OfficeArchiveLimits = Pick<FileExtractorLimits, 'maxArchiveEntries' | 'maxExpandedBytes' | 'maxInputBytes'>;
8
+ /** Inflated output is counted in slices of at most this many bytes. */
9
+ declare const officeInflateChunkBytes: number;
10
+ /**
11
+ * Whether a part SheetJS would decode as BOM-marked UTF-16 contains a hyperlink tag. Covers
12
+ * SheetJS `cc2str` UTF-16 BOM decoding (little- and big-endian from byte 2, including its
13
+ * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
14
+ */
15
+ declare const utf16PartHasHyperlink: (content: Uint8Array) => boolean;
16
+ /** Hyperlink start tags (Latin-1 text of the original bytes) found while stripping, per part. */
17
+ type StrippedHyperlinkTags = ReadonlyMap<string, ReadonlyArray<string>>;
18
+ type NormalizedOfficeArchive = {
19
+ /** Every validated (and, for XLSX, hyperlink-stripped) part by entry name. */readonly parts: Readonly<Record<string, Uint8Array>>; /** XLSX only: removed hyperlink tags, at most `maxHyperlinkTags` across the workbook. */
20
+ readonly hyperlinkTags: StrippedHyperlinkTags;
21
+ };
22
+ type ReadOfficeArchiveOptions = {
23
+ /** XLSX: hyperlink start tags to capture across the workbook (default 0). */readonly maxHyperlinkTags?: number;
24
+ };
25
+ /** A fresh stored-entry ZIP of validated parts; the attacker's ZIP index is never reused. */
26
+ declare const storedArchive: (parts: Readonly<Record<string, Uint8Array>>) => Uint8Array<ArrayBuffer>;
27
+ /**
28
+ * Inflate bounded input chunks, count actual output, and return the validated parts. The
29
+ * attacker's ZIP index is discarded: parsers only get archives rebuilt from these parts.
30
+ *
31
+ * For XLSX, every part loses its `<hyperlink>` tags: SheetJS expands each hyperlink range into
32
+ * per-cell objects before any budget runs, so one `ref="A1:XFD1048576"` exhausts memory. The
33
+ * removed tags are returned so links can still be shown. XLSX input that SheetJS would route to
34
+ * its binary (XLSB), ODS, or Numbers parsers is rejected early with a clear error; the guarantee
35
+ * is that SheetJS only receives `buildSheetJsInput`'s allowlisted archive.
36
+ */
37
+ declare const readOfficeArchive: (input: Uint8Array, format: OfficeFileFormat, limits: OfficeArchiveLimits, options?: ReadOfficeArchiveOptions) => Effect.Effect<NormalizedOfficeArchive, OfficeArchiveError, never>;
38
+ /**
39
+ * Validate a DOCX, XLSX, or PPTX archive with bounded inflation and return a rebuilt stored-entry
40
+ * ZIP of every validated part, for storage or for other parsers. XLSX parts lose their hyperlink
41
+ * tags, and XLSX input with ODS or Numbers marker entries or XLSB parts is rejected; `.bin` parts
42
+ * SheetJS never parses (printer settings, OLE objects) are kept so stored files still open.
43
+ *
44
+ * The output is not SheetJS input. Never run SheetJS on it directly: extract XLSX text through
45
+ * `FileExtractor`, which hands SheetJS only an allowlisted archive it builds itself (worksheet,
46
+ * shared-string, style, and core-property parts with generated content types and relationships).
47
+ */
48
+ declare const normalizeOfficeArchive: (bytes: Uint8Array, format: OfficeFileFormat, limits?: Partial<OfficeArchiveLimits>) => Effect.Effect<Uint8Array<ArrayBuffer>, OfficeArchiveError, never>;
49
+ //#endregion
50
+ export { NormalizedOfficeArchive, OfficeArchiveLimits, ReadOfficeArchiveOptions, StrippedHyperlinkTags, normalizeOfficeArchive, officeInflateChunkBytes, readOfficeArchive, storedArchive, utf16PartHasHyperlink };
51
+ //# sourceMappingURL=office-archive.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"office-archive.d.mts","names":[],"sources":["../../src/node/office-archive.ts"],"mappings":";;;;;;KAgBY,mBAAA,GAAsB,IAAI,CACpC,mBAAA;;cAgBW,uBAAA;;;;AAhBQ;AAgBrB;cA2Ia,qBAAA,GAAyB,OAAmB,EAAV,UAAU;;KAW7C,qBAAA,GAAwB,WAAW,SAAS,aAAA;AAAA,KAE5C,uBAAA;EAbC,uFAeF,KAAA,EAAO,QAAA,CAAS,MAAA,SAAe,UAAA;WAE/B,aAAA,EAAe,qBAAA;AAAA;AAAA,KAwCd,wBAAA;EA9CqB,sFAgDtB,gBAAgB;AAAA;AAhD0C;AAAA,cAoDxD,aAAA,GAAiB,KAAA,EAAO,QAAA,CAAS,MAAA,SAAe,UAAA,OAAY,UAAA,CAAA,WAAA;;;;;;;;;;;cAa5D,iBAAA,GACX,KAAA,EAAO,UAAA,EACP,MAAA,EAAQ,gBAAA,EACR,MAAA,EAAQ,mBAAA,EACR,OAAA,GAAS,wBAAA,KAA6B,MAAA,CAAA,MAAA,CAAA,uBAAA,EAAA,kBAAA;;;;;AA/DO;AAwC/C;;;;AAE2B;cA+Hd,sBAAA,GACX,KAAA,EAAO,UAAA,EACP,MAAA,EAAQ,gBAAA,EACR,MAAA,GAAQ,OAAA,CAAQ,mBAAA,MAAyB,MAAA,CAAA,MAAA,CAAA,UAAA,CAAA,WAAA,GAAA,kBAAA"}
@@ -0,0 +1,193 @@
1
+ import { OfficeArchiveError } from "../errors.mjs";
2
+ import { defaultFileExtractorLimits } from "../limits.mjs";
3
+ import { sheetJsTextViews } from "./sheetjs-xml.mjs";
4
+ import { contentTypesRouteToBinary, isAlternateFormatEntry, relationshipsRouteToBinary } from "./xlsx-routing.mjs";
5
+ import { Effect } from "effect";
6
+ import { Buffer } from "node:buffer";
7
+ import { Readable } from "node:stream";
8
+ import { createInflateRaw } from "node:zlib";
9
+ import { zipSync } from "fflate";
10
+ //#region src/node/office-archive.ts
11
+ const invalid = () => new OfficeArchiveError({ message: "Invalid Office archive." });
12
+ const tooLargeMessage = "Office archive expansion exceeds its declared size or limit.";
13
+ const tooLarge = (expandedBytes) => expandedBytes === void 0 ? new OfficeArchiveError({ message: tooLargeMessage }) : new OfficeArchiveError({
14
+ message: tooLargeMessage,
15
+ expandedBytes
16
+ });
17
+ const compressedChunkBytes = 1024;
18
+ /** Inflated output is counted in slices of at most this many bytes. */
19
+ const officeInflateChunkBytes = 16 * 1024;
20
+ const mainParts = {
21
+ docx: "word/document.xml",
22
+ pptx: "ppt/presentation.xml",
23
+ xlsx: "xl/workbook.xml"
24
+ };
25
+ /** Read the directory only as a bounded index. Its sizes are never trusted for allocation. */
26
+ const archiveEntries = (bytes, limits) => {
27
+ let end = bytes.length - 22;
28
+ const earliest = Math.max(0, end - 65535);
29
+ while (end >= earliest && bytes.readUInt32LE(end) !== 101010256) end -= 1;
30
+ if (end < earliest || end + 22 + bytes.readUInt16LE(end + 20) !== bytes.length) throw invalid();
31
+ const count = bytes.readUInt16LE(end + 10);
32
+ const directorySize = bytes.readUInt32LE(end + 12);
33
+ const directoryOffset = bytes.readUInt32LE(end + 16);
34
+ if (bytes.readUInt32LE(end + 4) !== 0 || bytes.readUInt16LE(end + 8) !== count || count === 0 || count > limits.maxArchiveEntries || directoryOffset + directorySize !== end) throw invalid();
35
+ const entries = [];
36
+ const names = /* @__PURE__ */ new Set();
37
+ let offset = directoryOffset;
38
+ let declaredTotal = 0;
39
+ for (let index = 0; index < count; index += 1) {
40
+ if (offset + 46 > end || bytes.readUInt32LE(offset) !== 33639248) throw invalid();
41
+ const flags = bytes.readUInt16LE(offset + 8);
42
+ const method = bytes.readUInt16LE(offset + 10);
43
+ const compressedSize = bytes.readUInt32LE(offset + 20);
44
+ const originalSize = bytes.readUInt32LE(offset + 24);
45
+ const nameSize = bytes.readUInt16LE(offset + 28);
46
+ const extraSize = bytes.readUInt16LE(offset + 30);
47
+ const commentSize = bytes.readUInt16LE(offset + 32);
48
+ const local = bytes.readUInt32LE(offset + 42);
49
+ const next = offset + 46 + nameSize + extraSize + commentSize;
50
+ if (next > end || nameSize === 0 || nameSize > 1024 || (flags & -2063) !== 0 || method !== 0 && method !== 8 || bytes.readUInt16LE(offset + 34) !== 0 || compressedSize === 4294967295 || originalSize === 4294967295 || local + 30 > directoryOffset) throw invalid();
51
+ const nameBytes = bytes.subarray(offset + 46, offset + 46 + nameSize);
52
+ const name = new TextDecoder("utf-8", { fatal: true }).decode(nameBytes);
53
+ if (/[\\\u0000-\u001f]/.test(name) || name.startsWith("/") || name.includes("//") || name.split("/").some((part) => part === ".." || part === ".") || names.has(name.toLowerCase()) || /vbaProject\.bin$/i.test(name)) throw invalid();
54
+ names.add(name.toLowerCase());
55
+ if (bytes.readUInt32LE(local) !== 67324752 || bytes.readUInt16LE(local + 6) !== flags || bytes.readUInt16LE(local + 8) !== method || bytes.readUInt16LE(local + 26) !== nameSize) throw invalid();
56
+ const start = local + 30 + nameSize + bytes.readUInt16LE(local + 28);
57
+ if (start + compressedSize > directoryOffset || !bytes.subarray(local + 30, local + 30 + nameSize).equals(nameBytes)) throw invalid();
58
+ for (const [position, expected] of [[18, compressedSize], [22, originalSize]]) {
59
+ const value = bytes.readUInt32LE(local + position);
60
+ if (value !== expected && !((flags & 8) !== 0 && value === 0)) throw invalid();
61
+ }
62
+ declaredTotal += originalSize;
63
+ if (declaredTotal > limits.maxExpandedBytes) throw tooLarge();
64
+ entries.push({
65
+ name,
66
+ start,
67
+ compressedSize,
68
+ originalSize,
69
+ method
70
+ });
71
+ offset = next;
72
+ }
73
+ if (offset !== end) throw invalid();
74
+ return entries;
75
+ };
76
+ /**
77
+ * A superset of SheetJS's `hlinkregex` (`/<(?:\w+:)?hyperlink [^<>]*>/`). A candidate never spans
78
+ * a `<`, so each scan stops at the next tag and the strip stays linear in the part size.
79
+ */
80
+ const hyperlinkTag = /<\/?(?:[\w.-]+:)?hyperlink\b[^<>]*>/gi;
81
+ const hyperlinkTagStart = /<\/?(?:[\w.-]+:)?hyperlink\b/i;
82
+ /**
83
+ * Whether a part SheetJS would decode as BOM-marked UTF-16 contains a hyperlink tag. Covers
84
+ * SheetJS `cc2str` UTF-16 BOM decoding (little- and big-endian from byte 2, including its
85
+ * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
86
+ */
87
+ const utf16PartHasHyperlink = (content) => sheetJsTextViews(content).slice(1).some((text) => hyperlinkTagStart.test(text));
88
+ const unsupportedXlsxParts = () => new OfficeArchiveError({ message: "XLSX archive contains binary (XLSB), ODS, or Numbers parts." });
89
+ const inflateEntry = async (compressed, record) => {
90
+ function* inputChunks() {
91
+ for (let offset = 0; offset < compressed.length; offset += compressedChunkBytes) yield compressed.subarray(offset, offset + compressedChunkBytes);
92
+ }
93
+ const source = Readable.from(inputChunks(), { highWaterMark: 1 });
94
+ const inflater = createInflateRaw({
95
+ chunkSize: officeInflateChunkBytes,
96
+ readableHighWaterMark: officeInflateChunkBytes,
97
+ writableHighWaterMark: compressedChunkBytes
98
+ });
99
+ source.pipe(inflater);
100
+ try {
101
+ for await (const value of inflater) {
102
+ if (!Buffer.isBuffer(value)) throw invalid();
103
+ for (let offset = 0; offset < value.length; offset += officeInflateChunkBytes) record(value.subarray(offset, offset + officeInflateChunkBytes));
104
+ }
105
+ if (inflater.bytesWritten !== compressed.length) throw invalid();
106
+ } finally {
107
+ source.destroy();
108
+ inflater.destroy();
109
+ }
110
+ };
111
+ /** A fresh stored-entry ZIP of validated parts; the attacker's ZIP index is never reused. */
112
+ const storedArchive = (parts) => zipSync({ ...parts }, { level: 0 });
113
+ /**
114
+ * Inflate bounded input chunks, count actual output, and return the validated parts. The
115
+ * attacker's ZIP index is discarded: parsers only get archives rebuilt from these parts.
116
+ *
117
+ * For XLSX, every part loses its `<hyperlink>` tags: SheetJS expands each hyperlink range into
118
+ * per-cell objects before any budget runs, so one `ref="A1:XFD1048576"` exhausts memory. The
119
+ * removed tags are returned so links can still be shown. XLSX input that SheetJS would route to
120
+ * its binary (XLSB), ODS, or Numbers parsers is rejected early with a clear error; the guarantee
121
+ * is that SheetJS only receives `buildSheetJsInput`'s allowlisted archive.
122
+ */
123
+ const readOfficeArchive = (input, format, limits, options = {}) => Effect.tryPromise({
124
+ try: async () => {
125
+ const bytes = Buffer.from(input.buffer, input.byteOffset, input.byteLength);
126
+ if (bytes.length < 22 || bytes.length > limits.maxInputBytes) throw invalid();
127
+ const entries = archiveEntries(bytes, limits);
128
+ const mainPart = mainParts[format];
129
+ const maxHyperlinkTags = options.maxHyperlinkTags ?? 0;
130
+ if (format === "xlsx" && entries.some((entry) => isAlternateFormatEntry(entry.name))) throw unsupportedXlsxParts();
131
+ if (!entries.some((entry) => entry.name === "[Content_Types].xml") || !entries.some((entry) => entry.name === mainPart)) throw invalid();
132
+ const validated = Object.create(null);
133
+ const hyperlinkTags = /* @__PURE__ */ new Map();
134
+ let capturedTags = 0;
135
+ let expandedBytes = 0;
136
+ for (const entry of entries) {
137
+ const chunks = [];
138
+ let entryBytes = 0;
139
+ const record = (chunk) => {
140
+ entryBytes += chunk.length;
141
+ expandedBytes += chunk.length;
142
+ if (expandedBytes > limits.maxExpandedBytes || entryBytes > entry.originalSize) throw tooLarge(expandedBytes);
143
+ chunks.push(chunk);
144
+ };
145
+ const compressed = bytes.subarray(entry.start, entry.start + entry.compressedSize);
146
+ if (entry.method === 0) record(compressed);
147
+ else await inflateEntry(compressed, record);
148
+ if (entryBytes !== entry.originalSize) throw invalid();
149
+ const content = Buffer.concat(chunks, entryBytes);
150
+ if (format !== "xlsx") {
151
+ validated[entry.name] = content;
152
+ continue;
153
+ }
154
+ const tags = [];
155
+ const rewritten = Buffer.from(content.toString("latin1").replace(hyperlinkTag, (tag) => {
156
+ if (!tag.startsWith("</") && capturedTags < maxHyperlinkTags) {
157
+ capturedTags += 1;
158
+ tags.push(tag);
159
+ }
160
+ return " ";
161
+ }), "latin1");
162
+ if (utf16PartHasHyperlink(rewritten)) throw invalid();
163
+ const lowerName = entry.name.toLowerCase();
164
+ if (lowerName === "[content_types].xml" && contentTypesRouteToBinary(rewritten) || lowerName.endsWith(".rels") && relationshipsRouteToBinary(rewritten)) throw unsupportedXlsxParts();
165
+ if (tags.length > 0) hyperlinkTags.set(entry.name, tags);
166
+ validated[entry.name] = rewritten;
167
+ }
168
+ return {
169
+ parts: validated,
170
+ hyperlinkTags
171
+ };
172
+ },
173
+ catch: (error) => error instanceof OfficeArchiveError ? error : invalid()
174
+ });
175
+ /**
176
+ * Validate a DOCX, XLSX, or PPTX archive with bounded inflation and return a rebuilt stored-entry
177
+ * ZIP of every validated part, for storage or for other parsers. XLSX parts lose their hyperlink
178
+ * tags, and XLSX input with ODS or Numbers marker entries or XLSB parts is rejected; `.bin` parts
179
+ * SheetJS never parses (printer settings, OLE objects) are kept so stored files still open.
180
+ *
181
+ * The output is not SheetJS input. Never run SheetJS on it directly: extract XLSX text through
182
+ * `FileExtractor`, which hands SheetJS only an allowlisted archive it builds itself (worksheet,
183
+ * shared-string, style, and core-property parts with generated content types and relationships).
184
+ */
185
+ const normalizeOfficeArchive = (bytes, format, limits = {}) => readOfficeArchive(bytes, format, {
186
+ maxArchiveEntries: limits.maxArchiveEntries ?? defaultFileExtractorLimits.maxArchiveEntries,
187
+ maxExpandedBytes: limits.maxExpandedBytes ?? defaultFileExtractorLimits.maxExpandedBytes,
188
+ maxInputBytes: limits.maxInputBytes ?? defaultFileExtractorLimits.maxInputBytes
189
+ }).pipe(Effect.map((normalized) => storedArchive(normalized.parts)));
190
+ //#endregion
191
+ export { normalizeOfficeArchive, officeInflateChunkBytes, readOfficeArchive, storedArchive, utf16PartHasHyperlink };
192
+
193
+ //# sourceMappingURL=office-archive.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"office-archive.mjs","names":[],"sources":["../../src/node/office-archive.ts"],"sourcesContent":["import { Buffer } from 'node:buffer'\nimport { Readable } from 'node:stream'\nimport { createInflateRaw } from 'node:zlib'\nimport { Effect } from 'effect'\nimport { zipSync } from 'fflate'\nimport { OfficeArchiveError } from '../errors.ts'\nimport type { OfficeFileFormat } from '../format.ts'\nimport { defaultFileExtractorLimits } from '../limits.ts'\nimport type { FileExtractorLimits } from '../limits.ts'\nimport { sheetJsTextViews } from './sheetjs-xml.ts'\nimport {\n contentTypesRouteToBinary,\n isAlternateFormatEntry,\n relationshipsRouteToBinary\n} from './xlsx-routing.ts'\n\nexport type OfficeArchiveLimits = Pick<\n FileExtractorLimits,\n 'maxArchiveEntries' | 'maxExpandedBytes' | 'maxInputBytes'\n>\n\nconst invalid = () => new OfficeArchiveError({ message: 'Invalid Office archive.' })\n\nconst tooLargeMessage = 'Office archive expansion exceeds its declared size or limit.'\n\nconst tooLarge = (expandedBytes?: number) =>\n expandedBytes === undefined\n ? new OfficeArchiveError({ message: tooLargeMessage })\n : new OfficeArchiveError({ message: tooLargeMessage, expandedBytes })\n\nconst compressedChunkBytes = 1024\n\n/** Inflated output is counted in slices of at most this many bytes. */\nexport const officeInflateChunkBytes = 16 * 1024\n\nconst mainParts: Readonly<Record<OfficeFileFormat, string>> = {\n docx: 'word/document.xml',\n pptx: 'ppt/presentation.xml',\n xlsx: 'xl/workbook.xml'\n}\n\ntype ArchiveEntry = {\n readonly name: string\n readonly start: number\n readonly compressedSize: number\n readonly originalSize: number\n readonly method: number\n}\n\n/** Read the directory only as a bounded index. Its sizes are never trusted for allocation. */\nconst archiveEntries = (bytes: Buffer, limits: OfficeArchiveLimits) => {\n let end = bytes.length - 22\n const earliest = Math.max(0, end - 65535)\n\n while (end >= earliest && bytes.readUInt32LE(end) !== 0x06054b50) end -= 1\n\n if (end < earliest || end + 22 + bytes.readUInt16LE(end + 20) !== bytes.length) throw invalid()\n\n const count = bytes.readUInt16LE(end + 10)\n const directorySize = bytes.readUInt32LE(end + 12)\n const directoryOffset = bytes.readUInt32LE(end + 16)\n\n if (\n bytes.readUInt32LE(end + 4) !== 0 ||\n bytes.readUInt16LE(end + 8) !== count ||\n count === 0 ||\n count > limits.maxArchiveEntries ||\n directoryOffset + directorySize !== end\n )\n throw invalid()\n\n const entries: Array<ArchiveEntry> = []\n // OPC part names are case-insensitive, and SheetJS looks entries up ignoring case.\n const names = new Set<string>()\n let offset = directoryOffset\n let declaredTotal = 0\n\n for (let index = 0; index < count; index += 1) {\n if (offset + 46 > end || bytes.readUInt32LE(offset) !== 0x02014b50) throw invalid()\n\n const flags = bytes.readUInt16LE(offset + 8)\n const method = bytes.readUInt16LE(offset + 10)\n const compressedSize = bytes.readUInt32LE(offset + 20)\n const originalSize = bytes.readUInt32LE(offset + 24)\n const nameSize = bytes.readUInt16LE(offset + 28)\n const extraSize = bytes.readUInt16LE(offset + 30)\n const commentSize = bytes.readUInt16LE(offset + 32)\n const local = bytes.readUInt32LE(offset + 42)\n const next = offset + 46 + nameSize + extraSize + commentSize\n\n // Reject encryption, unsupported methods, split/ZIP64 archives, and ambiguous paths.\n if (\n next > end ||\n nameSize === 0 ||\n nameSize > 1024 ||\n (flags & ~0x080e) !== 0 ||\n (method !== 0 && method !== 8) ||\n bytes.readUInt16LE(offset + 34) !== 0 ||\n compressedSize === 0xffffffff ||\n originalSize === 0xffffffff ||\n local + 30 > directoryOffset\n )\n throw invalid()\n\n const nameBytes = bytes.subarray(offset + 46, offset + 46 + nameSize)\n const name = new TextDecoder('utf-8', { fatal: true }).decode(nameBytes)\n\n // `//` is rejected (SheetJS collapses the first one), but directory entries ending in `/` stay.\n if (\n /[\\\\\\u0000-\\u001f]/.test(name) ||\n name.startsWith('/') ||\n name.includes('//') ||\n name.split('/').some(part => part === '..' || part === '.') ||\n names.has(name.toLowerCase()) ||\n /vbaProject\\.bin$/i.test(name)\n )\n throw invalid()\n\n names.add(name.toLowerCase())\n\n if (\n bytes.readUInt32LE(local) !== 0x04034b50 ||\n bytes.readUInt16LE(local + 6) !== flags ||\n bytes.readUInt16LE(local + 8) !== method ||\n bytes.readUInt16LE(local + 26) !== nameSize\n )\n throw invalid()\n\n const start = local + 30 + nameSize + bytes.readUInt16LE(local + 28)\n\n if (\n start + compressedSize > directoryOffset ||\n !bytes.subarray(local + 30, local + 30 + nameSize).equals(nameBytes)\n )\n throw invalid()\n\n // Data descriptors may leave local sizes zero; all nonzero declarations must agree.\n for (const [position, expected] of [\n [18, compressedSize],\n [22, originalSize]\n ] as const) {\n const value = bytes.readUInt32LE(local + position)\n\n if (value !== expected && !((flags & 8) !== 0 && value === 0)) throw invalid()\n }\n\n declaredTotal += originalSize\n\n if (declaredTotal > limits.maxExpandedBytes) throw tooLarge()\n\n entries.push({ name, start, compressedSize, originalSize, method })\n offset = next\n }\n\n if (offset !== end) throw invalid()\n\n return entries\n}\n\n/**\n * A superset of SheetJS's `hlinkregex` (`/<(?:\\w+:)?hyperlink [^<>]*>/`). A candidate never spans\n * a `<`, so each scan stops at the next tag and the strip stays linear in the part size.\n */\nconst hyperlinkTag = /<\\/?(?:[\\w.-]+:)?hyperlink\\b[^<>]*>/gi\n\nconst hyperlinkTagStart = /<\\/?(?:[\\w.-]+:)?hyperlink\\b/i\n\n/**\n * Whether a part SheetJS would decode as BOM-marked UTF-16 contains a hyperlink tag. Covers\n * SheetJS `cc2str` UTF-16 BOM decoding (little- and big-endian from byte 2, including its\n * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.\n */\nexport const utf16PartHasHyperlink = (content: Uint8Array) =>\n sheetJsTextViews(content)\n .slice(1)\n .some(text => hyperlinkTagStart.test(text))\n\nconst unsupportedXlsxParts = () =>\n new OfficeArchiveError({\n message: 'XLSX archive contains binary (XLSB), ODS, or Numbers parts.'\n })\n\n/** Hyperlink start tags (Latin-1 text of the original bytes) found while stripping, per part. */\nexport type StrippedHyperlinkTags = ReadonlyMap<string, ReadonlyArray<string>>\n\nexport type NormalizedOfficeArchive = {\n /** Every validated (and, for XLSX, hyperlink-stripped) part by entry name. */\n readonly parts: Readonly<Record<string, Uint8Array>>\n /** XLSX only: removed hyperlink tags, at most `maxHyperlinkTags` across the workbook. */\n readonly hyperlinkTags: StrippedHyperlinkTags\n}\n\nconst inflateEntry = async (compressed: Buffer, record: (chunk: Buffer) => void): Promise<void> => {\n function* inputChunks() {\n for (let offset = 0; offset < compressed.length; offset += compressedChunkBytes) {\n yield compressed.subarray(offset, offset + compressedChunkBytes)\n }\n }\n\n const source = Readable.from(inputChunks(), { highWaterMark: 1 })\n\n // Stream high-water marks are valid Transform options that `ZlibOptions` does not declare.\n const inflateOptions = {\n chunkSize: officeInflateChunkBytes,\n readableHighWaterMark: officeInflateChunkBytes,\n writableHighWaterMark: compressedChunkBytes\n }\n\n const inflater = createInflateRaw(inflateOptions)\n\n source.pipe(inflater)\n\n try {\n // Async iteration may combine buffered chunks; bound accounting slices explicitly.\n for await (const value of inflater) {\n if (!Buffer.isBuffer(value)) throw invalid()\n\n for (let offset = 0; offset < value.length; offset += officeInflateChunkBytes) {\n record(value.subarray(offset, offset + officeInflateChunkBytes))\n }\n }\n\n if (inflater.bytesWritten !== compressed.length) throw invalid()\n } finally {\n source.destroy()\n inflater.destroy()\n }\n}\n\nexport type ReadOfficeArchiveOptions = {\n /** XLSX: hyperlink start tags to capture across the workbook (default 0). */\n readonly maxHyperlinkTags?: number\n}\n\n/** A fresh stored-entry ZIP of validated parts; the attacker's ZIP index is never reused. */\nexport const storedArchive = (parts: Readonly<Record<string, Uint8Array>>) =>\n zipSync({ ...parts }, { level: 0 })\n\n/**\n * Inflate bounded input chunks, count actual output, and return the validated parts. The\n * attacker's ZIP index is discarded: parsers only get archives rebuilt from these parts.\n *\n * For XLSX, every part loses its `<hyperlink>` tags: SheetJS expands each hyperlink range into\n * per-cell objects before any budget runs, so one `ref=\"A1:XFD1048576\"` exhausts memory. The\n * removed tags are returned so links can still be shown. XLSX input that SheetJS would route to\n * its binary (XLSB), ODS, or Numbers parsers is rejected early with a clear error; the guarantee\n * is that SheetJS only receives `buildSheetJsInput`'s allowlisted archive.\n */\nexport const readOfficeArchive = (\n input: Uint8Array,\n format: OfficeFileFormat,\n limits: OfficeArchiveLimits,\n options: ReadOfficeArchiveOptions = {}\n) =>\n Effect.tryPromise({\n try: async (): Promise<NormalizedOfficeArchive> => {\n const bytes = Buffer.from(input.buffer, input.byteOffset, input.byteLength)\n\n if (bytes.length < 22 || bytes.length > limits.maxInputBytes) throw invalid()\n\n const entries = archiveEntries(bytes, limits)\n const mainPart = mainParts[format]\n const maxHyperlinkTags = options.maxHyperlinkTags ?? 0\n\n if (format === 'xlsx' && entries.some(entry => isAlternateFormatEntry(entry.name)))\n throw unsupportedXlsxParts()\n\n if (\n !entries.some(entry => entry.name === '[Content_Types].xml') ||\n !entries.some(entry => entry.name === mainPart)\n )\n throw invalid()\n\n const validated: Record<string, Uint8Array> = Object.create(null)\n const hyperlinkTags = new Map<string, Array<string>>()\n let capturedTags = 0\n let expandedBytes = 0\n\n for (const entry of entries) {\n const chunks: Array<Buffer> = []\n let entryBytes = 0\n\n const record = (chunk: Buffer) => {\n entryBytes += chunk.length\n expandedBytes += chunk.length\n\n if (expandedBytes > limits.maxExpandedBytes || entryBytes > entry.originalSize)\n throw tooLarge(expandedBytes)\n\n chunks.push(chunk)\n }\n\n const compressed = bytes.subarray(entry.start, entry.start + entry.compressedSize)\n\n if (entry.method === 0) record(compressed)\n else await inflateEntry(compressed, record)\n\n if (entryBytes !== entry.originalSize) throw invalid()\n\n const content = Buffer.concat(chunks, entryBytes)\n\n if (format !== 'xlsx') {\n validated[entry.name] = content\n continue\n }\n\n // Relationship targets need not end in .xml, so scan every entry without changing other\n // bytes (including UTF-8 and binary parts). A space prevents removal from joining\n // attacker-controlled fragments into a new parser-accepted hyperlink tag.\n const tags: Array<string> = []\n\n const rewritten = Buffer.from(\n content.toString('latin1').replace(hyperlinkTag, tag => {\n if (!tag.startsWith('</') && capturedTags < maxHyperlinkTags) {\n capturedTags += 1\n tags.push(tag)\n }\n\n return ' '\n }),\n 'latin1'\n )\n\n // SheetJS also decodes BOM-marked UTF-16 parts, which the Latin-1 strip cannot see\n // through, and the strip itself can shift byte alignment or create a BOM. Check the exact\n // bytes SheetJS will parse; Excel never writes UTF-16 parts, so reject rather than rewrite.\n if (utf16PartHasHyperlink(rewritten)) throw invalid()\n\n // Content types and relationships decide which parser SheetJS runs on each part.\n const lowerName = entry.name.toLowerCase()\n\n if (\n (lowerName === '[content_types].xml' && contentTypesRouteToBinary(rewritten)) ||\n (lowerName.endsWith('.rels') && relationshipsRouteToBinary(rewritten))\n )\n throw unsupportedXlsxParts()\n\n if (tags.length > 0) hyperlinkTags.set(entry.name, tags)\n\n validated[entry.name] = rewritten\n }\n\n return { parts: validated, hyperlinkTags }\n },\n // Out-of-range header reads (RangeError) and inflate failures are malformed archives too.\n catch: error => (error instanceof OfficeArchiveError ? error : invalid())\n })\n\n/**\n * Validate a DOCX, XLSX, or PPTX archive with bounded inflation and return a rebuilt stored-entry\n * ZIP of every validated part, for storage or for other parsers. XLSX parts lose their hyperlink\n * tags, and XLSX input with ODS or Numbers marker entries or XLSB parts is rejected; `.bin` parts\n * SheetJS never parses (printer settings, OLE objects) are kept so stored files still open.\n *\n * The output is not SheetJS input. Never run SheetJS on it directly: extract XLSX text through\n * `FileExtractor`, which hands SheetJS only an allowlisted archive it builds itself (worksheet,\n * shared-string, style, and core-property parts with generated content types and relationships).\n */\nexport const normalizeOfficeArchive = (\n bytes: Uint8Array,\n format: OfficeFileFormat,\n limits: Partial<OfficeArchiveLimits> = {}\n) =>\n readOfficeArchive(bytes, format, {\n maxArchiveEntries: limits.maxArchiveEntries ?? defaultFileExtractorLimits.maxArchiveEntries,\n maxExpandedBytes: limits.maxExpandedBytes ?? defaultFileExtractorLimits.maxExpandedBytes,\n maxInputBytes: limits.maxInputBytes ?? defaultFileExtractorLimits.maxInputBytes\n }).pipe(Effect.map(normalized => storedArchive(normalized.parts)))\n"],"mappings":";;;;;;;;;;AAqBA,MAAM,gBAAgB,IAAI,mBAAmB,EAAE,SAAS,0BAA0B,CAAC;AAEnF,MAAM,kBAAkB;AAExB,MAAM,YAAY,kBAChB,kBAAkB,KAAA,IACd,IAAI,mBAAmB,EAAE,SAAS,gBAAgB,CAAC,IACnD,IAAI,mBAAmB;CAAE,SAAS;CAAiB;AAAc,CAAC;AAExE,MAAM,uBAAuB;;AAG7B,MAAa,0BAA0B,KAAK;AAE5C,MAAM,YAAwD;CAC5D,MAAM;CACN,MAAM;CACN,MAAM;AACR;;AAWA,MAAM,kBAAkB,OAAe,WAAgC;CACrE,IAAI,MAAM,MAAM,SAAS;CACzB,MAAM,WAAW,KAAK,IAAI,GAAG,MAAM,KAAK;CAExC,OAAO,OAAO,YAAY,MAAM,aAAa,GAAG,MAAM,WAAY,OAAO;CAEzE,IAAI,MAAM,YAAY,MAAM,KAAK,MAAM,aAAa,MAAM,EAAE,MAAM,MAAM,QAAQ,MAAM,QAAQ;CAE9F,MAAM,QAAQ,MAAM,aAAa,MAAM,EAAE;CACzC,MAAM,gBAAgB,MAAM,aAAa,MAAM,EAAE;CACjD,MAAM,kBAAkB,MAAM,aAAa,MAAM,EAAE;CAEnD,IACE,MAAM,aAAa,MAAM,CAAC,MAAM,KAChC,MAAM,aAAa,MAAM,CAAC,MAAM,SAChC,UAAU,KACV,QAAQ,OAAO,qBACf,kBAAkB,kBAAkB,KAEpC,MAAM,QAAQ;CAEhB,MAAM,UAA+B,CAAC;CAEtC,MAAM,wBAAQ,IAAI,IAAY;CAC9B,IAAI,SAAS;CACb,IAAI,gBAAgB;CAEpB,KAAK,IAAI,QAAQ,GAAG,QAAQ,OAAO,SAAS,GAAG;EAC7C,IAAI,SAAS,KAAK,OAAO,MAAM,aAAa,MAAM,MAAM,UAAY,MAAM,QAAQ;EAElF,MAAM,QAAQ,MAAM,aAAa,SAAS,CAAC;EAC3C,MAAM,SAAS,MAAM,aAAa,SAAS,EAAE;EAC7C,MAAM,iBAAiB,MAAM,aAAa,SAAS,EAAE;EACrD,MAAM,eAAe,MAAM,aAAa,SAAS,EAAE;EACnD,MAAM,WAAW,MAAM,aAAa,SAAS,EAAE;EAC/C,MAAM,YAAY,MAAM,aAAa,SAAS,EAAE;EAChD,MAAM,cAAc,MAAM,aAAa,SAAS,EAAE;EAClD,MAAM,QAAQ,MAAM,aAAa,SAAS,EAAE;EAC5C,MAAM,OAAO,SAAS,KAAK,WAAW,YAAY;EAGlD,IACE,OAAO,OACP,aAAa,KACb,WAAW,SACV,QAAQ,WAAa,KACrB,WAAW,KAAK,WAAW,KAC5B,MAAM,aAAa,SAAS,EAAE,MAAM,KACpC,mBAAmB,cACnB,iBAAiB,cACjB,QAAQ,KAAK,iBAEb,MAAM,QAAQ;EAEhB,MAAM,YAAY,MAAM,SAAS,SAAS,IAAI,SAAS,KAAK,QAAQ;EACpE,MAAM,OAAO,IAAI,YAAY,SAAS,EAAE,OAAO,KAAK,CAAC,EAAE,OAAO,SAAS;EAGvE,IACE,oBAAoB,KAAK,IAAI,KAC7B,KAAK,WAAW,GAAG,KACnB,KAAK,SAAS,IAAI,KAClB,KAAK,MAAM,GAAG,EAAE,MAAK,SAAQ,SAAS,QAAQ,SAAS,GAAG,KAC1D,MAAM,IAAI,KAAK,YAAY,CAAC,KAC5B,oBAAoB,KAAK,IAAI,GAE7B,MAAM,QAAQ;EAEhB,MAAM,IAAI,KAAK,YAAY,CAAC;EAE5B,IACE,MAAM,aAAa,KAAK,MAAM,YAC9B,MAAM,aAAa,QAAQ,CAAC,MAAM,SAClC,MAAM,aAAa,QAAQ,CAAC,MAAM,UAClC,MAAM,aAAa,QAAQ,EAAE,MAAM,UAEnC,MAAM,QAAQ;EAEhB,MAAM,QAAQ,QAAQ,KAAK,WAAW,MAAM,aAAa,QAAQ,EAAE;EAEnE,IACE,QAAQ,iBAAiB,mBACzB,CAAC,MAAM,SAAS,QAAQ,IAAI,QAAQ,KAAK,QAAQ,EAAE,OAAO,SAAS,GAEnE,MAAM,QAAQ;EAGhB,KAAK,MAAM,CAAC,UAAU,aAAa,CACjC,CAAC,IAAI,cAAc,GACnB,CAAC,IAAI,YAAY,CACnB,GAAY;GACV,MAAM,QAAQ,MAAM,aAAa,QAAQ,QAAQ;GAEjD,IAAI,UAAU,YAAY,GAAG,QAAQ,OAAO,KAAK,UAAU,IAAI,MAAM,QAAQ;EAC/E;EAEA,iBAAiB;EAEjB,IAAI,gBAAgB,OAAO,kBAAkB,MAAM,SAAS;EAE5D,QAAQ,KAAK;GAAE;GAAM;GAAO;GAAgB;GAAc;EAAO,CAAC;EAClE,SAAS;CACX;CAEA,IAAI,WAAW,KAAK,MAAM,QAAQ;CAElC,OAAO;AACT;;;;;AAMA,MAAM,eAAe;AAErB,MAAM,oBAAoB;;;;;;AAO1B,MAAa,yBAAyB,YACpC,iBAAiB,OAAO,EACrB,MAAM,CAAC,EACP,MAAK,SAAQ,kBAAkB,KAAK,IAAI,CAAC;AAE9C,MAAM,6BACJ,IAAI,mBAAmB,EACrB,SAAS,8DACX,CAAC;AAYH,MAAM,eAAe,OAAO,YAAoB,WAAmD;CACjG,UAAU,cAAc;EACtB,KAAK,IAAI,SAAS,GAAG,SAAS,WAAW,QAAQ,UAAU,sBACzD,MAAM,WAAW,SAAS,QAAQ,SAAS,oBAAoB;CAEnE;CAEA,MAAM,SAAS,SAAS,KAAK,YAAY,GAAG,EAAE,eAAe,EAAE,CAAC;CAShE,MAAM,WAAW,iBAAiB;EALhC,WAAW;EACX,uBAAuB;EACvB,uBAAuB;CAGsB,CAAC;CAEhD,OAAO,KAAK,QAAQ;CAEpB,IAAI;EAEF,WAAW,MAAM,SAAS,UAAU;GAClC,IAAI,CAAC,OAAO,SAAS,KAAK,GAAG,MAAM,QAAQ;GAE3C,KAAK,IAAI,SAAS,GAAG,SAAS,MAAM,QAAQ,UAAU,yBACpD,OAAO,MAAM,SAAS,QAAQ,SAAS,uBAAuB,CAAC;EAEnE;EAEA,IAAI,SAAS,iBAAiB,WAAW,QAAQ,MAAM,QAAQ;CACjE,UAAU;EACR,OAAO,QAAQ;EACf,SAAS,QAAQ;CACnB;AACF;;AAQA,MAAa,iBAAiB,UAC5B,QAAQ,EAAE,GAAG,MAAM,GAAG,EAAE,OAAO,EAAE,CAAC;;;;;;;;;;;AAYpC,MAAa,qBACX,OACA,QACA,QACA,UAAoC,CAAC,MAErC,OAAO,WAAW;CAChB,KAAK,YAA8C;EACjD,MAAM,QAAQ,OAAO,KAAK,MAAM,QAAQ,MAAM,YAAY,MAAM,UAAU;EAE1E,IAAI,MAAM,SAAS,MAAM,MAAM,SAAS,OAAO,eAAe,MAAM,QAAQ;EAE5E,MAAM,UAAU,eAAe,OAAO,MAAM;EAC5C,MAAM,WAAW,UAAU;EAC3B,MAAM,mBAAmB,QAAQ,oBAAoB;EAErD,IAAI,WAAW,UAAU,QAAQ,MAAK,UAAS,uBAAuB,MAAM,IAAI,CAAC,GAC/E,MAAM,qBAAqB;EAE7B,IACE,CAAC,QAAQ,MAAK,UAAS,MAAM,SAAS,qBAAqB,KAC3D,CAAC,QAAQ,MAAK,UAAS,MAAM,SAAS,QAAQ,GAE9C,MAAM,QAAQ;EAEhB,MAAM,YAAwC,OAAO,OAAO,IAAI;EAChE,MAAM,gCAAgB,IAAI,IAA2B;EACrD,IAAI,eAAe;EACnB,IAAI,gBAAgB;EAEpB,KAAK,MAAM,SAAS,SAAS;GAC3B,MAAM,SAAwB,CAAC;GAC/B,IAAI,aAAa;GAEjB,MAAM,UAAU,UAAkB;IAChC,cAAc,MAAM;IACpB,iBAAiB,MAAM;IAEvB,IAAI,gBAAgB,OAAO,oBAAoB,aAAa,MAAM,cAChE,MAAM,SAAS,aAAa;IAE9B,OAAO,KAAK,KAAK;GACnB;GAEA,MAAM,aAAa,MAAM,SAAS,MAAM,OAAO,MAAM,QAAQ,MAAM,cAAc;GAEjF,IAAI,MAAM,WAAW,GAAG,OAAO,UAAU;QACpC,MAAM,aAAa,YAAY,MAAM;GAE1C,IAAI,eAAe,MAAM,cAAc,MAAM,QAAQ;GAErD,MAAM,UAAU,OAAO,OAAO,QAAQ,UAAU;GAEhD,IAAI,WAAW,QAAQ;IACrB,UAAU,MAAM,QAAQ;IACxB;GACF;GAKA,MAAM,OAAsB,CAAC;GAE7B,MAAM,YAAY,OAAO,KACvB,QAAQ,SAAS,QAAQ,EAAE,QAAQ,eAAc,QAAO;IACtD,IAAI,CAAC,IAAI,WAAW,IAAI,KAAK,eAAe,kBAAkB;KAC5D,gBAAgB;KAChB,KAAK,KAAK,GAAG;IACf;IAEA,OAAO;GACT,CAAC,GACD,QACF;GAKA,IAAI,sBAAsB,SAAS,GAAG,MAAM,QAAQ;GAGpD,MAAM,YAAY,MAAM,KAAK,YAAY;GAEzC,IACG,cAAc,yBAAyB,0BAA0B,SAAS,KAC1E,UAAU,SAAS,OAAO,KAAK,2BAA2B,SAAS,GAEpE,MAAM,qBAAqB;GAE7B,IAAI,KAAK,SAAS,GAAG,cAAc,IAAI,MAAM,MAAM,IAAI;GAEvD,UAAU,MAAM,QAAQ;EAC1B;EAEA,OAAO;GAAE,OAAO;GAAW;EAAc;CAC3C;CAEA,QAAO,UAAU,iBAAiB,qBAAqB,QAAQ,QAAQ;AACzE,CAAC;;;;;;;;;;;AAYH,MAAa,0BACX,OACA,QACA,SAAuC,CAAC,MAExC,kBAAkB,OAAO,QAAQ;CAC/B,mBAAmB,OAAO,qBAAqB,2BAA2B;CAC1E,kBAAkB,OAAO,oBAAoB,2BAA2B;CACxE,eAAe,OAAO,iBAAiB,2BAA2B;AACpE,CAAC,EAAE,KAAK,OAAO,KAAI,eAAc,cAAc,WAAW,KAAK,CAAC,CAAC"}
@@ -0,0 +1,6 @@
1
+ //#region src/node/pptx-text.d.ts
2
+ /** Slide text in slide order, then speaker notes, from the parts of a validated archive. */
3
+ declare const extractPptxText: (parts: Readonly<Record<string, Uint8Array>>) => string;
4
+ //#endregion
5
+ export { extractPptxText };
6
+ //# sourceMappingURL=pptx-text.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"pptx-text.d.mts","names":[],"sources":["../../src/node/pptx-text.ts"],"mappings":";;cAkHa,eAAA,GAAmB,KAAA,EAAO,QAAA,CAAS,MAAA,SAAe,UAAA"}
@@ -0,0 +1,63 @@
1
+ import { decodeXmlEntities } from "./xml-text.mjs";
2
+ import { strFromU8 } from "fflate";
3
+ //#region src/node/pptx-text.ts
4
+ const slideXmlFile = /^ppt\/slides\/slide(\d+)\.xml$/;
5
+ const notesXmlFile = /^ppt\/notesSlides\/notesSlide(\d+)\.xml$/;
6
+ const optionalXmlPrefix = `(?:[A-Za-z_][\\w.-]*:)?`;
7
+ const xmlElement = (localName) => ({
8
+ start: new RegExp(`<${optionalXmlPrefix}${localName}\\b[^<>]*>`, "g"),
9
+ end: new RegExp(`</${optionalXmlPrefix}${localName}>`, "g")
10
+ });
11
+ const paragraphXml = xmlElement("p");
12
+ const textXml = xmlElement("t");
13
+ const lineBreakXml = new RegExp(`<${optionalXmlPrefix}br\\b[^<>]*/>`, "g");
14
+ const tabXml = new RegExp(`<${optionalXmlPrefix}tab\\b[^<>]*/>`, "g");
15
+ /** Each start tag paired with the next end tag after it (the old lazy `[\s\S]*?` match). */
16
+ const elementMatches = (xml, element) => {
17
+ const matches = [];
18
+ let position = 0;
19
+ for (;;) {
20
+ element.start.lastIndex = position;
21
+ const start = element.start.exec(xml);
22
+ if (start === null) return matches;
23
+ const contentStart = start.index + start[0].length;
24
+ element.end.lastIndex = contentStart;
25
+ const end = element.end.exec(xml);
26
+ if (end === null) return matches;
27
+ position = end.index + end[0].length;
28
+ matches.push({
29
+ outer: xml.slice(start.index, position),
30
+ inner: xml.slice(contentStart, end.index)
31
+ });
32
+ }
33
+ };
34
+ const indexedXmlFile = (fileName, bytes, pattern, group) => {
35
+ const indexText = pattern.exec(fileName)?.[1];
36
+ if (indexText === void 0) return void 0;
37
+ const index = Number.parseInt(indexText, 10);
38
+ if (!Number.isInteger(index)) return void 0;
39
+ return {
40
+ fileName,
41
+ bytes,
42
+ group,
43
+ index
44
+ };
45
+ };
46
+ const pptxXmlFile = (fileName, bytes) => indexedXmlFile(fileName, bytes, slideXmlFile, 0) ?? indexedXmlFile(fileName, bytes, notesXmlFile, 1);
47
+ const comparePptxXmlFiles = (left, right) => left.group - right.group || left.index - right.index || left.fileName.localeCompare(right.fileName);
48
+ const extractParagraphText = (paragraph) => {
49
+ return elementMatches(paragraph.replace(lineBreakXml, "<a:t>\n</a:t>").replace(tabXml, "<a:t> </a:t>"), textXml).map((match) => decodeXmlEntities(match.inner)).join("").trim();
50
+ };
51
+ const extractXmlText = (xml) => {
52
+ const paragraphs = elementMatches(xml, paragraphXml).map((match) => match.outer);
53
+ return (paragraphs.length > 0 ? paragraphs : [xml]).map(extractParagraphText).filter((text) => text.length > 0).join("\n");
54
+ };
55
+ /** Slide text in slide order, then speaker notes, from the parts of a validated archive. */
56
+ const extractPptxText = (parts) => Object.entries(parts).flatMap(([fileName, fileBytes]) => {
57
+ const xmlFile = pptxXmlFile(fileName, fileBytes);
58
+ return xmlFile === void 0 ? [] : [xmlFile];
59
+ }).sort(comparePptxXmlFiles).map((file) => extractXmlText(strFromU8(file.bytes))).filter((text) => text.length > 0).join("\n\n");
60
+ //#endregion
61
+ export { extractPptxText };
62
+
63
+ //# sourceMappingURL=pptx-text.mjs.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"pptx-text.mjs","names":[],"sources":["../../src/node/pptx-text.ts"],"sourcesContent":["import { strFromU8 } from 'fflate'\nimport { decodeXmlEntities } from './xml-text.ts'\n\ntype PptxXmlFile = {\n readonly fileName: string\n readonly bytes: Uint8Array\n readonly group: number\n readonly index: number\n}\n\nconst slideXmlFile = /^ppt\\/slides\\/slide(\\d+)\\.xml$/\n\nconst notesXmlFile = /^ppt\\/notesSlides\\/notesSlide(\\d+)\\.xml$/\n\nconst xmlName = '[A-Za-z_][\\\\w.-]*'\n\nconst optionalXmlPrefix = `(?:${xmlName}:)?`\n\n// Every pattern stops at the next `<` (`[^<>]`), and elements are paired in one forward pass, so\n// extraction stays linear in the part size even for unterminated or unbalanced markup.\ntype XmlElement = { readonly start: RegExp; readonly end: RegExp }\n\nconst xmlElement = (localName: string): XmlElement => ({\n start: new RegExp(`<${optionalXmlPrefix}${localName}\\\\b[^<>]*>`, 'g'),\n end: new RegExp(`</${optionalXmlPrefix}${localName}>`, 'g')\n})\n\nconst paragraphXml = xmlElement('p')\n\nconst textXml = xmlElement('t')\n\nconst lineBreakXml = new RegExp(`<${optionalXmlPrefix}br\\\\b[^<>]*/>`, 'g')\n\nconst tabXml = new RegExp(`<${optionalXmlPrefix}tab\\\\b[^<>]*/>`, 'g')\n\ntype ElementMatch = {\n /** Start tag through end tag. */\n readonly outer: string\n /** Text between the start and end tags. */\n readonly inner: string\n}\n\n/** Each start tag paired with the next end tag after it (the old lazy `[\\s\\S]*?` match). */\nconst elementMatches = (xml: string, element: XmlElement): ReadonlyArray<ElementMatch> => {\n const matches: Array<ElementMatch> = []\n let position = 0\n\n for (;;) {\n element.start.lastIndex = position\n const start = element.start.exec(xml)\n\n if (start === null) return matches\n\n const contentStart = start.index + start[0].length\n element.end.lastIndex = contentStart\n const end = element.end.exec(xml)\n\n // No end tag after this start means none after any later start either.\n if (end === null) return matches\n\n position = end.index + end[0].length\n matches.push({\n outer: xml.slice(start.index, position),\n inner: xml.slice(contentStart, end.index)\n })\n }\n}\n\nconst indexedXmlFile = (\n fileName: string,\n bytes: Uint8Array,\n pattern: RegExp,\n group: number\n): PptxXmlFile | undefined => {\n const indexText = pattern.exec(fileName)?.[1]\n\n if (indexText === undefined) return undefined\n\n const index = Number.parseInt(indexText, 10)\n\n if (!Number.isInteger(index)) return undefined\n\n return { fileName, bytes, group, index }\n}\n\nconst pptxXmlFile = (fileName: string, bytes: Uint8Array): PptxXmlFile | undefined =>\n indexedXmlFile(fileName, bytes, slideXmlFile, 0) ??\n indexedXmlFile(fileName, bytes, notesXmlFile, 1)\n\nconst comparePptxXmlFiles = (left: PptxXmlFile, right: PptxXmlFile) =>\n left.group - right.group ||\n left.index - right.index ||\n left.fileName.localeCompare(right.fileName)\n\nconst extractParagraphText = (paragraph: string) => {\n const xml = paragraph.replace(lineBreakXml, '<a:t>\\n</a:t>').replace(tabXml, '<a:t>\\t</a:t>')\n\n return elementMatches(xml, textXml)\n .map(match => decodeXmlEntities(match.inner))\n .join('')\n .trim()\n}\n\nconst extractXmlText = (xml: string) => {\n const paragraphs = elementMatches(xml, paragraphXml).map(match => match.outer)\n const textSources = paragraphs.length > 0 ? paragraphs : [xml]\n\n return textSources\n .map(extractParagraphText)\n .filter(text => text.length > 0)\n .join('\\n')\n}\n\n/** Slide text in slide order, then speaker notes, from the parts of a validated archive. */\nexport const extractPptxText = (parts: Readonly<Record<string, Uint8Array>>) =>\n Object.entries(parts)\n .flatMap(([fileName, fileBytes]) => {\n const xmlFile = pptxXmlFile(fileName, fileBytes)\n\n return xmlFile === undefined ? [] : [xmlFile]\n })\n .sort(comparePptxXmlFiles)\n .map(file => extractXmlText(strFromU8(file.bytes)))\n .filter(text => text.length > 0)\n .join('\\n\\n')\n"],"mappings":";;;AAUA,MAAM,eAAe;AAErB,MAAM,eAAe;AAIrB,MAAM,oBAAoB;AAM1B,MAAM,cAAc,eAAmC;CACrD,OAAO,IAAI,OAAO,IAAI,oBAAoB,UAAU,aAAa,GAAG;CACpE,KAAK,IAAI,OAAO,KAAK,oBAAoB,UAAU,IAAI,GAAG;AAC5D;AAEA,MAAM,eAAe,WAAW,GAAG;AAEnC,MAAM,UAAU,WAAW,GAAG;AAE9B,MAAM,eAAe,IAAI,OAAO,IAAI,kBAAkB,gBAAgB,GAAG;AAEzE,MAAM,SAAS,IAAI,OAAO,IAAI,kBAAkB,iBAAiB,GAAG;;AAUpE,MAAM,kBAAkB,KAAa,YAAqD;CACxF,MAAM,UAA+B,CAAC;CACtC,IAAI,WAAW;CAEf,SAAS;EACP,QAAQ,MAAM,YAAY;EAC1B,MAAM,QAAQ,QAAQ,MAAM,KAAK,GAAG;EAEpC,IAAI,UAAU,MAAM,OAAO;EAE3B,MAAM,eAAe,MAAM,QAAQ,MAAM,GAAG;EAC5C,QAAQ,IAAI,YAAY;EACxB,MAAM,MAAM,QAAQ,IAAI,KAAK,GAAG;EAGhC,IAAI,QAAQ,MAAM,OAAO;EAEzB,WAAW,IAAI,QAAQ,IAAI,GAAG;EAC9B,QAAQ,KAAK;GACX,OAAO,IAAI,MAAM,MAAM,OAAO,QAAQ;GACtC,OAAO,IAAI,MAAM,cAAc,IAAI,KAAK;EAC1C,CAAC;CACH;AACF;AAEA,MAAM,kBACJ,UACA,OACA,SACA,UAC4B;CAC5B,MAAM,YAAY,QAAQ,KAAK,QAAQ,IAAI;CAE3C,IAAI,cAAc,KAAA,GAAW,OAAO,KAAA;CAEpC,MAAM,QAAQ,OAAO,SAAS,WAAW,EAAE;CAE3C,IAAI,CAAC,OAAO,UAAU,KAAK,GAAG,OAAO,KAAA;CAErC,OAAO;EAAE;EAAU;EAAO;EAAO;CAAM;AACzC;AAEA,MAAM,eAAe,UAAkB,UACrC,eAAe,UAAU,OAAO,cAAc,CAAC,KAC/C,eAAe,UAAU,OAAO,cAAc,CAAC;AAEjD,MAAM,uBAAuB,MAAmB,UAC9C,KAAK,QAAQ,MAAM,SACnB,KAAK,QAAQ,MAAM,SACnB,KAAK,SAAS,cAAc,MAAM,QAAQ;AAE5C,MAAM,wBAAwB,cAAsB;CAGlD,OAAO,eAFK,UAAU,QAAQ,cAAc,eAAe,EAAE,QAAQ,QAAQ,cAErD,GAAG,OAAO,EAC/B,KAAI,UAAS,kBAAkB,MAAM,KAAK,CAAC,EAC3C,KAAK,EAAE,EACP,KAAK;AACV;AAEA,MAAM,kBAAkB,QAAgB;CACtC,MAAM,aAAa,eAAe,KAAK,YAAY,EAAE,KAAI,UAAS,MAAM,KAAK;CAG7E,QAFoB,WAAW,SAAS,IAAI,aAAa,CAAC,GAAG,GAG1D,IAAI,oBAAoB,EACxB,QAAO,SAAQ,KAAK,SAAS,CAAC,EAC9B,KAAK,IAAI;AACd;;AAGA,MAAa,mBAAmB,UAC9B,OAAO,QAAQ,KAAK,EACjB,SAAS,CAAC,UAAU,eAAe;CAClC,MAAM,UAAU,YAAY,UAAU,SAAS;CAE/C,OAAO,YAAY,KAAA,IAAY,CAAC,IAAI,CAAC,OAAO;AAC9C,CAAC,EACA,KAAK,mBAAmB,EACxB,KAAI,SAAQ,eAAe,UAAU,KAAK,KAAK,CAAC,CAAC,EACjD,QAAO,SAAQ,KAAK,SAAS,CAAC,EAC9B,KAAK,MAAM"}
@@ -0,0 +1,89 @@
1
+ //#region src/node/sheetjs-xml.d.ts
2
+ /**
3
+ * The texts SheetJS can read from a part: its Latin-1 ("binary") view and, for BOM-marked parts,
4
+ * the UTF-16 decodings of `cc2str` (little- and big-endian from byte 2, including its
5
+ * `arr[1]/arr[2]` Buffer check) plus an extra odd-offset big-endian decode.
6
+ */
7
+ declare const sheetJsTextViews: (content: Uint8Array) => ReadonlyArray<string>;
8
+ type SheetJsTag = {
9
+ /** The tag up to its first space, line feed, or carriage return (SheetJS `y[0]`). */readonly head: string; /** Raw (still escaped) attribute values by SheetJS key, plus lower-cased copies. */
10
+ readonly attributes: ReadonlyMap<string, string>;
11
+ };
12
+ /**
13
+ * Port of SheetJS `parsexmltag`: exact-case keys (plus lower-cased copies), a namespace prefix
14
+ * dropped, an unprefixed name cut at its first `_`, the last value winning. Values are raw.
15
+ */
16
+ declare const parseSheetJsTag: (tag: string) => SheetJsTag;
17
+ /** Every tag SheetJS's pattern finds in `text`, in document order. */
18
+ declare function sheetJsTags(text: string): Generator<SheetJsTag>;
19
+ /** SheetJS `strip_ns`: the first `<prefix:` (or `</prefix:`) loses its prefix. */
20
+ declare const stripSheetJsNamespace: (head: string) => string;
21
+ /** SheetJS `utf8read` in Node: the Latin-1 ("binary") string read back as UTF-8. */
22
+ declare const sheetJsUtf8Read: (binary: string) => string;
23
+ /**
24
+ * SheetJS `unescapexml` for text without CDATA, quirks included: entity names match ignoring
25
+ * case but only lower-case ones map (`&QUOT;` becomes U+0000), `&#X41;` is read as decimal, and
26
+ * numeric references wrap at U+FFFF. Returns `undefined` for text with a CDATA marker, which
27
+ * SheetJS splits recursively (and never ends for an unterminated one).
28
+ */
29
+ declare const sheetJsUnescapeXml: (text: string) => string | undefined;
30
+ /** An attribute value as SheetJS reads text attributes: `unescapexml(utf8read(raw))`. */
31
+ declare const sheetJsAttributeText: (raw: string) => string | undefined;
32
+ /**
33
+ * Escape `text` for a double-quoted attribute of a generated UTF-8 part so that SheetJS's
34
+ * `unescapexml(utf8read(…))` returns `text` exactly: markup characters become entities, an `_`
35
+ * that would start an `_xHHHH_` code becomes `_x005F_`, and control characters, U+FFFE, U+FFFF,
36
+ * and lone surrogates become `_xHHHH_` codes.
37
+ */
38
+ declare const sheetJsAttributeEscape: (text: string) => string;
39
+ /** SheetJS's own `XML_HEADER`, used for every generated part. */
40
+ declare const sheetJsXmlHeader = "<?xml version=\"1.0\" encoding=\"UTF-8\" standalone=\"yes\"?>\r\n";
41
+ declare const spreadsheetMainNamespace = "http://schemas.openxmlformats.org/spreadsheetml/2006/main";
42
+ declare const officeDocumentRelationshipsNamespace = "http://schemas.openxmlformats.org/officeDocument/2006/relationships";
43
+ /**
44
+ * `text` without its simple opening tags, `<(?:[\w.-]+:)?[\w.-]+>`: a superset of the tags SheetJS
45
+ * removes before decoding (`<(?:\w+:)?(?:si|sstItem)>` in `parse_sst_xml`, `<(?:\w+:)?r>` in
46
+ * `parse_rs`). One forward scan: a `<` that does not start such a tag is kept and the scan resumes
47
+ * at the next `<`, so every character is read at most twice.
48
+ */
49
+ declare const withoutSimpleTags: (text: string) => string;
50
+ /**
51
+ * Whether SheetJS could meet a CDATA marker in this part. SheetJS's `unescapexml` handles CDATA by
52
+ * recursing on a string two characters shorter and copying the whole tail at every level, so an
53
+ * unterminated marker costs quadratic time and memory.
54
+ *
55
+ * What SheetJS hands to `unescapexml` in a worksheet or shared-strings part comes from the part
56
+ * text through two kinds of transformation:
57
+ *
58
+ * - Decodes: raw (`<v>` of every cell), `utf8read(raw)` (shared and inline strings), and
59
+ * `utf8read(unescapexml(raw))` (cells of type `str`, decoded again after `utf8read`).
60
+ * `utf8read` keeps only the low byte of each character, so U+013C from `_x013C_`, `&#x13C;`, or
61
+ * `&#316;` becomes `<`.
62
+ * - Tag removal before decoding: `parse_sst_xml` removes every `<si>`/`<sstItem>` opening tag
63
+ * from the whole shared-strings table, and `parse_rs` removes every `<r>` opening tag from rich
64
+ * text (after `utf8read`). Inline strings (`t="inlineStr"`) call `parse_si` without options,
65
+ * so their rich text is processed even with `cellHTML: false`. `A<<r>![CDATA[B` thus reaches
66
+ * `unescapexml` as `A<![CDATA[B`.
67
+ *
68
+ * So the check rejects a text view (`sheetJsTextViews`) when:
69
+ *
70
+ * - the view, or `utf8read` of it, contains `<<` or `<!`. A marker assembled by removing tags
71
+ * needs a literal `<` (in the view, or from `utf8read`) followed by a removed tag, or by `!`,
72
+ * and every removed tag starts with `<`;
73
+ * - the view, or the view without any simple opening tag (`withoutSimpleTags`, a superset of
74
+ * SheetJS's removals), meets the marker raw or after any chain of up to two steps of
75
+ * `unescapexml` and `utf8read`, in any order (a superset of SheetJS's decode sequences).
76
+ *
77
+ * Every step is a linear pass, at most a few per view. The first rule deliberately fails closed:
78
+ * it also rejects XML comments, `<!DOCTYPE`, and any other `<!…` declaration, which Excel,
79
+ * LibreOffice, and Google Sheets never write in worksheets or shared strings (nor a literal `<<`).
80
+ */
81
+ declare const sheetJsCouldReadCdata: (content: Uint8Array) => boolean;
82
+ /**
83
+ * The `dc:title` of a core-properties part, read as SheetJS `parse_core_props` reads it, without
84
+ * handing the part to SheetJS. A title holding CDATA is ignored.
85
+ */
86
+ declare const coreTitle: (content: Uint8Array | undefined) => string | undefined;
87
+ //#endregion
88
+ export { SheetJsTag, coreTitle, officeDocumentRelationshipsNamespace, parseSheetJsTag, sheetJsAttributeEscape, sheetJsAttributeText, sheetJsCouldReadCdata, sheetJsTags, sheetJsTextViews, sheetJsUnescapeXml, sheetJsUtf8Read, sheetJsXmlHeader, spreadsheetMainNamespace, stripSheetJsNamespace, withoutSimpleTags };
89
+ //# sourceMappingURL=sheetjs-xml.d.mts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"sheetjs-xml.d.mts","names":[],"sources":["../../src/node/sheetjs-xml.ts"],"mappings":";;AAgBA;;;;cAAa,gBAAA,GAAoB,OAAA,EAAS,UAAA,KAAa,aAAa;AAAA,KA2BxD,UAAA;EA3B2C,8FA6B5C,IAAA,UA7ByD;EAAA,SA+BzD,UAAA,EAAY,WAAW;AAAA;;;;;cAOrB,eAAA,GAAmB,GAAA,aAAc,UAwC7C;;iBAGgB,WAAA,CAAY,IAAA,WAAe,SAAS,CAAC,UAAA;AA3CtD;AAAA,cAgDa,qBAAA,GAAyB,IAAY;;cAGrC,eAAA,GAAmB,MAAc;AAX7C;AAGD;;;;;AAHC,cA2BY,kBAAA,GAAsB,IAAY;;cAgBlC,oBAAA,GAAwB,GAAW;AAxCgB;AAKhE;;;;AAAkD;AALc,cAgEnD,sBAAA,GAA0B,IAAY;;cAuCtC,gBAAA;AAAA,cAEA,wBAAA;AAAA,cAEA,oCAAA;AAnFb;;;;AAA+C;AAgB/C;AAhBA,cAiIa,iBAAA,GAAqB,IAAY;;;AAjHE;AAwBhD;;;;AAAmD;AAuCnD;;;;AAA6B;AAE7B;;;;AAAqC;AAErC;;;;AAAiD;AA8CjD;;;;AAA8C;AA4D9C;;;cAAa,qBAAA,GAAyB,OAAmB,EAAV,UAAU;AAAA;AAkCzD;;;AAlCyD,cAkC5C,SAAA,GAAa,OAA+B,EAAtB,UAAU"}