@bevel-software/platform-core-backend 0.11.2 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. package/THIRD-PARTY-NOTICES.md +1165 -427
  2. package/dist/core/create-core-server.js +1 -1
  3. package/dist/core/create-core-server.js.map +1 -1
  4. package/dist/core/create-core-services.d.ts +2 -0
  5. package/dist/core/create-core-services.d.ts.map +1 -1
  6. package/dist/core/create-core-services.js +5 -0
  7. package/dist/core/create-core-services.js.map +1 -1
  8. package/dist/core-config.d.ts +7 -0
  9. package/dist/core-config.d.ts.map +1 -1
  10. package/dist/core-config.js +9 -0
  11. package/dist/core-config.js.map +1 -1
  12. package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
  13. package/dist/modules/code-mode/code-mode.tool.js +7 -1
  14. package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
  15. package/dist/modules/kb-fs/clone-config.d.ts +40 -2
  16. package/dist/modules/kb-fs/clone-config.d.ts.map +1 -1
  17. package/dist/modules/kb-fs/clone-config.js +94 -2
  18. package/dist/modules/kb-fs/clone-config.js.map +1 -1
  19. package/dist/modules/workflow/workflow.service.d.ts +38 -0
  20. package/dist/modules/workflow/workflow.service.d.ts.map +1 -1
  21. package/dist/modules/workflow/workflow.service.js +112 -6
  22. package/dist/modules/workflow/workflow.service.js.map +1 -1
  23. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
  24. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
  25. package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
  26. package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
  27. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
  28. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
  29. package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
  30. package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
  31. package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
  32. package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
  33. package/dist/modules/workspace/file-readers/document-reader.js +59 -0
  34. package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
  35. package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
  36. package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
  37. package/dist/modules/workspace/file-readers/email-reader.js +19 -0
  38. package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
  39. package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
  40. package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
  41. package/dist/modules/workspace/file-readers/email-text.js +151 -0
  42. package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
  43. package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
  44. package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
  45. package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
  46. package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
  47. package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
  48. package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
  49. package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
  50. package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
  51. package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
  52. package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
  53. package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
  54. package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
  55. package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
  56. package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
  57. package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
  58. package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
  59. package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
  60. package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
  61. package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
  62. package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
  63. package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
  64. package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
  65. package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
  66. package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
  67. package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
  68. package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
  69. package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
  70. package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
  71. package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
  72. package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
  73. package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
  74. package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
  75. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
  76. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
  77. package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
  78. package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
  79. package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
  80. package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
  81. package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
  82. package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
  83. package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
  84. package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
  85. package/dist/modules/workspace/file-readers/file-reader.js +55 -0
  86. package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
  87. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
  88. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
  89. package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
  90. package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
  91. package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
  92. package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
  93. package/dist/modules/workspace/file-readers/image-read.js +108 -0
  94. package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
  95. package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
  96. package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
  97. package/dist/modules/workspace/file-readers/image-reader.js +30 -0
  98. package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
  99. package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
  100. package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
  101. package/dist/modules/workspace/file-readers/odf-text.js +116 -0
  102. package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
  103. package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
  104. package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
  105. package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
  106. package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
  107. package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
  108. package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
  109. package/dist/modules/workspace/file-readers/text-reader.js +117 -0
  110. package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
  111. package/dist/modules/workspace/startup/kb-git.d.ts.map +1 -1
  112. package/dist/modules/workspace/startup/kb-git.js +21 -4
  113. package/dist/modules/workspace/startup/kb-git.js.map +1 -1
  114. package/dist/modules/workspace/workspace.service.d.ts +52 -8
  115. package/dist/modules/workspace/workspace.service.d.ts.map +1 -1
  116. package/dist/modules/workspace/workspace.service.js +121 -23
  117. package/dist/modules/workspace/workspace.service.js.map +1 -1
  118. package/dist/modules/workspace/workspace.tools.d.ts +2 -1
  119. package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
  120. package/dist/modules/workspace/workspace.tools.js +158 -15
  121. package/dist/modules/workspace/workspace.tools.js.map +1 -1
  122. package/package.json +11 -6
  123. package/src/core/create-core-server.ts +1 -1
  124. package/src/core/create-core-services.ts +6 -0
  125. package/src/core-config.ts +9 -0
  126. package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
  127. package/src/modules/code-mode/code-mode.tool.ts +7 -1
  128. package/src/modules/kb-fs/__tests__/clone-config.test.ts +63 -2
  129. package/src/modules/kb-fs/clone-config.ts +97 -2
  130. package/src/modules/secrets-vault/secrets-vault.routes.ts +582 -582
  131. package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
  132. package/src/modules/workflow/__tests__/workflow.service.commitFileWhileLocked.test.ts +11 -5
  133. package/src/modules/workflow/__tests__/workflow.service.releaseLock.test.ts +172 -7
  134. package/src/modules/workflow/workflow.service.ts +118 -6
  135. package/src/modules/workspace/__tests__/workspace.service.test.ts +1 -1
  136. package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
  137. package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
  138. package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
  139. package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
  140. package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
  141. package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
  142. package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
  143. package/src/modules/workspace/file-readers/document-reader.ts +64 -0
  144. package/src/modules/workspace/file-readers/email-reader.ts +21 -0
  145. package/src/modules/workspace/file-readers/email-text.ts +193 -0
  146. package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
  147. package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
  148. package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
  149. package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
  150. package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
  151. package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
  152. package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
  153. package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
  154. package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
  155. package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
  156. package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
  157. package/src/modules/workspace/file-readers/file-reader.ts +104 -0
  158. package/src/modules/workspace/file-readers/image-read.ts +122 -0
  159. package/src/modules/workspace/file-readers/image-reader.ts +39 -0
  160. package/src/modules/workspace/file-readers/odf-text.ts +123 -0
  161. package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
  162. package/src/modules/workspace/file-readers/text-reader.ts +131 -0
  163. package/src/modules/workspace/startup/__tests__/kb-startup-runner.test.ts +141 -0
  164. package/src/modules/workspace/startup/kb-git.ts +20 -7
  165. package/src/modules/workspace/workspace.service.ts +132 -25
  166. package/src/modules/workspace/workspace.tools.ts +174 -12
@@ -0,0 +1,17 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ /**
3
+ * Extract the BODY text of a `.odt` (OpenDocument Text) document.
4
+ *
5
+ * An odt is a zip whose main part is `content.xml`; the body lives under
6
+ * `<office:text>`. Headers/footers are skipped like docx — in ODF they live in
7
+ * `styles.xml`, which is never opened, so reading `content.xml` alone IS the
8
+ * body-only extraction.
9
+ *
10
+ * Paragraphs (`<text:p>`) and headings (`<text:h>`, heading text as its own
11
+ * line) become lines in document order; `<text:span>` runs inside concatenate
12
+ * with NO separator, and the ODF whitespace elements (`<text:tab/>`,
13
+ * `<text:line-break/>`, `<text:s text:c="N"/>`) render as real characters —
14
+ * see `odfParagraphText`.
15
+ */
16
+ export declare function extractOdt(bytes: Buffer): ExtractResult;
17
+ //# sourceMappingURL=extract-odt.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extract-odt.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-odt.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAI5D;;;;;;;;;;;;;GAaG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CA6BvD"}
@@ -0,0 +1,45 @@
1
+ import { localBlocks, removeLocalElements } from './ooxml-text.js';
2
+ import { odfParagraphBlocks, odfParagraphText, readOdfContentXml } from './odf-text.js';
3
+ /**
4
+ * Extract the BODY text of a `.odt` (OpenDocument Text) document.
5
+ *
6
+ * An odt is a zip whose main part is `content.xml`; the body lives under
7
+ * `<office:text>`. Headers/footers are skipped like docx — in ODF they live in
8
+ * `styles.xml`, which is never opened, so reading `content.xml` alone IS the
9
+ * body-only extraction.
10
+ *
11
+ * Paragraphs (`<text:p>`) and headings (`<text:h>`, heading text as its own
12
+ * line) become lines in document order; `<text:span>` runs inside concatenate
13
+ * with NO separator, and the ODF whitespace elements (`<text:tab/>`,
14
+ * `<text:line-break/>`, `<text:s text:c="N"/>`) render as real characters —
15
+ * see `odfParagraphText`.
16
+ */
17
+ export function extractOdt(bytes) {
18
+ const content = readOdfContentXml(bytes, '.odt');
19
+ if (!content.ok)
20
+ return content;
21
+ // Table cells contain their own <text:p>, so the flat paragraph scan renders
22
+ // table text too (one line per cell paragraph, like the raw document order).
23
+ // The text BODY, read by the parser and matched on its LOCAL name: a
24
+ // comment mentioning `</office:text>` used to terminate the body early and
25
+ // drop every paragraph after it, and an ODT binding the office namespace to
26
+ // another prefix had no body at all.
27
+ const body = localBlocks(content.xml, 'text')[0] ?? content.xml;
28
+ // Tracked-change bookkeeping is not body text: `<text:tracked-changes>`
29
+ // stores every DELETION's content as ordinary paragraphs, so the flat scan
30
+ // below would read deleted text back in as document lines. Removed by its
31
+ // parsed element boundaries before the paragraph walk.
32
+ // `<office:annotation>` is a COMMENT on the document, stored as ordinary
33
+ // paragraphs: read flat, a reviewer's note came back as a document line.
34
+ const visible = removeLocalElements(body, ['tracked-changes', 'annotation']);
35
+ const lines = odfParagraphBlocks(visible).map(odfParagraphText);
36
+ const paragraphs = lines.length;
37
+ while (lines.length > 0 && lines[lines.length - 1].trim() === '')
38
+ lines.pop();
39
+ return {
40
+ ok: true,
41
+ summary: `${paragraphs} paragraph${paragraphs === 1 ? '' : 's'}; layout, images and formatting omitted`,
42
+ text: lines.join('\n'),
43
+ };
44
+ }
45
+ //# sourceMappingURL=extract-odt.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extract-odt.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-odt.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,WAAW,EAAE,mBAAmB,EAAE,MAAM,iBAAiB,CAAC;AACnE,OAAO,EAAE,kBAAkB,EAAE,gBAAgB,EAAE,iBAAiB,EAAE,MAAM,eAAe,CAAC;AAExF;;;;;;;;;;;;;GAaG;AACH,MAAM,UAAU,UAAU,CAAC,KAAa;IACtC,MAAM,OAAO,GAAG,iBAAiB,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC;IACjD,IAAI,CAAC,OAAO,CAAC,EAAE;QAAE,OAAO,OAAO,CAAC;IAEhC,6EAA6E;IAC7E,6EAA6E;IAC7E,qEAAqE;IACrE,2EAA2E;IAC3E,4EAA4E;IAC5E,qCAAqC;IACrC,MAAM,IAAI,GAAG,WAAW,CAAC,OAAO,CAAC,GAAG,EAAE,MAAM,CAAC,CAAC,CAAC,CAAC,IAAI,OAAO,CAAC,GAAG,CAAC;IAEhE,wEAAwE;IACxE,2EAA2E;IAC3E,0EAA0E;IAC1E,uDAAuD;IACvD,yEAAyE;IACzE,yEAAyE;IACzE,MAAM,OAAO,GAAG,mBAAmB,CAAC,IAAI,EAAE,CAAC,iBAAiB,EAAE,YAAY,CAAC,CAAC,CAAC;IAE7E,MAAM,KAAK,GAAG,kBAAkB,CAAC,OAAO,CAAC,CAAC,GAAG,CAAC,gBAAgB,CAAC,CAAC;IAChE,MAAM,UAAU,GAAG,KAAK,CAAC,MAAM,CAAC;IAChC,OAAO,KAAK,CAAC,MAAM,GAAG,CAAC,IAAI,KAAK,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,IAAI,EAAE,KAAK,EAAE;QAAE,KAAK,CAAC,GAAG,EAAE,CAAC;IAE9E,OAAO;QACL,EAAE,EAAE,IAAI;QACR,OAAO,EAAE,GAAG,UAAU,aAAa,UAAU,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,yCAAyC;QACvG,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC;KACvB,CAAC;AACJ,CAAC"}
@@ -0,0 +1,3 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ export declare function extractPdf(bytes: Buffer): Promise<ExtractResult>;
3
+ //# sourceMappingURL=extract-pdf.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extract-pdf.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-pdf.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAwD5D,wBAAsB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,OAAO,CAAC,aAAa,CAAC,CAmGtE"}
@@ -0,0 +1,176 @@
1
+ import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
2
+ /**
3
+ * Extract a PDF's TEXT LAYER page by page with pdf.js (`pdfjs-dist`,
4
+ * Mozilla's maintained renderer — chosen over the unmaintained thin wrappers
5
+ * around it). The LEGACY build is the one supported under Node; it is loaded
6
+ * lazily (and once) because it is a heavyweight module most deployments only
7
+ * need after the first PDF read.
8
+ *
9
+ * Layout heuristic: text items on one line are joined with single spaces; a
10
+ * new line starts when pdf.js flags an EOL or the item's Y position jumps.
11
+ * A PDF with NO text layer (a scan) extracts to just the `[page N]` markers,
12
+ * and the summary says "no text layer (scanned document?)" — no OCR in v1.
13
+ */
14
+ /**
15
+ * How much DECODED text one PDF may yield before extraction gives up.
16
+ *
17
+ * The raw-size cap below bounds what arrives; it does not bound what comes
18
+ * out. PDF text lives in compressed streams, so a file comfortably under
19
+ * 50 MB can decode to far more than that, and every character of it is held
20
+ * in `lines` until the extraction returns. This bound is the decoded
21
+ * counterpart, checked as the text accumulates rather than after.
22
+ */
23
+ const MAX_PDF_TEXT_CHARS = 20 * 1024 * 1024; // 20M chars of extracted text
24
+ /**
25
+ * How many pages one PDF may have before extraction gives up.
26
+ *
27
+ * The decoded-text bound does not cover a document whose cost is its PAGE
28
+ * COUNT rather than its prose: every page costs a `getPage`, a
29
+ * `getTextContent` and a retained `[page N]` marker even when it holds no
30
+ * text at all, so a file declaring hundreds of thousands of empty pages spends
31
+ * minutes and megabytes without ever tripping a character budget. Real
32
+ * documents do not come close — a 2,000-page manual is an outlier.
33
+ */
34
+ const MAX_PDF_PAGES = 10_000;
35
+ /**
36
+ * How many text items one PAGE may hold. Items arrive through
37
+ * `streamTextContent` in small chunks (~100 items each), so this bound — like
38
+ * the character budget — fires while the page is still streaming, not after
39
+ * it has materialized. It exists because item COUNT is its own cost: each
40
+ * item is a retained heap object, and a page of empty-string items would
41
+ * never trip the character budget.
42
+ */
43
+ const MAX_PDF_ITEMS_PER_PAGE = 200_000;
44
+ /** The typed failure both decoded-text bounds return. */
45
+ function overBudget() {
46
+ return {
47
+ ok: false,
48
+ message: `could not be extracted as a PDF (its text decodes to over ${MAX_PDF_TEXT_CHARS} characters — over the extraction limit)`,
49
+ };
50
+ }
51
+ export async function extractPdf(bytes) {
52
+ // The same bounded-read guard the zip-based extractors apply per part: a
53
+ // PDF has no compressed container to pre-scan, so the bound is simply the
54
+ // file's raw size, checked before pdf.js parses anything.
55
+ if (bytes.length > MAX_DOC_PART_BYTES) {
56
+ return {
57
+ ok: false,
58
+ message: `could not be extracted as a PDF (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
59
+ };
60
+ }
61
+ let doc;
62
+ try {
63
+ doc = await openPdf(bytes);
64
+ }
65
+ catch (err) {
66
+ return { ok: false, message: `could not be parsed as a PDF (${err.message})` };
67
+ }
68
+ try {
69
+ if (doc.numPages > MAX_PDF_PAGES) {
70
+ return {
71
+ ok: false,
72
+ message: `could not be extracted as a PDF (it declares ${doc.numPages} pages — over the ${MAX_PDF_PAGES}-page extraction limit)`,
73
+ };
74
+ }
75
+ const lines = [];
76
+ let textChars = 0;
77
+ let anyText = false;
78
+ for (let n = 1; n <= doc.numPages; n++) {
79
+ lines.push(`[page ${n}]`);
80
+ // The marker is retained text like any other line: a document whose cost
81
+ // is its page count must reach the same bound as one whose cost is prose.
82
+ textChars += n.toString().length + 8;
83
+ if (textChars > MAX_PDF_TEXT_CHARS)
84
+ return overBudget();
85
+ const page = await doc.getPage(n);
86
+ // `streamTextContent` delivers the page's items in small chunks (~100
87
+ // items each, `getTextContent` is just this stream materialized), so
88
+ // both budgets fire WHILE the page streams: a crafted single page can
89
+ // no longer build its whole item array before a bound trips. Once one
90
+ // does, the reader is cancelled and pdf.js stops producing.
91
+ const reader = page.streamTextContent().getReader();
92
+ let pageItems = 0;
93
+ let line = '';
94
+ let lastY;
95
+ const flush = () => {
96
+ if (line.trim() !== '') {
97
+ lines.push(line);
98
+ anyText = true;
99
+ // The '\n' the final join emits for this line is retained text too.
100
+ textChars += 1;
101
+ }
102
+ line = '';
103
+ };
104
+ for (;;) {
105
+ const { done, value: chunk } = await reader.read();
106
+ if (done)
107
+ break;
108
+ pageItems += chunk.items.length;
109
+ if (pageItems > MAX_PDF_ITEMS_PER_PAGE) {
110
+ await reader.cancel().catch(() => undefined);
111
+ page.cleanup();
112
+ return {
113
+ ok: false,
114
+ message: `could not be extracted as a PDF (page ${n} holds more than ${MAX_PDF_ITEMS_PER_PAGE} text items — over the extraction limit)`,
115
+ };
116
+ }
117
+ for (const item of chunk.items) {
118
+ if (!('str' in item))
119
+ continue; // marked-content item — no text
120
+ const y = item.transform?.[5];
121
+ // Y-position jump = new visual line (1pt tolerance for kerning wobble).
122
+ if (typeof y === 'number') {
123
+ if (lastY !== undefined && Math.abs(y - lastY) > 1)
124
+ flush();
125
+ lastY = y;
126
+ }
127
+ if (item.str !== '') {
128
+ // COUNTED before it is kept — the join space included: the bound
129
+ // exists to stop the decoded text from accumulating, so it must
130
+ // fire mid-page and cover every character the result will hold.
131
+ textChars += item.str.length + (line === '' ? 0 : 1);
132
+ if (textChars > MAX_PDF_TEXT_CHARS) {
133
+ await reader.cancel().catch(() => undefined);
134
+ return overBudget();
135
+ }
136
+ line += (line === '' ? '' : ' ') + item.str;
137
+ }
138
+ if (item.hasEOL)
139
+ flush();
140
+ }
141
+ }
142
+ flush();
143
+ page.cleanup();
144
+ }
145
+ const pages = `${doc.numPages} page${doc.numPages === 1 ? '' : 's'}`;
146
+ return anyText
147
+ ? { ok: true, summary: `${pages}; layout, images and formatting omitted`, text: lines.join('\n') }
148
+ : { ok: true, summary: `${pages}; no text layer (scanned document?)`, text: lines.join('\n') };
149
+ }
150
+ catch (err) {
151
+ return { ok: false, message: `could not extract the PDF's text (${err.message})` };
152
+ }
153
+ finally {
154
+ await doc.destroy();
155
+ }
156
+ }
157
+ let pdfjsPromise;
158
+ async function openPdf(bytes) {
159
+ pdfjsPromise ??= import('pdfjs-dist/legacy/build/pdf.mjs').catch((err) => {
160
+ // A FAILED load must not be memoized: left in place, the rejected promise
161
+ // would answer every later read and disable PDF extraction for the whole
162
+ // process. Reset so the next read retries the import.
163
+ pdfjsPromise = undefined;
164
+ throw err;
165
+ });
166
+ const { getDocument } = await pdfjsPromise;
167
+ return getDocument({
168
+ // Copy into a fresh Uint8Array: pdf.js TRANSFERS the buffer it is given
169
+ // (detaching it), and the caller's Buffer must stay usable for hashing.
170
+ data: new Uint8Array(bytes),
171
+ // Server side: no font rendering — text content is all we consume.
172
+ disableFontFace: true,
173
+ useSystemFonts: true,
174
+ }).promise;
175
+ }
176
+ //# sourceMappingURL=extract-pdf.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extract-pdf.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-pdf.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,kBAAkB,EAAE,MAAM,iBAAiB,CAAC;AAErD;;;;;;;;;;;GAWG;AACH;;;;;;;;GAQG;AACH,MAAM,kBAAkB,GAAG,EAAE,GAAG,IAAI,GAAG,IAAI,CAAC,CAAC,8BAA8B;AAE3E;;;;;;;;;GASG;AACH,MAAM,aAAa,GAAG,MAAM,CAAC;AAE7B;;;;;;;GAOG;AACH,MAAM,sBAAsB,GAAG,OAAO,CAAC;AAEvC,yDAAyD;AACzD,SAAS,UAAU;IACjB,OAAO;QACL,EAAE,EAAE,KAAK;QACT,OAAO,EAAE,6DAA6D,kBAAkB,0CAA0C;KACnI,CAAC;AACJ,CAAC;AAED,MAAM,CAAC,KAAK,UAAU,UAAU,CAAC,KAAa;IAC5C,yEAAyE;IACzE,0EAA0E;IAC1E,0DAA0D;IAC1D,IAAI,KAAK,CAAC,MAAM,GAAG,kBAAkB,EAAE,CAAC;QACtC,OAAO;YACL,EAAE,EAAE,KAAK;YACT,OAAO,EAAE,gDAAgD,KAAK,CAAC,MAAM,qBAAqB,kBAAkB,iCAAiC;SAC9I,CAAC;IACJ,CAAC;IACD,IAAI,GAAwC,CAAC;IAC7C,IAAI,CAAC;QACH,GAAG,GAAG,MAAM,OAAO,CAAC,KAAK,CAAC,CAAC;IAC7B,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,iCAAkC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAC5F,CAAC;IACD,IAAI,CAAC;QACH,IAAI,GAAG,CAAC,QAAQ,GAAG,aAAa,EAAE,CAAC;YACjC,OAAO;gBACL,EAAE,EAAE,KAAK;gBACT,OAAO,EAAE,gDAAgD,GAAG,CAAC,QAAQ,qBAAqB,aAAa,yBAAyB;aACjI,CAAC;QACJ,CAAC;QACD,MAAM,KAAK,GAAa,EAAE,CAAC;QAC3B,IAAI,SAAS,GAAG,CAAC,CAAC;QAClB,IAAI,OAAO,GAAG,KAAK,CAAC;QACpB,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,IAAI,GAAG,CAAC,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YACvC,KAAK,CAAC,IAAI,CAAC,SAAS,CAAC,GAAG,CAAC,CAAC;YAC1B,yEAAyE;YACzE,0EAA0E;YAC1E,SAAS,IAAI,CAAC,CAAC,QAAQ,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC;YACrC,IAAI,SAAS,GAAG,kBAAkB;gBAAE,OAAO,UAAU,EAAE,CAAC;YACxD,MAAM,IAAI,GAAG,MAAM,GAAG,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC;YAClC,sEAAsE;YACtE,qEAAqE;YACrE,sEAAsE;YACtE,sEAAsE;YACtE,4DAA4D;YAC5D,MAAM,MAAM,GACV,IAAI,CAAC,iBAAiB,EACvB,CAAC,SAAS,EAAE,CAAC;YACd,IAAI,SAAS,GAAG,CAAC,CAAC;YAClB,IAAI,IAAI,GAAG,EAAE,CAAC;YACd,IAAI,KAAyB,CAAC;YAC9B,MAAM,KAAK,GAAG,GAAS,EAAE;gBACvB,IAAI,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE,EAAE,CAAC;oBACvB,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;oBACjB,OAAO,GAAG,IAAI,CAAC;oBACf,oEAAoE;oBACpE,SAAS,IAAI,CAAC,CAAC;gBACjB,CAAC;gBACD,IAAI,GAAG,EAAE,CAAC;YACZ,CAAC,CAAC;YACF,SAAS,CAAC;gBACR,MAAM,EAAE,IAAI,EAAE,KAAK,EAAE,KAAK,EAAE,GAAG,MAAM,MAAM,CAAC,IAAI,EAAE,CAAC;gBACnD,IAAI,IAAI;oBAAE,MAAM;gBAChB,SAAS,IAAI,KAAK,CAAC,KAAK,CAAC,MAAM,CAAC;gBAChC,IAAI,SAAS,GAAG,sBAAsB,EAAE,CAAC;oBACvC,MAAM,MAAM,CAAC,MAAM,EAAE,CAAC,KAAK,CAAC,GAAG,EAAE,CAAC,SAAS,CAAC,CAAC;oBAC7C,IAAI,CAAC,OAAO,EAAE,CAAC;oBACf,OAAO;wBACL,EAAE,EAAE,KAAK;wBACT,OAAO,EAAE,yCAAyC,CAAC,oBAAoB,sBAAsB,0CAA0C;qBACxI,CAAC;gBACJ,CAAC;gBACD,KAAK,MAAM,IAAI,IAAI,KAAK,CAAC,KAAK,EAAE,CAAC;oBAC/B,IAAI,CAAC,CAAC,KAAK,IAAI,IAAI,CAAC;wBAAE,SAAS,CAAC,gCAAgC;oBAChE,MAAM,CAAC,GAAG,IAAI,CAAC,SAAS,EAAE,CAAC,CAAC,CAAC,CAAC;oBAC9B,wEAAwE;oBACxE,IAAI,OAAO,CAAC,KAAK,QAAQ,EAAE,CAAC;wBAC1B,IAAI,KAAK,KAAK,SAAS,IAAI,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,KAAK,CAAC,GAAG,CAAC;4BAAE,KAAK,EAAE,CAAC;wBAC5D,KAAK,GAAG,CAAC,CAAC;oBACZ,CAAC;oBACD,IAAI,IAAI,CAAC,GAAG,KAAK,EAAE,EAAE,CAAC;wBACpB,iEAAiE;wBACjE,gEAAgE;wBAChE,gEAAgE;wBAChE,SAAS,IAAI,IAAI,CAAC,GAAG,CAAC,MAAM,GAAG,CAAC,IAAI,KAAK,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;wBACrD,IAAI,SAAS,GAAG,kBAAkB,EAAE,CAAC;4BACnC,MAAM,MAAM,CAAC,MAAM,EAAE,CAAC,KAAK,CAAC,GAAG,EAAE,CAAC,SAAS,CAAC,CAAC;4BAC7C,OAAO,UAAU,EAAE,CAAC;wBACtB,CAAC;wBACD,IAAI,IAAI,CAAC,IAAI,KAAK,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC;oBAC9C,CAAC;oBACD,IAAI,IAAI,CAAC,MAAM;wBAAE,KAAK,EAAE,CAAC;gBAC3B,CAAC;YACH,CAAC;YACD,KAAK,EAAE,CAAC;YACR,IAAI,CAAC,OAAO,EAAE,CAAC;QACjB,CAAC;QACD,MAAM,KAAK,GAAG,GAAG,GAAG,CAAC,QAAQ,QAAQ,GAAG,CAAC,QAAQ,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC;QACrE,OAAO,OAAO;YACZ,CAAC,CAAC,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,GAAG,KAAK,yCAAyC,EAAE,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE;YAClG,CAAC,CAAC,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,GAAG,KAAK,qCAAqC,EAAE,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;IACnG,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,qCAAsC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAChG,CAAC;YAAS,CAAC;QACT,MAAM,GAAG,CAAC,OAAO,EAAE,CAAC;IACtB,CAAC;AACH,CAAC;AAGD,IAAI,YAAwC,CAAC;AAE7C,KAAK,UAAU,OAAO,CAAC,KAAa;IAClC,YAAY,KAAK,MAAM,CAAC,iCAAiC,CAAC,CAAC,KAAK,CAAC,CAAC,GAAY,EAAE,EAAE;QAChF,0EAA0E;QAC1E,yEAAyE;QACzE,sDAAsD;QACtD,YAAY,GAAG,SAAS,CAAC;QACzB,MAAM,GAAG,CAAC;IACZ,CAAC,CAAC,CAAC;IACH,MAAM,EAAE,WAAW,EAAE,GAAG,MAAM,YAAY,CAAC;IAC3C,OAAO,WAAW,CAAC;QACjB,wEAAwE;QACxE,wEAAwE;QACxE,IAAI,EAAE,IAAI,UAAU,CAAC,KAAK,CAAC;QAC3B,mEAAmE;QACnE,eAAe,EAAE,IAAI;QACrB,cAAc,EAAE,IAAI;KACrB,CAAC,CAAC,OAAO,CAAC;AACb,CAAC"}
@@ -0,0 +1,37 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ /**
3
+ * Extract the text of a `.pptx` (PowerPoint) deck.
4
+ *
5
+ * Slides live at `ppt/slides/slideN.xml`; each is emitted under a `[slide N]`
6
+ * marker line, in the PRESENTATION's slide order — `ppt/presentation.xml`'s
7
+ * `<p:sldIdLst>`, resolved through its rels part (see
8
+ * `slideOrderFromPresentation`) — with numeric filename order as the fallback
9
+ * when the package has no readable list. Speaker notes
10
+ * follow their slide under `[slide N notes]` when non-empty. A slide's notes
11
+ * part is resolved through the slide's RELATIONSHIPS part (the `_rels` twin of
12
+ * the slide part's own NAME, relationship type ending `notesSlide`) — the
13
+ * package is free to number notes parts differently from slides — with the
14
+ * `notesSlideN.xml` convention as the fallback when the slide has no rels
15
+ * part at all. Within a slide, each `<a:p>` paragraph is a line;
16
+ * `<a:t>` runs concatenate with no separator (runs split mid-word).
17
+ *
18
+ * Bounded: every entry's DECLARED uncompressed size is checked before
19
+ * inflation (see `zipEntryOversize`), and the parts read for one deck may not
20
+ * exceed `MAX_DOC_TOTAL_BYTES` in total — over either bound is a typed
21
+ * failure, never an allocation.
22
+ */
23
+ export declare function extractPptx(bytes: Buffer): ExtractResult;
24
+ /**
25
+ * The Target of the first `notesSlide`-typed Relationship in a rels part, or
26
+ * undefined.
27
+ *
28
+ * Read by the parser: matched on the element's LOCAL name, so a producer that
29
+ * binds the relationships namespace to a prefix (`<r:Relationship r:Type=…>`)
30
+ * is read like any other — and a `<Relationship>`-looking fragment written
31
+ * inside a COMMENT or a CDATA section is text, not live metadata pointing the
32
+ * notes lookup at a part of its author's choosing.
33
+ */
34
+ export declare function notesTargetFromRels(relsXml: string): string | undefined;
35
+ /** Resolve an OPC relationship Target against the part's base directory. */
36
+ export declare function resolveRelTarget(baseDir: string, target: string): string;
37
+ //# sourceMappingURL=extract-pptx.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extract-pptx.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-pptx.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAY5D;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CA8DxD;AA0KD;;;;;;;;;GASG;AACH,wBAAgB,mBAAmB,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAWvE;AAED,4EAA4E;AAC5E,wBAAgB,gBAAgB,CAAC,OAAO,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM,GAAG,MAAM,CAWxE"}
@@ -0,0 +1,288 @@
1
+ import AdmZip from 'adm-zip';
2
+ import { MAX_DOC_TOTAL_BYTES, attrByLocalName, decodeXmlEntities, localBlocks, localElementBlocks, localName, paragraphRunText, zipEntryOversize, } from './ooxml-text.js';
3
+ /**
4
+ * Extract the text of a `.pptx` (PowerPoint) deck.
5
+ *
6
+ * Slides live at `ppt/slides/slideN.xml`; each is emitted under a `[slide N]`
7
+ * marker line, in the PRESENTATION's slide order — `ppt/presentation.xml`'s
8
+ * `<p:sldIdLst>`, resolved through its rels part (see
9
+ * `slideOrderFromPresentation`) — with numeric filename order as the fallback
10
+ * when the package has no readable list. Speaker notes
11
+ * follow their slide under `[slide N notes]` when non-empty. A slide's notes
12
+ * part is resolved through the slide's RELATIONSHIPS part (the `_rels` twin of
13
+ * the slide part's own NAME, relationship type ending `notesSlide`) — the
14
+ * package is free to number notes parts differently from slides — with the
15
+ * `notesSlideN.xml` convention as the fallback when the slide has no rels
16
+ * part at all. Within a slide, each `<a:p>` paragraph is a line;
17
+ * `<a:t>` runs concatenate with no separator (runs split mid-word).
18
+ *
19
+ * Bounded: every entry's DECLARED uncompressed size is checked before
20
+ * inflation (see `zipEntryOversize`), and the parts read for one deck may not
21
+ * exceed `MAX_DOC_TOTAL_BYTES` in total — over either bound is a typed
22
+ * failure, never an allocation.
23
+ */
24
+ export function extractPptx(bytes) {
25
+ let slides;
26
+ let notesBySlide;
27
+ let presOrder;
28
+ try {
29
+ const zip = new AdmZip(bytes);
30
+ const budget = { remaining: MAX_DOC_TOTAL_BYTES };
31
+ slides = collectNumbered(zip, /^ppt\/slides\/slide(\d+)\.xml$/, budget);
32
+ notesBySlide = collectNotes(zip, slides, budget);
33
+ presOrder = slideOrderFromPresentation(zip, budget);
34
+ }
35
+ catch (err) {
36
+ return { ok: false, message: `could not be parsed as a .pptx (${err.message})` };
37
+ }
38
+ if (slides.size === 0) {
39
+ return { ok: false, message: 'could not be parsed as a .pptx (no ppt/slides/slideN.xml inside the archive)' };
40
+ }
41
+ // Emission order: the presentation's own slide list when it resolves to
42
+ // selected parts, numeric filename order otherwise (parts the list does not
43
+ // name follow it, in filename order). When the LIST orders the deck, the
44
+ // markers number POSITIONS in it — what a viewer calls slide 1 — because a
45
+ // reordered deck's part filenames no longer mean anything positional.
46
+ const byFilename = [...slides.keys()].sort((a, b) => a - b);
47
+ let order = byFilename;
48
+ let positional = false;
49
+ if (presOrder !== undefined) {
50
+ const numByName = new Map();
51
+ for (const [n, slide] of slides)
52
+ numByName.set(slide.name, n);
53
+ const seen = new Set();
54
+ const fromList = [];
55
+ for (const name of presOrder) {
56
+ const n = numByName.get(name);
57
+ if (n !== undefined && !seen.has(n)) {
58
+ seen.add(n);
59
+ fromList.push(n);
60
+ }
61
+ }
62
+ if (fromList.length > 0) {
63
+ order = [...fromList, ...byFilename.filter((n) => !seen.has(n))];
64
+ positional = true;
65
+ }
66
+ }
67
+ const lines = [];
68
+ let anyNotes = false;
69
+ order.forEach((n, i) => {
70
+ const label = positional ? i + 1 : n;
71
+ lines.push(`[slide ${label}]`);
72
+ lines.push(...paragraphLines(slides.get(n).xml));
73
+ const notesXml = notesBySlide.get(n);
74
+ const noteLines = notesXml !== undefined ? paragraphLines(notesXml) : [];
75
+ if (noteLines.length > 0) {
76
+ anyNotes = true;
77
+ lines.push(`[slide ${label} notes]`);
78
+ lines.push(...noteLines);
79
+ }
80
+ });
81
+ return {
82
+ ok: true,
83
+ summary: `${slides.size} slide${slides.size === 1 ? '' : 's'}${anyNotes ? ' + notes' : ''}; layout, images and formatting omitted`,
84
+ text: lines.join('\n'),
85
+ };
86
+ }
87
+ /** Non-empty paragraph texts of one slide/notes part, in document order. */
88
+ function paragraphLines(xml) {
89
+ const out = [];
90
+ for (const p of localBlocks(xml, 'p')) {
91
+ const text = paragraphRunText(p, 't');
92
+ if (text.trim() !== '')
93
+ out.push(text);
94
+ }
95
+ return out;
96
+ }
97
+ /** `entry`'s bytes as UTF-8, after the per-part and aggregate bounds. Throws over either. */
98
+ function readEntryBounded(entry, budget) {
99
+ const oversize = zipEntryOversize(entry);
100
+ if (oversize)
101
+ throw new Error(oversize);
102
+ budget.remaining -= entry.header.size;
103
+ if (budget.remaining < 0) {
104
+ throw new Error(`the archive's parts exceed the ${MAX_DOC_TOTAL_BYTES}-byte (200 MB) total extraction limit`);
105
+ }
106
+ return entry.getData().toString('utf8');
107
+ }
108
+ /**
109
+ * Entries matching `re` (capture 1 = number), decoded as UTF-8, keyed by
110
+ * number. Two part names can parse to the SAME number (`slide1.xml` and
111
+ * `slide01.xml`); the winner is the FIRST in ascending part-name order —
112
+ * deterministic regardless of zip entry order, and the same policy as the
113
+ * browser twin (`pptxOutline.ts`), so viewer and `read_file` agree. Losing
114
+ * duplicates are never inflated (no budget charge).
115
+ *
116
+ * The winner's NAME rides along with its bytes because everything else about
117
+ * a slide hangs off the part name, not the number: see `collectNotes`.
118
+ */
119
+ function collectNumbered(zip, re, budget) {
120
+ const matched = [];
121
+ for (const entry of zip.getEntries()) {
122
+ const m = re.exec(entry.entryName);
123
+ if (!m)
124
+ continue;
125
+ // A crafted name can spell a number past 2^53 (or Infinity): distinct
126
+ // parts would collide in the map and one would silently vanish.
127
+ const n = parseInt(m[1], 10);
128
+ if (Number.isSafeInteger(n))
129
+ matched.push([n, entry.entryName, entry]);
130
+ }
131
+ // A zip may list the SAME part name twice. Sorting by name leaves those two
132
+ // in archive order, so which one wins depends on how the file was written —
133
+ // and two archives with identical parts would extract differently. A part
134
+ // claimed twice is not a part this reader can resolve, so it is dropped.
135
+ const claims = new Map();
136
+ for (const [, name] of matched)
137
+ claims.set(name, (claims.get(name) ?? 0) + 1);
138
+ const unique = matched.filter(([, name]) => claims.get(name) === 1);
139
+ unique.sort((a, b) => (a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0));
140
+ const out = new Map();
141
+ for (const [n, name, entry] of unique) {
142
+ if (!out.has(n))
143
+ out.set(n, { name, xml: readEntryBounded(entry, budget) });
144
+ }
145
+ return out;
146
+ }
147
+ const NOTES_REL_TYPE_SUFFIX = '/notesSlide';
148
+ const SLIDE_REL_TYPE_SUFFIX = '/slide';
149
+ /**
150
+ * The value of the attribute whose LOCAL name is `want` AND that carries a
151
+ * namespace prefix. `<p:sldId>` holds both its own `id` and the relationship
152
+ * reference `r:id`; plain local-name matching answers with whichever is
153
+ * written first, so the relationship id must be the PREFIXED one.
154
+ */
155
+ function prefixedAttrByLocalName(attributes, want) {
156
+ for (const [key, value] of Object.entries(attributes)) {
157
+ if (key === 'xmlns' || key.startsWith('xmlns:'))
158
+ continue;
159
+ if (key.includes(':') && localName(key) === want)
160
+ return value;
161
+ }
162
+ return undefined;
163
+ }
164
+ /**
165
+ * Slide part names in PRESENTATION order: `ppt/presentation.xml`'s
166
+ * `<p:sldIdLst>` entries, each `r:id` resolved through the presentation's own
167
+ * rels part — or undefined when the package has no readable list. Reordering
168
+ * slides in PowerPoint rewrites the sldIdLst and leaves the part names alone,
169
+ * so `slide1.xml` need not be the deck's first slide; the numeric filename
170
+ * sort is only the fallback for packages without the list.
171
+ */
172
+ function slideOrderFromPresentation(zip, budget) {
173
+ const rels = zip.getEntry('ppt/_rels/presentation.xml.rels');
174
+ const pres = zip.getEntry('ppt/presentation.xml');
175
+ if (!rels || !pres)
176
+ return undefined;
177
+ const targetById = new Map();
178
+ for (const rel of localElementBlocks(readEntryBounded(rels, budget), ['Relationship'])) {
179
+ const type = attrByLocalName(rel.attributes, 'Type');
180
+ if (type === undefined || !type.endsWith(SLIDE_REL_TYPE_SUFFIX))
181
+ continue;
182
+ const id = attrByLocalName(rel.attributes, 'Id');
183
+ // Targets are RAW in the rels (see `notesTargetFromRels`) and relative to
184
+ // the presentation part's directory.
185
+ const target = attrByLocalName(rel.attributes, 'Target');
186
+ if (id !== undefined && target !== undefined) {
187
+ targetById.set(id, resolveRelTarget('ppt', decodeXmlEntities(target)));
188
+ }
189
+ }
190
+ if (targetById.size === 0)
191
+ return undefined;
192
+ const list = localBlocks(readEntryBounded(pres, budget), 'sldIdLst')[0];
193
+ if (list === undefined)
194
+ return undefined;
195
+ const order = [];
196
+ for (const sld of localElementBlocks(list, ['sldId'])) {
197
+ const rid = prefixedAttrByLocalName(sld.attributes, 'id');
198
+ const target = rid !== undefined ? targetById.get(rid) : undefined;
199
+ if (target !== undefined)
200
+ order.push(target);
201
+ }
202
+ return order.length > 0 ? order : undefined;
203
+ }
204
+ /**
205
+ * The OPC relationships part of `partName` — `dir/_rels/base.rels`. Derived
206
+ * from the part NAME, never from the slide number: when a deck ships both
207
+ * `slide1.xml` and `slide01.xml`, the name-ordering rule picks `slide01.xml`,
208
+ * whose rels part is `slide01.xml.rels`. Rebuilding the path from the number
209
+ * asked for `slide1.xml.rels` — a part belonging to the OTHER file — and so
210
+ * either lost that slide's speaker notes or attached the losing part's.
211
+ */
212
+ function relsPartName(partName) {
213
+ const cut = partName.lastIndexOf('/');
214
+ return `${partName.slice(0, cut)}/_rels/${partName.slice(cut + 1)}.rels`;
215
+ }
216
+ /**
217
+ * The conventional notes part for a slide part — `slide01.xml` →
218
+ * `notesSlide01.xml`. Derived from the name for the same reason as the rels
219
+ * path, so a zero-padded deck's fallback lands on the matching notes part.
220
+ */
221
+ function conventionalNotesPart(partName) {
222
+ const base = partName.slice(partName.lastIndexOf('/') + 1);
223
+ return `ppt/notesSlides/notes${base[0].toUpperCase()}${base.slice(1)}`;
224
+ }
225
+ /**
226
+ * Slide number → its notes part's XML, resolved through each slide's `.rels`
227
+ * part; the conventional-name fallback ONLY for a slide without a rels part.
228
+ * Both paths come from the SELECTED part's name (see `relsPartName`).
229
+ */
230
+ function collectNotes(zip, slides, budget) {
231
+ const out = new Map();
232
+ for (const [n, slide] of slides) {
233
+ const rels = zip.getEntry(relsPartName(slide.name));
234
+ let notesPart;
235
+ if (rels) {
236
+ const target = notesTargetFromRels(readEntryBounded(rels, budget));
237
+ notesPart = target !== undefined ? resolveRelTarget('ppt/slides', target) : undefined;
238
+ }
239
+ else {
240
+ notesPart = conventionalNotesPart(slide.name);
241
+ }
242
+ if (notesPart === undefined)
243
+ continue;
244
+ const entry = zip.getEntry(notesPart);
245
+ if (entry)
246
+ out.set(n, readEntryBounded(entry, budget));
247
+ }
248
+ return out;
249
+ }
250
+ /**
251
+ * The Target of the first `notesSlide`-typed Relationship in a rels part, or
252
+ * undefined.
253
+ *
254
+ * Read by the parser: matched on the element's LOCAL name, so a producer that
255
+ * binds the relationships namespace to a prefix (`<r:Relationship r:Type=…>`)
256
+ * is read like any other — and a `<Relationship>`-looking fragment written
257
+ * inside a COMMENT or a CDATA section is text, not live metadata pointing the
258
+ * notes lookup at a part of its author's choosing.
259
+ */
260
+ export function notesTargetFromRels(relsXml) {
261
+ for (const rel of localElementBlocks(relsXml, ['Relationship'])) {
262
+ const type = attrByLocalName(rel.attributes, 'Type');
263
+ if (type !== undefined && type.endsWith(NOTES_REL_TYPE_SUFFIX)) {
264
+ // Attribute values are RAW here (the block reader does not decode), and a
265
+ // part name may legally contain `&`, written `&amp;` in the rels.
266
+ const target = attrByLocalName(rel.attributes, 'Target');
267
+ return target !== undefined ? decodeXmlEntities(target) : undefined;
268
+ }
269
+ }
270
+ return undefined;
271
+ }
272
+ /** Resolve an OPC relationship Target against the part's base directory. */
273
+ export function resolveRelTarget(baseDir, target) {
274
+ const parts = target.startsWith('/')
275
+ ? target.slice(1).split('/')
276
+ : [...baseDir.split('/'), ...target.split('/')];
277
+ const out = [];
278
+ for (const p of parts) {
279
+ if (p === '' || p === '.')
280
+ continue;
281
+ if (p === '..')
282
+ out.pop();
283
+ else
284
+ out.push(p);
285
+ }
286
+ return out.join('/');
287
+ }
288
+ //# sourceMappingURL=extract-pptx.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extract-pptx.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-pptx.ts"],"names":[],"mappings":"AAAA,OAAO,MAAM,MAAM,SAAS,CAAC;AAE7B,OAAO,EACL,mBAAmB,EACnB,eAAe,EACf,iBAAiB,EACjB,WAAW,EACX,kBAAkB,EAClB,SAAS,EACT,gBAAgB,EAChB,gBAAgB,GACjB,MAAM,iBAAiB,CAAC;AAEzB;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,UAAU,WAAW,CAAC,KAAa;IACvC,IAAI,MAA8B,CAAC;IACnC,IAAI,YAAiC,CAAC;IACtC,IAAI,SAA+B,CAAC;IACpC,IAAI,CAAC;QACH,MAAM,GAAG,GAAG,IAAI,MAAM,CAAC,KAAK,CAAC,CAAC;QAC9B,MAAM,MAAM,GAAG,EAAE,SAAS,EAAE,mBAAmB,EAAE,CAAC;QAClD,MAAM,GAAG,eAAe,CAAC,GAAG,EAAE,gCAAgC,EAAE,MAAM,CAAC,CAAC;QACxE,YAAY,GAAG,YAAY,CAAC,GAAG,EAAE,MAAM,EAAE,MAAM,CAAC,CAAC;QACjD,SAAS,GAAG,0BAA0B,CAAC,GAAG,EAAE,MAAM,CAAC,CAAC;IACtD,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,mCAAoC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAC9F,CAAC;IACD,IAAI,MAAM,CAAC,IAAI,KAAK,CAAC,EAAE,CAAC;QACtB,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,8EAA8E,EAAE,CAAC;IAChH,CAAC;IAED,wEAAwE;IACxE,4EAA4E;IAC5E,yEAAyE;IACzE,2EAA2E;IAC3E,sEAAsE;IACtE,MAAM,UAAU,GAAG,CAAC,GAAG,MAAM,CAAC,IAAI,EAAE,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IAC5D,IAAI,KAAK,GAAG,UAAU,CAAC;IACvB,IAAI,UAAU,GAAG,KAAK,CAAC;IACvB,IAAI,SAAS,KAAK,SAAS,EAAE,CAAC;QAC5B,MAAM,SAAS,GAAG,IAAI,GAAG,EAAkB,CAAC;QAC5C,KAAK,MAAM,CAAC,CAAC,EAAE,KAAK,CAAC,IAAI,MAAM;YAAE,SAAS,CAAC,GAAG,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC;QAC9D,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;QAC/B,MAAM,QAAQ,GAAa,EAAE,CAAC;QAC9B,KAAK,MAAM,IAAI,IAAI,SAAS,EAAE,CAAC;YAC7B,MAAM,CAAC,GAAG,SAAS,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;YAC9B,IAAI,CAAC,KAAK,SAAS,IAAI,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC;gBACpC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC;gBACZ,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;YACnB,CAAC;QACH,CAAC;QACD,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YACxB,KAAK,GAAG,CAAC,GAAG,QAAQ,EAAE,GAAG,UAAU,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;YACjE,UAAU,GAAG,IAAI,CAAC;QACpB,CAAC;IACH,CAAC;IAED,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,QAAQ,GAAG,KAAK,CAAC;IACrB,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE;QACrB,MAAM,KAAK,GAAG,UAAU,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;QACrC,KAAK,CAAC,IAAI,CAAC,UAAU,KAAK,GAAG,CAAC,CAAC;QAC/B,KAAK,CAAC,IAAI,CAAC,GAAG,cAAc,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,CAAE,CAAC,GAAG,CAAC,CAAC,CAAC;QAClD,MAAM,QAAQ,GAAG,YAAY,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC;QACrC,MAAM,SAAS,GAAG,QAAQ,KAAK,SAAS,CAAC,CAAC,CAAC,cAAc,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;QACzE,IAAI,SAAS,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YACzB,QAAQ,GAAG,IAAI,CAAC;YAChB,KAAK,CAAC,IAAI,CAAC,UAAU,KAAK,SAAS,CAAC,CAAC;YACrC,KAAK,CAAC,IAAI,CAAC,GAAG,SAAS,CAAC,CAAC;QAC3B,CAAC;IACH,CAAC,CAAC,CAAC;IACH,OAAO;QACL,EAAE,EAAE,IAAI;QACR,OAAO,EAAE,GAAG,MAAM,CAAC,IAAI,SAAS,MAAM,CAAC,IAAI,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,GAAG,QAAQ,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,EAAE,yCAAyC;QAClI,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC;KACvB,CAAC;AACJ,CAAC;AAED,4EAA4E;AAC5E,SAAS,cAAc,CAAC,GAAW;IACjC,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,MAAM,CAAC,IAAI,WAAW,CAAC,GAAG,EAAE,GAAG,CAAC,EAAE,CAAC;QACtC,MAAM,IAAI,GAAG,gBAAgB,CAAC,CAAC,EAAE,GAAG,CAAC,CAAC;QACtC,IAAI,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE;YAAE,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IACzC,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAOD,6FAA6F;AAC7F,SAAS,gBAAgB,CAAC,KAAuB,EAAE,MAAkB;IACnE,MAAM,QAAQ,GAAG,gBAAgB,CAAC,KAAK,CAAC,CAAC;IACzC,IAAI,QAAQ;QAAE,MAAM,IAAI,KAAK,CAAC,QAAQ,CAAC,CAAC;IACxC,MAAM,CAAC,SAAS,IAAI,KAAK,CAAC,MAAM,CAAC,IAAI,CAAC;IACtC,IAAI,MAAM,CAAC,SAAS,GAAG,CAAC,EAAE,CAAC;QACzB,MAAM,IAAI,KAAK,CAAC,kCAAkC,mBAAmB,uCAAuC,CAAC,CAAC;IAChH,CAAC;IACD,OAAO,KAAK,CAAC,OAAO,EAAE,CAAC,QAAQ,CAAC,MAAM,CAAC,CAAC;AAC1C,CAAC;AASD;;;;;;;;;;GAUG;AACH,SAAS,eAAe,CAAC,GAAW,EAAE,EAAU,EAAE,MAAkB;IAClE,MAAM,OAAO,GAA8C,EAAE,CAAC;IAC9D,KAAK,MAAM,KAAK,IAAI,GAAG,CAAC,UAAU,EAAE,EAAE,CAAC;QACrC,MAAM,CAAC,GAAG,EAAE,CAAC,IAAI,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC;QACnC,IAAI,CAAC,CAAC;YAAE,SAAS;QACjB,sEAAsE;QACtE,gEAAgE;QAChE,MAAM,CAAC,GAAG,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC;QAC7B,IAAI,MAAM,CAAC,aAAa,CAAC,CAAC,CAAC;YAAE,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,KAAK,CAAC,SAAS,EAAE,KAAK,CAAC,CAAC,CAAC;IACzE,CAAC;IACD,4EAA4E;IAC5E,4EAA4E;IAC5E,0EAA0E;IAC1E,yEAAyE;IACzE,MAAM,MAAM,GAAG,IAAI,GAAG,EAAkB,CAAC;IACzC,KAAK,MAAM,CAAC,EAAE,IAAI,CAAC,IAAI,OAAO;QAAE,MAAM,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IAC9E,MAAM,MAAM,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,IAAI,CAAC,EAAE,EAAE,CAAC,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC;IACpE,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAChE,MAAM,GAAG,GAAG,IAAI,GAAG,EAAqB,CAAC;IACzC,KAAK,MAAM,CAAC,CAAC,EAAE,IAAI,EAAE,KAAK,CAAC,IAAI,MAAM,EAAE,CAAC;QACtC,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC;YAAE,GAAG,CAAC,GAAG,CAAC,CAAC,EAAE,EAAE,IAAI,EAAE,GAAG,EAAE,gBAAgB,CAAC,KAAK,EAAE,MAAM,CAAC,EAAE,CAAC,CAAC;IAC9E,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED,MAAM,qBAAqB,GAAG,aAAa,CAAC;AAC5C,MAAM,qBAAqB,GAAG,QAAQ,CAAC;AAEvC;;;;;GAKG;AACH,SAAS,uBAAuB,CAAC,UAAkC,EAAE,IAAY;IAC/E,KAAK,MAAM,CAAC,GAAG,EAAE,KAAK,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,UAAU,CAAC,EAAE,CAAC;QACtD,IAAI,GAAG,KAAK,OAAO,IAAI,GAAG,CAAC,UAAU,CAAC,QAAQ,CAAC;YAAE,SAAS;QAC1D,IAAI,GAAG,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,SAAS,CAAC,GAAG,CAAC,KAAK,IAAI;YAAE,OAAO,KAAK,CAAC;IACjE,CAAC;IACD,OAAO,SAAS,CAAC;AACnB,CAAC;AAED;;;;;;;GAOG;AACH,SAAS,0BAA0B,CAAC,GAAW,EAAE,MAAkB;IACjE,MAAM,IAAI,GAAG,GAAG,CAAC,QAAQ,CAAC,iCAAiC,CAAC,CAAC;IAC7D,MAAM,IAAI,GAAG,GAAG,CAAC,QAAQ,CAAC,sBAAsB,CAAC,CAAC;IAClD,IAAI,CAAC,IAAI,IAAI,CAAC,IAAI;QAAE,OAAO,SAAS,CAAC;IACrC,MAAM,UAAU,GAAG,IAAI,GAAG,EAAkB,CAAC;IAC7C,KAAK,MAAM,GAAG,IAAI,kBAAkB,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,CAAC,EAAE,CAAC,cAAc,CAAC,CAAC,EAAE,CAAC;QACvF,MAAM,IAAI,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,MAAM,CAAC,CAAC;QACrD,IAAI,IAAI,KAAK,SAAS,IAAI,CAAC,IAAI,CAAC,QAAQ,CAAC,qBAAqB,CAAC;YAAE,SAAS;QAC1E,MAAM,EAAE,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,IAAI,CAAC,CAAC;QACjD,0EAA0E;QAC1E,qCAAqC;QACrC,MAAM,MAAM,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAC;QACzD,IAAI,EAAE,KAAK,SAAS,IAAI,MAAM,KAAK,SAAS,EAAE,CAAC;YAC7C,UAAU,CAAC,GAAG,CAAC,EAAE,EAAE,gBAAgB,CAAC,KAAK,EAAE,iBAAiB,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC;QACzE,CAAC;IACH,CAAC;IACD,IAAI,UAAU,CAAC,IAAI,KAAK,CAAC;QAAE,OAAO,SAAS,CAAC;IAC5C,MAAM,IAAI,GAAG,WAAW,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,CAAC,EAAE,UAAU,CAAC,CAAC,CAAC,CAAC,CAAC;IACxE,IAAI,IAAI,KAAK,SAAS;QAAE,OAAO,SAAS,CAAC;IACzC,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,KAAK,MAAM,GAAG,IAAI,kBAAkB,CAAC,IAAI,EAAE,CAAC,OAAO,CAAC,CAAC,EAAE,CAAC;QACtD,MAAM,GAAG,GAAG,uBAAuB,CAAC,GAAG,CAAC,UAAU,EAAE,IAAI,CAAC,CAAC;QAC1D,MAAM,MAAM,GAAG,GAAG,KAAK,SAAS,CAAC,CAAC,CAAC,UAAU,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC;QACnE,IAAI,MAAM,KAAK,SAAS;YAAE,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;IAC/C,CAAC;IACD,OAAO,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,SAAS,CAAC;AAC9C,CAAC;AAED;;;;;;;GAOG;AACH,SAAS,YAAY,CAAC,QAAgB;IACpC,MAAM,GAAG,GAAG,QAAQ,CAAC,WAAW,CAAC,GAAG,CAAC,CAAC;IACtC,OAAO,GAAG,QAAQ,CAAC,KAAK,CAAC,CAAC,EAAE,GAAG,CAAC,UAAU,QAAQ,CAAC,KAAK,CAAC,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC;AAC3E,CAAC;AAED;;;;GAIG;AACH,SAAS,qBAAqB,CAAC,QAAgB;IAC7C,MAAM,IAAI,GAAG,QAAQ,CAAC,KAAK,CAAC,QAAQ,CAAC,WAAW,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC;IAC3D,OAAO,wBAAwB,IAAI,CAAC,CAAC,CAAC,CAAC,WAAW,EAAE,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC,EAAE,CAAC;AACzE,CAAC;AAED;;;;GAIG;AACH,SAAS,YAAY,CAAC,GAAW,EAAE,MAA8B,EAAE,MAAkB;IACnF,MAAM,GAAG,GAAG,IAAI,GAAG,EAAkB,CAAC;IACtC,KAAK,MAAM,CAAC,CAAC,EAAE,KAAK,CAAC,IAAI,MAAM,EAAE,CAAC;QAChC,MAAM,IAAI,GAAG,GAAG,CAAC,QAAQ,CAAC,YAAY,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC;QACpD,IAAI,SAA6B,CAAC;QAClC,IAAI,IAAI,EAAE,CAAC;YACT,MAAM,MAAM,GAAG,mBAAmB,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,CAAC,CAAC,CAAC;YACnE,SAAS,GAAG,MAAM,KAAK,SAAS,CAAC,CAAC,CAAC,gBAAgB,CAAC,YAAY,EAAE,MAAM,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC;QACxF,CAAC;aAAM,CAAC;YACN,SAAS,GAAG,qBAAqB,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC;QAChD,CAAC;QACD,IAAI,SAAS,KAAK,SAAS;YAAE,SAAS;QACtC,MAAM,KAAK,GAAG,GAAG,CAAC,QAAQ,CAAC,SAAS,CAAC,CAAC;QACtC,IAAI,KAAK;YAAE,GAAG,CAAC,GAAG,CAAC,CAAC,EAAE,gBAAgB,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC,CAAC;IACzD,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;;;;;GASG;AACH,MAAM,UAAU,mBAAmB,CAAC,OAAe;IACjD,KAAK,MAAM,GAAG,IAAI,kBAAkB,CAAC,OAAO,EAAE,CAAC,cAAc,CAAC,CAAC,EAAE,CAAC;QAChE,MAAM,IAAI,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,MAAM,CAAC,CAAC;QACrD,IAAI,IAAI,KAAK,SAAS,IAAI,IAAI,CAAC,QAAQ,CAAC,qBAAqB,CAAC,EAAE,CAAC;YAC/D,0EAA0E;YAC1E,kEAAkE;YAClE,MAAM,MAAM,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAC;YACzD,OAAO,MAAM,KAAK,SAAS,CAAC,CAAC,CAAC,iBAAiB,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC;QACtE,CAAC;IACH,CAAC;IACD,OAAO,SAAS,CAAC;AACnB,CAAC;AAED,4EAA4E;AAC5E,MAAM,UAAU,gBAAgB,CAAC,OAAe,EAAE,MAAc;IAC9D,MAAM,KAAK,GAAG,MAAM,CAAC,UAAU,CAAC,GAAG,CAAC;QAClC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC;QAC5B,CAAC,CAAC,CAAC,GAAG,OAAO,CAAC,KAAK,CAAC,GAAG,CAAC,EAAE,GAAG,MAAM,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC;IAClD,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,MAAM,CAAC,IAAI,KAAK,EAAE,CAAC;QACtB,IAAI,CAAC,KAAK,EAAE,IAAI,CAAC,KAAK,GAAG;YAAE,SAAS;QACpC,IAAI,CAAC,KAAK,IAAI;YAAE,GAAG,CAAC,GAAG,EAAE,CAAC;;YACrB,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;IACnB,CAAC;IACD,OAAO,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AACvB,CAAC"}
@@ -0,0 +1,10 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ /**
3
+ * Extract a `.xlsx` (Excel) workbook via SheetJS: per sheet a `[sheet: Name]`
4
+ * marker, then the rows as tab-separated values (each cell's FORMATTED value,
5
+ * e.g. dates as dates; tabs/newlines INSIDE a cell become single spaces so a
6
+ * cell's line break never reads as a row boundary), with trailing empty rows
7
+ * trimmed.
8
+ */
9
+ export declare function extractXlsx(bytes: Buffer): ExtractResult;
10
+ //# sourceMappingURL=extract-xlsx.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extract-xlsx.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-xlsx.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAY5D;;;;;;GAMG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CA0ExD"}