@bevel-software/platform-core-backend 0.11.2 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +1165 -427
- package/dist/core/create-core-server.js +1 -1
- package/dist/core/create-core-server.js.map +1 -1
- package/dist/core/create-core-services.d.ts +2 -0
- package/dist/core/create-core-services.d.ts.map +1 -1
- package/dist/core/create-core-services.js +5 -0
- package/dist/core/create-core-services.js.map +1 -1
- package/dist/core-config.d.ts +7 -0
- package/dist/core-config.d.ts.map +1 -1
- package/dist/core-config.js +9 -0
- package/dist/core-config.js.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.js +7 -1
- package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
- package/dist/modules/kb-fs/clone-config.d.ts +40 -2
- package/dist/modules/kb-fs/clone-config.d.ts.map +1 -1
- package/dist/modules/kb-fs/clone-config.js +94 -2
- package/dist/modules/kb-fs/clone-config.js.map +1 -1
- package/dist/modules/workflow/workflow.service.d.ts +38 -0
- package/dist/modules/workflow/workflow.service.d.ts.map +1 -1
- package/dist/modules/workflow/workflow.service.js +112 -6
- package/dist/modules/workflow/workflow.service.js.map +1 -1
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.js +59 -0
- package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.js +19 -0
- package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.js +151 -0
- package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
- package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
- package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
- package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
- package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
- package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
- package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.js +55 -0
- package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.js +108 -0
- package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.js +30 -0
- package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.js +116 -0
- package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.js +117 -0
- package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
- package/dist/modules/workspace/startup/kb-git.d.ts.map +1 -1
- package/dist/modules/workspace/startup/kb-git.js +21 -4
- package/dist/modules/workspace/startup/kb-git.js.map +1 -1
- package/dist/modules/workspace/workspace.service.d.ts +52 -8
- package/dist/modules/workspace/workspace.service.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.service.js +121 -23
- package/dist/modules/workspace/workspace.service.js.map +1 -1
- package/dist/modules/workspace/workspace.tools.d.ts +2 -1
- package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.tools.js +158 -15
- package/dist/modules/workspace/workspace.tools.js.map +1 -1
- package/package.json +11 -6
- package/src/core/create-core-server.ts +1 -1
- package/src/core/create-core-services.ts +6 -0
- package/src/core-config.ts +9 -0
- package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
- package/src/modules/code-mode/code-mode.tool.ts +7 -1
- package/src/modules/kb-fs/__tests__/clone-config.test.ts +63 -2
- package/src/modules/kb-fs/clone-config.ts +97 -2
- package/src/modules/secrets-vault/secrets-vault.routes.ts +582 -582
- package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
- package/src/modules/workflow/__tests__/workflow.service.commitFileWhileLocked.test.ts +11 -5
- package/src/modules/workflow/__tests__/workflow.service.releaseLock.test.ts +172 -7
- package/src/modules/workflow/workflow.service.ts +118 -6
- package/src/modules/workspace/__tests__/workspace.service.test.ts +1 -1
- package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
- package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
- package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
- package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
- package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
- package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
- package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
- package/src/modules/workspace/file-readers/document-reader.ts +64 -0
- package/src/modules/workspace/file-readers/email-reader.ts +21 -0
- package/src/modules/workspace/file-readers/email-text.ts +193 -0
- package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
- package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
- package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
- package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
- package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
- package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
- package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
- package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
- package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
- package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
- package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
- package/src/modules/workspace/file-readers/file-reader.ts +104 -0
- package/src/modules/workspace/file-readers/image-read.ts +122 -0
- package/src/modules/workspace/file-readers/image-reader.ts +39 -0
- package/src/modules/workspace/file-readers/odf-text.ts +123 -0
- package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
- package/src/modules/workspace/file-readers/text-reader.ts +131 -0
- package/src/modules/workspace/startup/__tests__/kb-startup-runner.test.ts +141 -0
- package/src/modules/workspace/startup/kb-git.ts +20 -7
- package/src/modules/workspace/workspace.service.ts +132 -25
- package/src/modules/workspace/workspace.tools.ts +174 -12
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Extract the BODY text of a `.odt` (OpenDocument Text) document.
|
|
4
|
+
*
|
|
5
|
+
* An odt is a zip whose main part is `content.xml`; the body lives under
|
|
6
|
+
* `<office:text>`. Headers/footers are skipped like docx — in ODF they live in
|
|
7
|
+
* `styles.xml`, which is never opened, so reading `content.xml` alone IS the
|
|
8
|
+
* body-only extraction.
|
|
9
|
+
*
|
|
10
|
+
* Paragraphs (`<text:p>`) and headings (`<text:h>`, heading text as its own
|
|
11
|
+
* line) become lines in document order; `<text:span>` runs inside concatenate
|
|
12
|
+
* with NO separator, and the ODF whitespace elements (`<text:tab/>`,
|
|
13
|
+
* `<text:line-break/>`, `<text:s text:c="N"/>`) render as real characters —
|
|
14
|
+
* see `odfParagraphText`.
|
|
15
|
+
*/
|
|
16
|
+
export declare function extractOdt(bytes: Buffer): ExtractResult;
|
|
17
|
+
//# sourceMappingURL=extract-odt.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-odt.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-odt.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAI5D;;;;;;;;;;;;;GAaG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CA6BvD"}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { localBlocks, removeLocalElements } from './ooxml-text.js';
|
|
2
|
+
import { odfParagraphBlocks, odfParagraphText, readOdfContentXml } from './odf-text.js';
|
|
3
|
+
/**
|
|
4
|
+
* Extract the BODY text of a `.odt` (OpenDocument Text) document.
|
|
5
|
+
*
|
|
6
|
+
* An odt is a zip whose main part is `content.xml`; the body lives under
|
|
7
|
+
* `<office:text>`. Headers/footers are skipped like docx — in ODF they live in
|
|
8
|
+
* `styles.xml`, which is never opened, so reading `content.xml` alone IS the
|
|
9
|
+
* body-only extraction.
|
|
10
|
+
*
|
|
11
|
+
* Paragraphs (`<text:p>`) and headings (`<text:h>`, heading text as its own
|
|
12
|
+
* line) become lines in document order; `<text:span>` runs inside concatenate
|
|
13
|
+
* with NO separator, and the ODF whitespace elements (`<text:tab/>`,
|
|
14
|
+
* `<text:line-break/>`, `<text:s text:c="N"/>`) render as real characters —
|
|
15
|
+
* see `odfParagraphText`.
|
|
16
|
+
*/
|
|
17
|
+
export function extractOdt(bytes) {
|
|
18
|
+
const content = readOdfContentXml(bytes, '.odt');
|
|
19
|
+
if (!content.ok)
|
|
20
|
+
return content;
|
|
21
|
+
// Table cells contain their own <text:p>, so the flat paragraph scan renders
|
|
22
|
+
// table text too (one line per cell paragraph, like the raw document order).
|
|
23
|
+
// The text BODY, read by the parser and matched on its LOCAL name: a
|
|
24
|
+
// comment mentioning `</office:text>` used to terminate the body early and
|
|
25
|
+
// drop every paragraph after it, and an ODT binding the office namespace to
|
|
26
|
+
// another prefix had no body at all.
|
|
27
|
+
const body = localBlocks(content.xml, 'text')[0] ?? content.xml;
|
|
28
|
+
// Tracked-change bookkeeping is not body text: `<text:tracked-changes>`
|
|
29
|
+
// stores every DELETION's content as ordinary paragraphs, so the flat scan
|
|
30
|
+
// below would read deleted text back in as document lines. Removed by its
|
|
31
|
+
// parsed element boundaries before the paragraph walk.
|
|
32
|
+
// `<office:annotation>` is a COMMENT on the document, stored as ordinary
|
|
33
|
+
// paragraphs: read flat, a reviewer's note came back as a document line.
|
|
34
|
+
const visible = removeLocalElements(body, ['tracked-changes', 'annotation']);
|
|
35
|
+
const lines = odfParagraphBlocks(visible).map(odfParagraphText);
|
|
36
|
+
const paragraphs = lines.length;
|
|
37
|
+
while (lines.length > 0 && lines[lines.length - 1].trim() === '')
|
|
38
|
+
lines.pop();
|
|
39
|
+
return {
|
|
40
|
+
ok: true,
|
|
41
|
+
summary: `${paragraphs} paragraph${paragraphs === 1 ? '' : 's'}; layout, images and formatting omitted`,
|
|
42
|
+
text: lines.join('\n'),
|
|
43
|
+
};
|
|
44
|
+
}
|
|
45
|
+
//# sourceMappingURL=extract-odt.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-odt.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-odt.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,WAAW,EAAE,mBAAmB,EAAE,MAAM,iBAAiB,CAAC;AACnE,OAAO,EAAE,kBAAkB,EAAE,gBAAgB,EAAE,iBAAiB,EAAE,MAAM,eAAe,CAAC;AAExF;;;;;;;;;;;;;GAaG;AACH,MAAM,UAAU,UAAU,CAAC,KAAa;IACtC,MAAM,OAAO,GAAG,iBAAiB,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC;IACjD,IAAI,CAAC,OAAO,CAAC,EAAE;QAAE,OAAO,OAAO,CAAC;IAEhC,6EAA6E;IAC7E,6EAA6E;IAC7E,qEAAqE;IACrE,2EAA2E;IAC3E,4EAA4E;IAC5E,qCAAqC;IACrC,MAAM,IAAI,GAAG,WAAW,CAAC,OAAO,CAAC,GAAG,EAAE,MAAM,CAAC,CAAC,CAAC,CAAC,IAAI,OAAO,CAAC,GAAG,CAAC;IAEhE,wEAAwE;IACxE,2EAA2E;IAC3E,0EAA0E;IAC1E,uDAAuD;IACvD,yEAAyE;IACzE,yEAAyE;IACzE,MAAM,OAAO,GAAG,mBAAmB,CAAC,IAAI,EAAE,CAAC,iBAAiB,EAAE,YAAY,CAAC,CAAC,CAAC;IAE7E,MAAM,KAAK,GAAG,kBAAkB,CAAC,OAAO,CAAC,CAAC,GAAG,CAAC,gBAAgB,CAAC,CAAC;IAChE,MAAM,UAAU,GAAG,KAAK,CAAC,MAAM,CAAC;IAChC,OAAO,KAAK,CAAC,MAAM,GAAG,CAAC,IAAI,KAAK,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,IAAI,EAAE,KAAK,EAAE;QAAE,KAAK,CAAC,GAAG,EAAE,CAAC;IAE9E,OAAO;QACL,EAAE,EAAE,IAAI;QACR,OAAO,EAAE,GAAG,UAAU,aAAa,UAAU,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,yCAAyC;QACvG,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC;KACvB,CAAC;AACJ,CAAC"}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-pdf.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-pdf.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAwD5D,wBAAsB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,OAAO,CAAC,aAAa,CAAC,CAmGtE"}
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
|
|
2
|
+
/**
|
|
3
|
+
* Extract a PDF's TEXT LAYER page by page with pdf.js (`pdfjs-dist`,
|
|
4
|
+
* Mozilla's maintained renderer — chosen over the unmaintained thin wrappers
|
|
5
|
+
* around it). The LEGACY build is the one supported under Node; it is loaded
|
|
6
|
+
* lazily (and once) because it is a heavyweight module most deployments only
|
|
7
|
+
* need after the first PDF read.
|
|
8
|
+
*
|
|
9
|
+
* Layout heuristic: text items on one line are joined with single spaces; a
|
|
10
|
+
* new line starts when pdf.js flags an EOL or the item's Y position jumps.
|
|
11
|
+
* A PDF with NO text layer (a scan) extracts to just the `[page N]` markers,
|
|
12
|
+
* and the summary says "no text layer (scanned document?)" — no OCR in v1.
|
|
13
|
+
*/
|
|
14
|
+
/**
|
|
15
|
+
* How much DECODED text one PDF may yield before extraction gives up.
|
|
16
|
+
*
|
|
17
|
+
* The raw-size cap below bounds what arrives; it does not bound what comes
|
|
18
|
+
* out. PDF text lives in compressed streams, so a file comfortably under
|
|
19
|
+
* 50 MB can decode to far more than that, and every character of it is held
|
|
20
|
+
* in `lines` until the extraction returns. This bound is the decoded
|
|
21
|
+
* counterpart, checked as the text accumulates rather than after.
|
|
22
|
+
*/
|
|
23
|
+
const MAX_PDF_TEXT_CHARS = 20 * 1024 * 1024; // 20M chars of extracted text
|
|
24
|
+
/**
|
|
25
|
+
* How many pages one PDF may have before extraction gives up.
|
|
26
|
+
*
|
|
27
|
+
* The decoded-text bound does not cover a document whose cost is its PAGE
|
|
28
|
+
* COUNT rather than its prose: every page costs a `getPage`, a
|
|
29
|
+
* `getTextContent` and a retained `[page N]` marker even when it holds no
|
|
30
|
+
* text at all, so a file declaring hundreds of thousands of empty pages spends
|
|
31
|
+
* minutes and megabytes without ever tripping a character budget. Real
|
|
32
|
+
* documents do not come close — a 2,000-page manual is an outlier.
|
|
33
|
+
*/
|
|
34
|
+
const MAX_PDF_PAGES = 10_000;
|
|
35
|
+
/**
|
|
36
|
+
* How many text items one PAGE may hold. Items arrive through
|
|
37
|
+
* `streamTextContent` in small chunks (~100 items each), so this bound — like
|
|
38
|
+
* the character budget — fires while the page is still streaming, not after
|
|
39
|
+
* it has materialized. It exists because item COUNT is its own cost: each
|
|
40
|
+
* item is a retained heap object, and a page of empty-string items would
|
|
41
|
+
* never trip the character budget.
|
|
42
|
+
*/
|
|
43
|
+
const MAX_PDF_ITEMS_PER_PAGE = 200_000;
|
|
44
|
+
/** The typed failure both decoded-text bounds return. */
|
|
45
|
+
function overBudget() {
|
|
46
|
+
return {
|
|
47
|
+
ok: false,
|
|
48
|
+
message: `could not be extracted as a PDF (its text decodes to over ${MAX_PDF_TEXT_CHARS} characters — over the extraction limit)`,
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
export async function extractPdf(bytes) {
|
|
52
|
+
// The same bounded-read guard the zip-based extractors apply per part: a
|
|
53
|
+
// PDF has no compressed container to pre-scan, so the bound is simply the
|
|
54
|
+
// file's raw size, checked before pdf.js parses anything.
|
|
55
|
+
if (bytes.length > MAX_DOC_PART_BYTES) {
|
|
56
|
+
return {
|
|
57
|
+
ok: false,
|
|
58
|
+
message: `could not be extracted as a PDF (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
|
|
59
|
+
};
|
|
60
|
+
}
|
|
61
|
+
let doc;
|
|
62
|
+
try {
|
|
63
|
+
doc = await openPdf(bytes);
|
|
64
|
+
}
|
|
65
|
+
catch (err) {
|
|
66
|
+
return { ok: false, message: `could not be parsed as a PDF (${err.message})` };
|
|
67
|
+
}
|
|
68
|
+
try {
|
|
69
|
+
if (doc.numPages > MAX_PDF_PAGES) {
|
|
70
|
+
return {
|
|
71
|
+
ok: false,
|
|
72
|
+
message: `could not be extracted as a PDF (it declares ${doc.numPages} pages — over the ${MAX_PDF_PAGES}-page extraction limit)`,
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
const lines = [];
|
|
76
|
+
let textChars = 0;
|
|
77
|
+
let anyText = false;
|
|
78
|
+
for (let n = 1; n <= doc.numPages; n++) {
|
|
79
|
+
lines.push(`[page ${n}]`);
|
|
80
|
+
// The marker is retained text like any other line: a document whose cost
|
|
81
|
+
// is its page count must reach the same bound as one whose cost is prose.
|
|
82
|
+
textChars += n.toString().length + 8;
|
|
83
|
+
if (textChars > MAX_PDF_TEXT_CHARS)
|
|
84
|
+
return overBudget();
|
|
85
|
+
const page = await doc.getPage(n);
|
|
86
|
+
// `streamTextContent` delivers the page's items in small chunks (~100
|
|
87
|
+
// items each, `getTextContent` is just this stream materialized), so
|
|
88
|
+
// both budgets fire WHILE the page streams: a crafted single page can
|
|
89
|
+
// no longer build its whole item array before a bound trips. Once one
|
|
90
|
+
// does, the reader is cancelled and pdf.js stops producing.
|
|
91
|
+
const reader = page.streamTextContent().getReader();
|
|
92
|
+
let pageItems = 0;
|
|
93
|
+
let line = '';
|
|
94
|
+
let lastY;
|
|
95
|
+
const flush = () => {
|
|
96
|
+
if (line.trim() !== '') {
|
|
97
|
+
lines.push(line);
|
|
98
|
+
anyText = true;
|
|
99
|
+
// The '\n' the final join emits for this line is retained text too.
|
|
100
|
+
textChars += 1;
|
|
101
|
+
}
|
|
102
|
+
line = '';
|
|
103
|
+
};
|
|
104
|
+
for (;;) {
|
|
105
|
+
const { done, value: chunk } = await reader.read();
|
|
106
|
+
if (done)
|
|
107
|
+
break;
|
|
108
|
+
pageItems += chunk.items.length;
|
|
109
|
+
if (pageItems > MAX_PDF_ITEMS_PER_PAGE) {
|
|
110
|
+
await reader.cancel().catch(() => undefined);
|
|
111
|
+
page.cleanup();
|
|
112
|
+
return {
|
|
113
|
+
ok: false,
|
|
114
|
+
message: `could not be extracted as a PDF (page ${n} holds more than ${MAX_PDF_ITEMS_PER_PAGE} text items — over the extraction limit)`,
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
for (const item of chunk.items) {
|
|
118
|
+
if (!('str' in item))
|
|
119
|
+
continue; // marked-content item — no text
|
|
120
|
+
const y = item.transform?.[5];
|
|
121
|
+
// Y-position jump = new visual line (1pt tolerance for kerning wobble).
|
|
122
|
+
if (typeof y === 'number') {
|
|
123
|
+
if (lastY !== undefined && Math.abs(y - lastY) > 1)
|
|
124
|
+
flush();
|
|
125
|
+
lastY = y;
|
|
126
|
+
}
|
|
127
|
+
if (item.str !== '') {
|
|
128
|
+
// COUNTED before it is kept — the join space included: the bound
|
|
129
|
+
// exists to stop the decoded text from accumulating, so it must
|
|
130
|
+
// fire mid-page and cover every character the result will hold.
|
|
131
|
+
textChars += item.str.length + (line === '' ? 0 : 1);
|
|
132
|
+
if (textChars > MAX_PDF_TEXT_CHARS) {
|
|
133
|
+
await reader.cancel().catch(() => undefined);
|
|
134
|
+
return overBudget();
|
|
135
|
+
}
|
|
136
|
+
line += (line === '' ? '' : ' ') + item.str;
|
|
137
|
+
}
|
|
138
|
+
if (item.hasEOL)
|
|
139
|
+
flush();
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
flush();
|
|
143
|
+
page.cleanup();
|
|
144
|
+
}
|
|
145
|
+
const pages = `${doc.numPages} page${doc.numPages === 1 ? '' : 's'}`;
|
|
146
|
+
return anyText
|
|
147
|
+
? { ok: true, summary: `${pages}; layout, images and formatting omitted`, text: lines.join('\n') }
|
|
148
|
+
: { ok: true, summary: `${pages}; no text layer (scanned document?)`, text: lines.join('\n') };
|
|
149
|
+
}
|
|
150
|
+
catch (err) {
|
|
151
|
+
return { ok: false, message: `could not extract the PDF's text (${err.message})` };
|
|
152
|
+
}
|
|
153
|
+
finally {
|
|
154
|
+
await doc.destroy();
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
let pdfjsPromise;
|
|
158
|
+
async function openPdf(bytes) {
|
|
159
|
+
pdfjsPromise ??= import('pdfjs-dist/legacy/build/pdf.mjs').catch((err) => {
|
|
160
|
+
// A FAILED load must not be memoized: left in place, the rejected promise
|
|
161
|
+
// would answer every later read and disable PDF extraction for the whole
|
|
162
|
+
// process. Reset so the next read retries the import.
|
|
163
|
+
pdfjsPromise = undefined;
|
|
164
|
+
throw err;
|
|
165
|
+
});
|
|
166
|
+
const { getDocument } = await pdfjsPromise;
|
|
167
|
+
return getDocument({
|
|
168
|
+
// Copy into a fresh Uint8Array: pdf.js TRANSFERS the buffer it is given
|
|
169
|
+
// (detaching it), and the caller's Buffer must stay usable for hashing.
|
|
170
|
+
data: new Uint8Array(bytes),
|
|
171
|
+
// Server side: no font rendering — text content is all we consume.
|
|
172
|
+
disableFontFace: true,
|
|
173
|
+
useSystemFonts: true,
|
|
174
|
+
}).promise;
|
|
175
|
+
}
|
|
176
|
+
//# sourceMappingURL=extract-pdf.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-pdf.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-pdf.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,kBAAkB,EAAE,MAAM,iBAAiB,CAAC;AAErD;;;;;;;;;;;GAWG;AACH;;;;;;;;GAQG;AACH,MAAM,kBAAkB,GAAG,EAAE,GAAG,IAAI,GAAG,IAAI,CAAC,CAAC,8BAA8B;AAE3E;;;;;;;;;GASG;AACH,MAAM,aAAa,GAAG,MAAM,CAAC;AAE7B;;;;;;;GAOG;AACH,MAAM,sBAAsB,GAAG,OAAO,CAAC;AAEvC,yDAAyD;AACzD,SAAS,UAAU;IACjB,OAAO;QACL,EAAE,EAAE,KAAK;QACT,OAAO,EAAE,6DAA6D,kBAAkB,0CAA0C;KACnI,CAAC;AACJ,CAAC;AAED,MAAM,CAAC,KAAK,UAAU,UAAU,CAAC,KAAa;IAC5C,yEAAyE;IACzE,0EAA0E;IAC1E,0DAA0D;IAC1D,IAAI,KAAK,CAAC,MAAM,GAAG,kBAAkB,EAAE,CAAC;QACtC,OAAO;YACL,EAAE,EAAE,KAAK;YACT,OAAO,EAAE,gDAAgD,KAAK,CAAC,MAAM,qBAAqB,kBAAkB,iCAAiC;SAC9I,CAAC;IACJ,CAAC;IACD,IAAI,GAAwC,CAAC;IAC7C,IAAI,CAAC;QACH,GAAG,GAAG,MAAM,OAAO,CAAC,KAAK,CAAC,CAAC;IAC7B,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,iCAAkC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAC5F,CAAC;IACD,IAAI,CAAC;QACH,IAAI,GAAG,CAAC,QAAQ,GAAG,aAAa,EAAE,CAAC;YACjC,OAAO;gBACL,EAAE,EAAE,KAAK;gBACT,OAAO,EAAE,gDAAgD,GAAG,CAAC,QAAQ,qBAAqB,aAAa,yBAAyB;aACjI,CAAC;QACJ,CAAC;QACD,MAAM,KAAK,GAAa,EAAE,CAAC;QAC3B,IAAI,SAAS,GAAG,CAAC,CAAC;QAClB,IAAI,OAAO,GAAG,KAAK,CAAC;QACpB,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,IAAI,GAAG,CAAC,QAAQ,EAAE,CAAC,EAAE,EAAE,CAAC;YACvC,KAAK,CAAC,IAAI,CAAC,SAAS,CAAC,GAAG,CAAC,CAAC;YAC1B,yEAAyE;YACzE,0EAA0E;YAC1E,SAAS,IAAI,CAAC,CAAC,QAAQ,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC;YACrC,IAAI,SAAS,GAAG,kBAAkB;gBAAE,OAAO,UAAU,EAAE,CAAC;YACxD,MAAM,IAAI,GAAG,MAAM,GAAG,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC;YAClC,sEAAsE;YACtE,qEAAqE;YACrE,sEAAsE;YACtE,sEAAsE;YACtE,4DAA4D;YAC5D,MAAM,MAAM,GACV,IAAI,CAAC,iBAAiB,EACvB,CAAC,SAAS,EAAE,CAAC;YACd,IAAI,SAAS,GAAG,CAAC,CAAC;YAClB,IAAI,IAAI,GAAG,EAAE,CAAC;YACd,IAAI,KAAyB,CAAC;YAC9B,MAAM,KAAK,GAAG,GAAS,EAAE;gBACvB,IAAI,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE,EAAE,CAAC;oBACvB,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;oBACjB,OAAO,GAAG,IAAI,CAAC;oBACf,oEAAoE;oBACpE,SAAS,IAAI,CAAC,CAAC;gBACjB,CAAC;gBACD,IAAI,GAAG,EAAE,CAAC;YACZ,CAAC,CAAC;YACF,SAAS,CAAC;gBACR,MAAM,EAAE,IAAI,EAAE,KAAK,EAAE,KAAK,EAAE,GAAG,MAAM,MAAM,CAAC,IAAI,EAAE,CAAC;gBACnD,IAAI,IAAI;oBAAE,MAAM;gBAChB,SAAS,IAAI,KAAK,CAAC,KAAK,CAAC,MAAM,CAAC;gBAChC,IAAI,SAAS,GAAG,sBAAsB,EAAE,CAAC;oBACvC,MAAM,MAAM,CAAC,MAAM,EAAE,CAAC,KAAK,CAAC,GAAG,EAAE,CAAC,SAAS,CAAC,CAAC;oBAC7C,IAAI,CAAC,OAAO,EAAE,CAAC;oBACf,OAAO;wBACL,EAAE,EAAE,KAAK;wBACT,OAAO,EAAE,yCAAyC,CAAC,oBAAoB,sBAAsB,0CAA0C;qBACxI,CAAC;gBACJ,CAAC;gBACD,KAAK,MAAM,IAAI,IAAI,KAAK,CAAC,KAAK,EAAE,CAAC;oBAC/B,IAAI,CAAC,CAAC,KAAK,IAAI,IAAI,CAAC;wBAAE,SAAS,CAAC,gCAAgC;oBAChE,MAAM,CAAC,GAAG,IAAI,CAAC,SAAS,EAAE,CAAC,CAAC,CAAC,CAAC;oBAC9B,wEAAwE;oBACxE,IAAI,OAAO,CAAC,KAAK,QAAQ,EAAE,CAAC;wBAC1B,IAAI,KAAK,KAAK,SAAS,IAAI,IAAI,CAAC,GAAG,CAAC,CAAC,GAAG,KAAK,CAAC,GAAG,CAAC;4BAAE,KAAK,EAAE,CAAC;wBAC5D,KAAK,GAAG,CAAC,CAAC;oBACZ,CAAC;oBACD,IAAI,IAAI,CAAC,GAAG,KAAK,EAAE,EAAE,CAAC;wBACpB,iEAAiE;wBACjE,gEAAgE;wBAChE,gEAAgE;wBAChE,SAAS,IAAI,IAAI,CAAC,GAAG,CAAC,MAAM,GAAG,CAAC,IAAI,KAAK,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;wBACrD,IAAI,SAAS,GAAG,kBAAkB,EAAE,CAAC;4BACnC,MAAM,MAAM,CAAC,MAAM,EAAE,CAAC,KAAK,CAAC,GAAG,EAAE,CAAC,SAAS,CAAC,CAAC;4BAC7C,OAAO,UAAU,EAAE,CAAC;wBACtB,CAAC;wBACD,IAAI,IAAI,CAAC,IAAI,KAAK,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC;oBAC9C,CAAC;oBACD,IAAI,IAAI,CAAC,MAAM;wBAAE,KAAK,EAAE,CAAC;gBAC3B,CAAC;YACH,CAAC;YACD,KAAK,EAAE,CAAC;YACR,IAAI,CAAC,OAAO,EAAE,CAAC;QACjB,CAAC;QACD,MAAM,KAAK,GAAG,GAAG,GAAG,CAAC,QAAQ,QAAQ,GAAG,CAAC,QAAQ,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC;QACrE,OAAO,OAAO;YACZ,CAAC,CAAC,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,GAAG,KAAK,yCAAyC,EAAE,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE;YAClG,CAAC,CAAC,EAAE,EAAE,EAAE,IAAI,EAAE,OAAO,EAAE,GAAG,KAAK,qCAAqC,EAAE,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;IACnG,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,qCAAsC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAChG,CAAC;YAAS,CAAC;QACT,MAAM,GAAG,CAAC,OAAO,EAAE,CAAC;IACtB,CAAC;AACH,CAAC;AAGD,IAAI,YAAwC,CAAC;AAE7C,KAAK,UAAU,OAAO,CAAC,KAAa;IAClC,YAAY,KAAK,MAAM,CAAC,iCAAiC,CAAC,CAAC,KAAK,CAAC,CAAC,GAAY,EAAE,EAAE;QAChF,0EAA0E;QAC1E,yEAAyE;QACzE,sDAAsD;QACtD,YAAY,GAAG,SAAS,CAAC;QACzB,MAAM,GAAG,CAAC;IACZ,CAAC,CAAC,CAAC;IACH,MAAM,EAAE,WAAW,EAAE,GAAG,MAAM,YAAY,CAAC;IAC3C,OAAO,WAAW,CAAC;QACjB,wEAAwE;QACxE,wEAAwE;QACxE,IAAI,EAAE,IAAI,UAAU,CAAC,KAAK,CAAC;QAC3B,mEAAmE;QACnE,eAAe,EAAE,IAAI;QACrB,cAAc,EAAE,IAAI;KACrB,CAAC,CAAC,OAAO,CAAC;AACb,CAAC"}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Extract the text of a `.pptx` (PowerPoint) deck.
|
|
4
|
+
*
|
|
5
|
+
* Slides live at `ppt/slides/slideN.xml`; each is emitted under a `[slide N]`
|
|
6
|
+
* marker line, in the PRESENTATION's slide order — `ppt/presentation.xml`'s
|
|
7
|
+
* `<p:sldIdLst>`, resolved through its rels part (see
|
|
8
|
+
* `slideOrderFromPresentation`) — with numeric filename order as the fallback
|
|
9
|
+
* when the package has no readable list. Speaker notes
|
|
10
|
+
* follow their slide under `[slide N notes]` when non-empty. A slide's notes
|
|
11
|
+
* part is resolved through the slide's RELATIONSHIPS part (the `_rels` twin of
|
|
12
|
+
* the slide part's own NAME, relationship type ending `notesSlide`) — the
|
|
13
|
+
* package is free to number notes parts differently from slides — with the
|
|
14
|
+
* `notesSlideN.xml` convention as the fallback when the slide has no rels
|
|
15
|
+
* part at all. Within a slide, each `<a:p>` paragraph is a line;
|
|
16
|
+
* `<a:t>` runs concatenate with no separator (runs split mid-word).
|
|
17
|
+
*
|
|
18
|
+
* Bounded: every entry's DECLARED uncompressed size is checked before
|
|
19
|
+
* inflation (see `zipEntryOversize`), and the parts read for one deck may not
|
|
20
|
+
* exceed `MAX_DOC_TOTAL_BYTES` in total — over either bound is a typed
|
|
21
|
+
* failure, never an allocation.
|
|
22
|
+
*/
|
|
23
|
+
export declare function extractPptx(bytes: Buffer): ExtractResult;
|
|
24
|
+
/**
|
|
25
|
+
* The Target of the first `notesSlide`-typed Relationship in a rels part, or
|
|
26
|
+
* undefined.
|
|
27
|
+
*
|
|
28
|
+
* Read by the parser: matched on the element's LOCAL name, so a producer that
|
|
29
|
+
* binds the relationships namespace to a prefix (`<r:Relationship r:Type=…>`)
|
|
30
|
+
* is read like any other — and a `<Relationship>`-looking fragment written
|
|
31
|
+
* inside a COMMENT or a CDATA section is text, not live metadata pointing the
|
|
32
|
+
* notes lookup at a part of its author's choosing.
|
|
33
|
+
*/
|
|
34
|
+
export declare function notesTargetFromRels(relsXml: string): string | undefined;
|
|
35
|
+
/** Resolve an OPC relationship Target against the part's base directory. */
|
|
36
|
+
export declare function resolveRelTarget(baseDir: string, target: string): string;
|
|
37
|
+
//# sourceMappingURL=extract-pptx.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-pptx.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-pptx.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAY5D;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CA8DxD;AA0KD;;;;;;;;;GASG;AACH,wBAAgB,mBAAmB,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAWvE;AAED,4EAA4E;AAC5E,wBAAgB,gBAAgB,CAAC,OAAO,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM,GAAG,MAAM,CAWxE"}
|
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
import AdmZip from 'adm-zip';
|
|
2
|
+
import { MAX_DOC_TOTAL_BYTES, attrByLocalName, decodeXmlEntities, localBlocks, localElementBlocks, localName, paragraphRunText, zipEntryOversize, } from './ooxml-text.js';
|
|
3
|
+
/**
|
|
4
|
+
* Extract the text of a `.pptx` (PowerPoint) deck.
|
|
5
|
+
*
|
|
6
|
+
* Slides live at `ppt/slides/slideN.xml`; each is emitted under a `[slide N]`
|
|
7
|
+
* marker line, in the PRESENTATION's slide order — `ppt/presentation.xml`'s
|
|
8
|
+
* `<p:sldIdLst>`, resolved through its rels part (see
|
|
9
|
+
* `slideOrderFromPresentation`) — with numeric filename order as the fallback
|
|
10
|
+
* when the package has no readable list. Speaker notes
|
|
11
|
+
* follow their slide under `[slide N notes]` when non-empty. A slide's notes
|
|
12
|
+
* part is resolved through the slide's RELATIONSHIPS part (the `_rels` twin of
|
|
13
|
+
* the slide part's own NAME, relationship type ending `notesSlide`) — the
|
|
14
|
+
* package is free to number notes parts differently from slides — with the
|
|
15
|
+
* `notesSlideN.xml` convention as the fallback when the slide has no rels
|
|
16
|
+
* part at all. Within a slide, each `<a:p>` paragraph is a line;
|
|
17
|
+
* `<a:t>` runs concatenate with no separator (runs split mid-word).
|
|
18
|
+
*
|
|
19
|
+
* Bounded: every entry's DECLARED uncompressed size is checked before
|
|
20
|
+
* inflation (see `zipEntryOversize`), and the parts read for one deck may not
|
|
21
|
+
* exceed `MAX_DOC_TOTAL_BYTES` in total — over either bound is a typed
|
|
22
|
+
* failure, never an allocation.
|
|
23
|
+
*/
|
|
24
|
+
export function extractPptx(bytes) {
|
|
25
|
+
let slides;
|
|
26
|
+
let notesBySlide;
|
|
27
|
+
let presOrder;
|
|
28
|
+
try {
|
|
29
|
+
const zip = new AdmZip(bytes);
|
|
30
|
+
const budget = { remaining: MAX_DOC_TOTAL_BYTES };
|
|
31
|
+
slides = collectNumbered(zip, /^ppt\/slides\/slide(\d+)\.xml$/, budget);
|
|
32
|
+
notesBySlide = collectNotes(zip, slides, budget);
|
|
33
|
+
presOrder = slideOrderFromPresentation(zip, budget);
|
|
34
|
+
}
|
|
35
|
+
catch (err) {
|
|
36
|
+
return { ok: false, message: `could not be parsed as a .pptx (${err.message})` };
|
|
37
|
+
}
|
|
38
|
+
if (slides.size === 0) {
|
|
39
|
+
return { ok: false, message: 'could not be parsed as a .pptx (no ppt/slides/slideN.xml inside the archive)' };
|
|
40
|
+
}
|
|
41
|
+
// Emission order: the presentation's own slide list when it resolves to
|
|
42
|
+
// selected parts, numeric filename order otherwise (parts the list does not
|
|
43
|
+
// name follow it, in filename order). When the LIST orders the deck, the
|
|
44
|
+
// markers number POSITIONS in it — what a viewer calls slide 1 — because a
|
|
45
|
+
// reordered deck's part filenames no longer mean anything positional.
|
|
46
|
+
const byFilename = [...slides.keys()].sort((a, b) => a - b);
|
|
47
|
+
let order = byFilename;
|
|
48
|
+
let positional = false;
|
|
49
|
+
if (presOrder !== undefined) {
|
|
50
|
+
const numByName = new Map();
|
|
51
|
+
for (const [n, slide] of slides)
|
|
52
|
+
numByName.set(slide.name, n);
|
|
53
|
+
const seen = new Set();
|
|
54
|
+
const fromList = [];
|
|
55
|
+
for (const name of presOrder) {
|
|
56
|
+
const n = numByName.get(name);
|
|
57
|
+
if (n !== undefined && !seen.has(n)) {
|
|
58
|
+
seen.add(n);
|
|
59
|
+
fromList.push(n);
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
if (fromList.length > 0) {
|
|
63
|
+
order = [...fromList, ...byFilename.filter((n) => !seen.has(n))];
|
|
64
|
+
positional = true;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
const lines = [];
|
|
68
|
+
let anyNotes = false;
|
|
69
|
+
order.forEach((n, i) => {
|
|
70
|
+
const label = positional ? i + 1 : n;
|
|
71
|
+
lines.push(`[slide ${label}]`);
|
|
72
|
+
lines.push(...paragraphLines(slides.get(n).xml));
|
|
73
|
+
const notesXml = notesBySlide.get(n);
|
|
74
|
+
const noteLines = notesXml !== undefined ? paragraphLines(notesXml) : [];
|
|
75
|
+
if (noteLines.length > 0) {
|
|
76
|
+
anyNotes = true;
|
|
77
|
+
lines.push(`[slide ${label} notes]`);
|
|
78
|
+
lines.push(...noteLines);
|
|
79
|
+
}
|
|
80
|
+
});
|
|
81
|
+
return {
|
|
82
|
+
ok: true,
|
|
83
|
+
summary: `${slides.size} slide${slides.size === 1 ? '' : 's'}${anyNotes ? ' + notes' : ''}; layout, images and formatting omitted`,
|
|
84
|
+
text: lines.join('\n'),
|
|
85
|
+
};
|
|
86
|
+
}
|
|
87
|
+
/** Non-empty paragraph texts of one slide/notes part, in document order. */
|
|
88
|
+
function paragraphLines(xml) {
|
|
89
|
+
const out = [];
|
|
90
|
+
for (const p of localBlocks(xml, 'p')) {
|
|
91
|
+
const text = paragraphRunText(p, 't');
|
|
92
|
+
if (text.trim() !== '')
|
|
93
|
+
out.push(text);
|
|
94
|
+
}
|
|
95
|
+
return out;
|
|
96
|
+
}
|
|
97
|
+
/** `entry`'s bytes as UTF-8, after the per-part and aggregate bounds. Throws over either. */
|
|
98
|
+
function readEntryBounded(entry, budget) {
|
|
99
|
+
const oversize = zipEntryOversize(entry);
|
|
100
|
+
if (oversize)
|
|
101
|
+
throw new Error(oversize);
|
|
102
|
+
budget.remaining -= entry.header.size;
|
|
103
|
+
if (budget.remaining < 0) {
|
|
104
|
+
throw new Error(`the archive's parts exceed the ${MAX_DOC_TOTAL_BYTES}-byte (200 MB) total extraction limit`);
|
|
105
|
+
}
|
|
106
|
+
return entry.getData().toString('utf8');
|
|
107
|
+
}
|
|
108
|
+
/**
|
|
109
|
+
* Entries matching `re` (capture 1 = number), decoded as UTF-8, keyed by
|
|
110
|
+
* number. Two part names can parse to the SAME number (`slide1.xml` and
|
|
111
|
+
* `slide01.xml`); the winner is the FIRST in ascending part-name order —
|
|
112
|
+
* deterministic regardless of zip entry order, and the same policy as the
|
|
113
|
+
* browser twin (`pptxOutline.ts`), so viewer and `read_file` agree. Losing
|
|
114
|
+
* duplicates are never inflated (no budget charge).
|
|
115
|
+
*
|
|
116
|
+
* The winner's NAME rides along with its bytes because everything else about
|
|
117
|
+
* a slide hangs off the part name, not the number: see `collectNotes`.
|
|
118
|
+
*/
|
|
119
|
+
function collectNumbered(zip, re, budget) {
|
|
120
|
+
const matched = [];
|
|
121
|
+
for (const entry of zip.getEntries()) {
|
|
122
|
+
const m = re.exec(entry.entryName);
|
|
123
|
+
if (!m)
|
|
124
|
+
continue;
|
|
125
|
+
// A crafted name can spell a number past 2^53 (or Infinity): distinct
|
|
126
|
+
// parts would collide in the map and one would silently vanish.
|
|
127
|
+
const n = parseInt(m[1], 10);
|
|
128
|
+
if (Number.isSafeInteger(n))
|
|
129
|
+
matched.push([n, entry.entryName, entry]);
|
|
130
|
+
}
|
|
131
|
+
// A zip may list the SAME part name twice. Sorting by name leaves those two
|
|
132
|
+
// in archive order, so which one wins depends on how the file was written —
|
|
133
|
+
// and two archives with identical parts would extract differently. A part
|
|
134
|
+
// claimed twice is not a part this reader can resolve, so it is dropped.
|
|
135
|
+
const claims = new Map();
|
|
136
|
+
for (const [, name] of matched)
|
|
137
|
+
claims.set(name, (claims.get(name) ?? 0) + 1);
|
|
138
|
+
const unique = matched.filter(([, name]) => claims.get(name) === 1);
|
|
139
|
+
unique.sort((a, b) => (a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0));
|
|
140
|
+
const out = new Map();
|
|
141
|
+
for (const [n, name, entry] of unique) {
|
|
142
|
+
if (!out.has(n))
|
|
143
|
+
out.set(n, { name, xml: readEntryBounded(entry, budget) });
|
|
144
|
+
}
|
|
145
|
+
return out;
|
|
146
|
+
}
|
|
147
|
+
const NOTES_REL_TYPE_SUFFIX = '/notesSlide';
|
|
148
|
+
const SLIDE_REL_TYPE_SUFFIX = '/slide';
|
|
149
|
+
/**
|
|
150
|
+
* The value of the attribute whose LOCAL name is `want` AND that carries a
|
|
151
|
+
* namespace prefix. `<p:sldId>` holds both its own `id` and the relationship
|
|
152
|
+
* reference `r:id`; plain local-name matching answers with whichever is
|
|
153
|
+
* written first, so the relationship id must be the PREFIXED one.
|
|
154
|
+
*/
|
|
155
|
+
function prefixedAttrByLocalName(attributes, want) {
|
|
156
|
+
for (const [key, value] of Object.entries(attributes)) {
|
|
157
|
+
if (key === 'xmlns' || key.startsWith('xmlns:'))
|
|
158
|
+
continue;
|
|
159
|
+
if (key.includes(':') && localName(key) === want)
|
|
160
|
+
return value;
|
|
161
|
+
}
|
|
162
|
+
return undefined;
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* Slide part names in PRESENTATION order: `ppt/presentation.xml`'s
|
|
166
|
+
* `<p:sldIdLst>` entries, each `r:id` resolved through the presentation's own
|
|
167
|
+
* rels part — or undefined when the package has no readable list. Reordering
|
|
168
|
+
* slides in PowerPoint rewrites the sldIdLst and leaves the part names alone,
|
|
169
|
+
* so `slide1.xml` need not be the deck's first slide; the numeric filename
|
|
170
|
+
* sort is only the fallback for packages without the list.
|
|
171
|
+
*/
|
|
172
|
+
function slideOrderFromPresentation(zip, budget) {
|
|
173
|
+
const rels = zip.getEntry('ppt/_rels/presentation.xml.rels');
|
|
174
|
+
const pres = zip.getEntry('ppt/presentation.xml');
|
|
175
|
+
if (!rels || !pres)
|
|
176
|
+
return undefined;
|
|
177
|
+
const targetById = new Map();
|
|
178
|
+
for (const rel of localElementBlocks(readEntryBounded(rels, budget), ['Relationship'])) {
|
|
179
|
+
const type = attrByLocalName(rel.attributes, 'Type');
|
|
180
|
+
if (type === undefined || !type.endsWith(SLIDE_REL_TYPE_SUFFIX))
|
|
181
|
+
continue;
|
|
182
|
+
const id = attrByLocalName(rel.attributes, 'Id');
|
|
183
|
+
// Targets are RAW in the rels (see `notesTargetFromRels`) and relative to
|
|
184
|
+
// the presentation part's directory.
|
|
185
|
+
const target = attrByLocalName(rel.attributes, 'Target');
|
|
186
|
+
if (id !== undefined && target !== undefined) {
|
|
187
|
+
targetById.set(id, resolveRelTarget('ppt', decodeXmlEntities(target)));
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
if (targetById.size === 0)
|
|
191
|
+
return undefined;
|
|
192
|
+
const list = localBlocks(readEntryBounded(pres, budget), 'sldIdLst')[0];
|
|
193
|
+
if (list === undefined)
|
|
194
|
+
return undefined;
|
|
195
|
+
const order = [];
|
|
196
|
+
for (const sld of localElementBlocks(list, ['sldId'])) {
|
|
197
|
+
const rid = prefixedAttrByLocalName(sld.attributes, 'id');
|
|
198
|
+
const target = rid !== undefined ? targetById.get(rid) : undefined;
|
|
199
|
+
if (target !== undefined)
|
|
200
|
+
order.push(target);
|
|
201
|
+
}
|
|
202
|
+
return order.length > 0 ? order : undefined;
|
|
203
|
+
}
|
|
204
|
+
/**
|
|
205
|
+
* The OPC relationships part of `partName` — `dir/_rels/base.rels`. Derived
|
|
206
|
+
* from the part NAME, never from the slide number: when a deck ships both
|
|
207
|
+
* `slide1.xml` and `slide01.xml`, the name-ordering rule picks `slide01.xml`,
|
|
208
|
+
* whose rels part is `slide01.xml.rels`. Rebuilding the path from the number
|
|
209
|
+
* asked for `slide1.xml.rels` — a part belonging to the OTHER file — and so
|
|
210
|
+
* either lost that slide's speaker notes or attached the losing part's.
|
|
211
|
+
*/
|
|
212
|
+
function relsPartName(partName) {
|
|
213
|
+
const cut = partName.lastIndexOf('/');
|
|
214
|
+
return `${partName.slice(0, cut)}/_rels/${partName.slice(cut + 1)}.rels`;
|
|
215
|
+
}
|
|
216
|
+
/**
|
|
217
|
+
* The conventional notes part for a slide part — `slide01.xml` →
|
|
218
|
+
* `notesSlide01.xml`. Derived from the name for the same reason as the rels
|
|
219
|
+
* path, so a zero-padded deck's fallback lands on the matching notes part.
|
|
220
|
+
*/
|
|
221
|
+
function conventionalNotesPart(partName) {
|
|
222
|
+
const base = partName.slice(partName.lastIndexOf('/') + 1);
|
|
223
|
+
return `ppt/notesSlides/notes${base[0].toUpperCase()}${base.slice(1)}`;
|
|
224
|
+
}
|
|
225
|
+
/**
|
|
226
|
+
* Slide number → its notes part's XML, resolved through each slide's `.rels`
|
|
227
|
+
* part; the conventional-name fallback ONLY for a slide without a rels part.
|
|
228
|
+
* Both paths come from the SELECTED part's name (see `relsPartName`).
|
|
229
|
+
*/
|
|
230
|
+
function collectNotes(zip, slides, budget) {
|
|
231
|
+
const out = new Map();
|
|
232
|
+
for (const [n, slide] of slides) {
|
|
233
|
+
const rels = zip.getEntry(relsPartName(slide.name));
|
|
234
|
+
let notesPart;
|
|
235
|
+
if (rels) {
|
|
236
|
+
const target = notesTargetFromRels(readEntryBounded(rels, budget));
|
|
237
|
+
notesPart = target !== undefined ? resolveRelTarget('ppt/slides', target) : undefined;
|
|
238
|
+
}
|
|
239
|
+
else {
|
|
240
|
+
notesPart = conventionalNotesPart(slide.name);
|
|
241
|
+
}
|
|
242
|
+
if (notesPart === undefined)
|
|
243
|
+
continue;
|
|
244
|
+
const entry = zip.getEntry(notesPart);
|
|
245
|
+
if (entry)
|
|
246
|
+
out.set(n, readEntryBounded(entry, budget));
|
|
247
|
+
}
|
|
248
|
+
return out;
|
|
249
|
+
}
|
|
250
|
+
/**
|
|
251
|
+
* The Target of the first `notesSlide`-typed Relationship in a rels part, or
|
|
252
|
+
* undefined.
|
|
253
|
+
*
|
|
254
|
+
* Read by the parser: matched on the element's LOCAL name, so a producer that
|
|
255
|
+
* binds the relationships namespace to a prefix (`<r:Relationship r:Type=…>`)
|
|
256
|
+
* is read like any other — and a `<Relationship>`-looking fragment written
|
|
257
|
+
* inside a COMMENT or a CDATA section is text, not live metadata pointing the
|
|
258
|
+
* notes lookup at a part of its author's choosing.
|
|
259
|
+
*/
|
|
260
|
+
export function notesTargetFromRels(relsXml) {
|
|
261
|
+
for (const rel of localElementBlocks(relsXml, ['Relationship'])) {
|
|
262
|
+
const type = attrByLocalName(rel.attributes, 'Type');
|
|
263
|
+
if (type !== undefined && type.endsWith(NOTES_REL_TYPE_SUFFIX)) {
|
|
264
|
+
// Attribute values are RAW here (the block reader does not decode), and a
|
|
265
|
+
// part name may legally contain `&`, written `&` in the rels.
|
|
266
|
+
const target = attrByLocalName(rel.attributes, 'Target');
|
|
267
|
+
return target !== undefined ? decodeXmlEntities(target) : undefined;
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
return undefined;
|
|
271
|
+
}
|
|
272
|
+
/** Resolve an OPC relationship Target against the part's base directory. */
|
|
273
|
+
export function resolveRelTarget(baseDir, target) {
|
|
274
|
+
const parts = target.startsWith('/')
|
|
275
|
+
? target.slice(1).split('/')
|
|
276
|
+
: [...baseDir.split('/'), ...target.split('/')];
|
|
277
|
+
const out = [];
|
|
278
|
+
for (const p of parts) {
|
|
279
|
+
if (p === '' || p === '.')
|
|
280
|
+
continue;
|
|
281
|
+
if (p === '..')
|
|
282
|
+
out.pop();
|
|
283
|
+
else
|
|
284
|
+
out.push(p);
|
|
285
|
+
}
|
|
286
|
+
return out.join('/');
|
|
287
|
+
}
|
|
288
|
+
//# sourceMappingURL=extract-pptx.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-pptx.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-pptx.ts"],"names":[],"mappings":"AAAA,OAAO,MAAM,MAAM,SAAS,CAAC;AAE7B,OAAO,EACL,mBAAmB,EACnB,eAAe,EACf,iBAAiB,EACjB,WAAW,EACX,kBAAkB,EAClB,SAAS,EACT,gBAAgB,EAChB,gBAAgB,GACjB,MAAM,iBAAiB,CAAC;AAEzB;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,UAAU,WAAW,CAAC,KAAa;IACvC,IAAI,MAA8B,CAAC;IACnC,IAAI,YAAiC,CAAC;IACtC,IAAI,SAA+B,CAAC;IACpC,IAAI,CAAC;QACH,MAAM,GAAG,GAAG,IAAI,MAAM,CAAC,KAAK,CAAC,CAAC;QAC9B,MAAM,MAAM,GAAG,EAAE,SAAS,EAAE,mBAAmB,EAAE,CAAC;QAClD,MAAM,GAAG,eAAe,CAAC,GAAG,EAAE,gCAAgC,EAAE,MAAM,CAAC,CAAC;QACxE,YAAY,GAAG,YAAY,CAAC,GAAG,EAAE,MAAM,EAAE,MAAM,CAAC,CAAC;QACjD,SAAS,GAAG,0BAA0B,CAAC,GAAG,EAAE,MAAM,CAAC,CAAC;IACtD,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,mCAAoC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAC9F,CAAC;IACD,IAAI,MAAM,CAAC,IAAI,KAAK,CAAC,EAAE,CAAC;QACtB,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,8EAA8E,EAAE,CAAC;IAChH,CAAC;IAED,wEAAwE;IACxE,4EAA4E;IAC5E,yEAAyE;IACzE,2EAA2E;IAC3E,sEAAsE;IACtE,MAAM,UAAU,GAAG,CAAC,GAAG,MAAM,CAAC,IAAI,EAAE,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IAC5D,IAAI,KAAK,GAAG,UAAU,CAAC;IACvB,IAAI,UAAU,GAAG,KAAK,CAAC;IACvB,IAAI,SAAS,KAAK,SAAS,EAAE,CAAC;QAC5B,MAAM,SAAS,GAAG,IAAI,GAAG,EAAkB,CAAC;QAC5C,KAAK,MAAM,CAAC,CAAC,EAAE,KAAK,CAAC,IAAI,MAAM;YAAE,SAAS,CAAC,GAAG,CAAC,KAAK,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC;QAC9D,MAAM,IAAI,GAAG,IAAI,GAAG,EAAU,CAAC;QAC/B,MAAM,QAAQ,GAAa,EAAE,CAAC;QAC9B,KAAK,MAAM,IAAI,IAAI,SAAS,EAAE,CAAC;YAC7B,MAAM,CAAC,GAAG,SAAS,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC;YAC9B,IAAI,CAAC,KAAK,SAAS,IAAI,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC;gBACpC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC;gBACZ,QAAQ,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;YACnB,CAAC;QACH,CAAC;QACD,IAAI,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YACxB,KAAK,GAAG,CAAC,GAAG,QAAQ,EAAE,GAAG,UAAU,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;YACjE,UAAU,GAAG,IAAI,CAAC;QACpB,CAAC;IACH,CAAC;IAED,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,QAAQ,GAAG,KAAK,CAAC;IACrB,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE;QACrB,MAAM,KAAK,GAAG,UAAU,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;QACrC,KAAK,CAAC,IAAI,CAAC,UAAU,KAAK,GAAG,CAAC,CAAC;QAC/B,KAAK,CAAC,IAAI,CAAC,GAAG,cAAc,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,CAAE,CAAC,GAAG,CAAC,CAAC,CAAC;QAClD,MAAM,QAAQ,GAAG,YAAY,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC;QACrC,MAAM,SAAS,GAAG,QAAQ,KAAK,SAAS,CAAC,CAAC,CAAC,cAAc,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;QACzE,IAAI,SAAS,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YACzB,QAAQ,GAAG,IAAI,CAAC;YAChB,KAAK,CAAC,IAAI,CAAC,UAAU,KAAK,SAAS,CAAC,CAAC;YACrC,KAAK,CAAC,IAAI,CAAC,GAAG,SAAS,CAAC,CAAC;QAC3B,CAAC;IACH,CAAC,CAAC,CAAC;IACH,OAAO;QACL,EAAE,EAAE,IAAI;QACR,OAAO,EAAE,GAAG,MAAM,CAAC,IAAI,SAAS,MAAM,CAAC,IAAI,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,GAAG,QAAQ,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,EAAE,yCAAyC;QAClI,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC;KACvB,CAAC;AACJ,CAAC;AAED,4EAA4E;AAC5E,SAAS,cAAc,CAAC,GAAW;IACjC,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,MAAM,CAAC,IAAI,WAAW,CAAC,GAAG,EAAE,GAAG,CAAC,EAAE,CAAC;QACtC,MAAM,IAAI,GAAG,gBAAgB,CAAC,CAAC,EAAE,GAAG,CAAC,CAAC;QACtC,IAAI,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE;YAAE,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IACzC,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAOD,6FAA6F;AAC7F,SAAS,gBAAgB,CAAC,KAAuB,EAAE,MAAkB;IACnE,MAAM,QAAQ,GAAG,gBAAgB,CAAC,KAAK,CAAC,CAAC;IACzC,IAAI,QAAQ;QAAE,MAAM,IAAI,KAAK,CAAC,QAAQ,CAAC,CAAC;IACxC,MAAM,CAAC,SAAS,IAAI,KAAK,CAAC,MAAM,CAAC,IAAI,CAAC;IACtC,IAAI,MAAM,CAAC,SAAS,GAAG,CAAC,EAAE,CAAC;QACzB,MAAM,IAAI,KAAK,CAAC,kCAAkC,mBAAmB,uCAAuC,CAAC,CAAC;IAChH,CAAC;IACD,OAAO,KAAK,CAAC,OAAO,EAAE,CAAC,QAAQ,CAAC,MAAM,CAAC,CAAC;AAC1C,CAAC;AASD;;;;;;;;;;GAUG;AACH,SAAS,eAAe,CAAC,GAAW,EAAE,EAAU,EAAE,MAAkB;IAClE,MAAM,OAAO,GAA8C,EAAE,CAAC;IAC9D,KAAK,MAAM,KAAK,IAAI,GAAG,CAAC,UAAU,EAAE,EAAE,CAAC;QACrC,MAAM,CAAC,GAAG,EAAE,CAAC,IAAI,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC;QACnC,IAAI,CAAC,CAAC;YAAE,SAAS;QACjB,sEAAsE;QACtE,gEAAgE;QAChE,MAAM,CAAC,GAAG,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC;QAC7B,IAAI,MAAM,CAAC,aAAa,CAAC,CAAC,CAAC;YAAE,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,KAAK,CAAC,SAAS,EAAE,KAAK,CAAC,CAAC,CAAC;IACzE,CAAC;IACD,4EAA4E;IAC5E,4EAA4E;IAC5E,0EAA0E;IAC1E,yEAAyE;IACzE,MAAM,MAAM,GAAG,IAAI,GAAG,EAAkB,CAAC;IACzC,KAAK,MAAM,CAAC,EAAE,IAAI,CAAC,IAAI,OAAO;QAAE,MAAM,CAAC,GAAG,CAAC,IAAI,EAAE,CAAC,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;IAC9E,MAAM,MAAM,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,IAAI,CAAC,EAAE,EAAE,CAAC,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC;IACpE,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IAChE,MAAM,GAAG,GAAG,IAAI,GAAG,EAAqB,CAAC;IACzC,KAAK,MAAM,CAAC,CAAC,EAAE,IAAI,EAAE,KAAK,CAAC,IAAI,MAAM,EAAE,CAAC;QACtC,IAAI,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC;YAAE,GAAG,CAAC,GAAG,CAAC,CAAC,EAAE,EAAE,IAAI,EAAE,GAAG,EAAE,gBAAgB,CAAC,KAAK,EAAE,MAAM,CAAC,EAAE,CAAC,CAAC;IAC9E,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED,MAAM,qBAAqB,GAAG,aAAa,CAAC;AAC5C,MAAM,qBAAqB,GAAG,QAAQ,CAAC;AAEvC;;;;;GAKG;AACH,SAAS,uBAAuB,CAAC,UAAkC,EAAE,IAAY;IAC/E,KAAK,MAAM,CAAC,GAAG,EAAE,KAAK,CAAC,IAAI,MAAM,CAAC,OAAO,CAAC,UAAU,CAAC,EAAE,CAAC;QACtD,IAAI,GAAG,KAAK,OAAO,IAAI,GAAG,CAAC,UAAU,CAAC,QAAQ,CAAC;YAAE,SAAS;QAC1D,IAAI,GAAG,CAAC,QAAQ,CAAC,GAAG,CAAC,IAAI,SAAS,CAAC,GAAG,CAAC,KAAK,IAAI;YAAE,OAAO,KAAK,CAAC;IACjE,CAAC;IACD,OAAO,SAAS,CAAC;AACnB,CAAC;AAED;;;;;;;GAOG;AACH,SAAS,0BAA0B,CAAC,GAAW,EAAE,MAAkB;IACjE,MAAM,IAAI,GAAG,GAAG,CAAC,QAAQ,CAAC,iCAAiC,CAAC,CAAC;IAC7D,MAAM,IAAI,GAAG,GAAG,CAAC,QAAQ,CAAC,sBAAsB,CAAC,CAAC;IAClD,IAAI,CAAC,IAAI,IAAI,CAAC,IAAI;QAAE,OAAO,SAAS,CAAC;IACrC,MAAM,UAAU,GAAG,IAAI,GAAG,EAAkB,CAAC;IAC7C,KAAK,MAAM,GAAG,IAAI,kBAAkB,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,CAAC,EAAE,CAAC,cAAc,CAAC,CAAC,EAAE,CAAC;QACvF,MAAM,IAAI,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,MAAM,CAAC,CAAC;QACrD,IAAI,IAAI,KAAK,SAAS,IAAI,CAAC,IAAI,CAAC,QAAQ,CAAC,qBAAqB,CAAC;YAAE,SAAS;QAC1E,MAAM,EAAE,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,IAAI,CAAC,CAAC;QACjD,0EAA0E;QAC1E,qCAAqC;QACrC,MAAM,MAAM,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAC;QACzD,IAAI,EAAE,KAAK,SAAS,IAAI,MAAM,KAAK,SAAS,EAAE,CAAC;YAC7C,UAAU,CAAC,GAAG,CAAC,EAAE,EAAE,gBAAgB,CAAC,KAAK,EAAE,iBAAiB,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC;QACzE,CAAC;IACH,CAAC;IACD,IAAI,UAAU,CAAC,IAAI,KAAK,CAAC;QAAE,OAAO,SAAS,CAAC;IAC5C,MAAM,IAAI,GAAG,WAAW,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,CAAC,EAAE,UAAU,CAAC,CAAC,CAAC,CAAC,CAAC;IACxE,IAAI,IAAI,KAAK,SAAS;QAAE,OAAO,SAAS,CAAC;IACzC,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,KAAK,MAAM,GAAG,IAAI,kBAAkB,CAAC,IAAI,EAAE,CAAC,OAAO,CAAC,CAAC,EAAE,CAAC;QACtD,MAAM,GAAG,GAAG,uBAAuB,CAAC,GAAG,CAAC,UAAU,EAAE,IAAI,CAAC,CAAC;QAC1D,MAAM,MAAM,GAAG,GAAG,KAAK,SAAS,CAAC,CAAC,CAAC,UAAU,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC;QACnE,IAAI,MAAM,KAAK,SAAS;YAAE,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;IAC/C,CAAC;IACD,OAAO,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,SAAS,CAAC;AAC9C,CAAC;AAED;;;;;;;GAOG;AACH,SAAS,YAAY,CAAC,QAAgB;IACpC,MAAM,GAAG,GAAG,QAAQ,CAAC,WAAW,CAAC,GAAG,CAAC,CAAC;IACtC,OAAO,GAAG,QAAQ,CAAC,KAAK,CAAC,CAAC,EAAE,GAAG,CAAC,UAAU,QAAQ,CAAC,KAAK,CAAC,GAAG,GAAG,CAAC,CAAC,OAAO,CAAC;AAC3E,CAAC;AAED;;;;GAIG;AACH,SAAS,qBAAqB,CAAC,QAAgB;IAC7C,MAAM,IAAI,GAAG,QAAQ,CAAC,KAAK,CAAC,QAAQ,CAAC,WAAW,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC;IAC3D,OAAO,wBAAwB,IAAI,CAAC,CAAC,CAAC,CAAC,WAAW,EAAE,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC,EAAE,CAAC;AACzE,CAAC;AAED;;;;GAIG;AACH,SAAS,YAAY,CAAC,GAAW,EAAE,MAA8B,EAAE,MAAkB;IACnF,MAAM,GAAG,GAAG,IAAI,GAAG,EAAkB,CAAC;IACtC,KAAK,MAAM,CAAC,CAAC,EAAE,KAAK,CAAC,IAAI,MAAM,EAAE,CAAC;QAChC,MAAM,IAAI,GAAG,GAAG,CAAC,QAAQ,CAAC,YAAY,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC;QACpD,IAAI,SAA6B,CAAC;QAClC,IAAI,IAAI,EAAE,CAAC;YACT,MAAM,MAAM,GAAG,mBAAmB,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,CAAC,CAAC,CAAC;YACnE,SAAS,GAAG,MAAM,KAAK,SAAS,CAAC,CAAC,CAAC,gBAAgB,CAAC,YAAY,EAAE,MAAM,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC;QACxF,CAAC;aAAM,CAAC;YACN,SAAS,GAAG,qBAAqB,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC;QAChD,CAAC;QACD,IAAI,SAAS,KAAK,SAAS;YAAE,SAAS;QACtC,MAAM,KAAK,GAAG,GAAG,CAAC,QAAQ,CAAC,SAAS,CAAC,CAAC;QACtC,IAAI,KAAK;YAAE,GAAG,CAAC,GAAG,CAAC,CAAC,EAAE,gBAAgB,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC,CAAC;IACzD,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;;;;;GASG;AACH,MAAM,UAAU,mBAAmB,CAAC,OAAe;IACjD,KAAK,MAAM,GAAG,IAAI,kBAAkB,CAAC,OAAO,EAAE,CAAC,cAAc,CAAC,CAAC,EAAE,CAAC;QAChE,MAAM,IAAI,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,MAAM,CAAC,CAAC;QACrD,IAAI,IAAI,KAAK,SAAS,IAAI,IAAI,CAAC,QAAQ,CAAC,qBAAqB,CAAC,EAAE,CAAC;YAC/D,0EAA0E;YAC1E,kEAAkE;YAClE,MAAM,MAAM,GAAG,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,QAAQ,CAAC,CAAC;YACzD,OAAO,MAAM,KAAK,SAAS,CAAC,CAAC,CAAC,iBAAiB,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC;QACtE,CAAC;IACH,CAAC;IACD,OAAO,SAAS,CAAC;AACnB,CAAC;AAED,4EAA4E;AAC5E,MAAM,UAAU,gBAAgB,CAAC,OAAe,EAAE,MAAc;IAC9D,MAAM,KAAK,GAAG,MAAM,CAAC,UAAU,CAAC,GAAG,CAAC;QAClC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC;QAC5B,CAAC,CAAC,CAAC,GAAG,OAAO,CAAC,KAAK,CAAC,GAAG,CAAC,EAAE,GAAG,MAAM,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC;IAClD,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,MAAM,CAAC,IAAI,KAAK,EAAE,CAAC;QACtB,IAAI,CAAC,KAAK,EAAE,IAAI,CAAC,KAAK,GAAG;YAAE,SAAS;QACpC,IAAI,CAAC,KAAK,IAAI;YAAE,GAAG,CAAC,GAAG,EAAE,CAAC;;YACrB,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;IACnB,CAAC;IACD,OAAO,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AACvB,CAAC"}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Extract a `.xlsx` (Excel) workbook via SheetJS: per sheet a `[sheet: Name]`
|
|
4
|
+
* marker, then the rows as tab-separated values (each cell's FORMATTED value,
|
|
5
|
+
* e.g. dates as dates; tabs/newlines INSIDE a cell become single spaces so a
|
|
6
|
+
* cell's line break never reads as a row boundary), with trailing empty rows
|
|
7
|
+
* trimmed.
|
|
8
|
+
*/
|
|
9
|
+
export declare function extractXlsx(bytes: Buffer): ExtractResult;
|
|
10
|
+
//# sourceMappingURL=extract-xlsx.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-xlsx.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-xlsx.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAY5D;;;;;;GAMG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CA0ExD"}
|