@bevel-software/platform-core-backend 0.11.2 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +1165 -427
- package/dist/core/create-core-server.js +1 -1
- package/dist/core/create-core-server.js.map +1 -1
- package/dist/core/create-core-services.d.ts +2 -0
- package/dist/core/create-core-services.d.ts.map +1 -1
- package/dist/core/create-core-services.js +5 -0
- package/dist/core/create-core-services.js.map +1 -1
- package/dist/core-config.d.ts +7 -0
- package/dist/core-config.d.ts.map +1 -1
- package/dist/core-config.js +9 -0
- package/dist/core-config.js.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.js +7 -1
- package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
- package/dist/modules/kb-fs/clone-config.d.ts +40 -2
- package/dist/modules/kb-fs/clone-config.d.ts.map +1 -1
- package/dist/modules/kb-fs/clone-config.js +94 -2
- package/dist/modules/kb-fs/clone-config.js.map +1 -1
- package/dist/modules/workflow/workflow.service.d.ts +38 -0
- package/dist/modules/workflow/workflow.service.d.ts.map +1 -1
- package/dist/modules/workflow/workflow.service.js +112 -6
- package/dist/modules/workflow/workflow.service.js.map +1 -1
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.js +59 -0
- package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.js +19 -0
- package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.js +151 -0
- package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
- package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
- package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
- package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
- package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
- package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
- package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.js +55 -0
- package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.js +108 -0
- package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.js +30 -0
- package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.js +116 -0
- package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.js +117 -0
- package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
- package/dist/modules/workspace/startup/kb-git.d.ts.map +1 -1
- package/dist/modules/workspace/startup/kb-git.js +21 -4
- package/dist/modules/workspace/startup/kb-git.js.map +1 -1
- package/dist/modules/workspace/workspace.service.d.ts +52 -8
- package/dist/modules/workspace/workspace.service.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.service.js +121 -23
- package/dist/modules/workspace/workspace.service.js.map +1 -1
- package/dist/modules/workspace/workspace.tools.d.ts +2 -1
- package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.tools.js +158 -15
- package/dist/modules/workspace/workspace.tools.js.map +1 -1
- package/package.json +11 -6
- package/src/core/create-core-server.ts +1 -1
- package/src/core/create-core-services.ts +6 -0
- package/src/core-config.ts +9 -0
- package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
- package/src/modules/code-mode/code-mode.tool.ts +7 -1
- package/src/modules/kb-fs/__tests__/clone-config.test.ts +63 -2
- package/src/modules/kb-fs/clone-config.ts +97 -2
- package/src/modules/secrets-vault/secrets-vault.routes.ts +582 -582
- package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
- package/src/modules/workflow/__tests__/workflow.service.commitFileWhileLocked.test.ts +11 -5
- package/src/modules/workflow/__tests__/workflow.service.releaseLock.test.ts +172 -7
- package/src/modules/workflow/workflow.service.ts +118 -6
- package/src/modules/workspace/__tests__/workspace.service.test.ts +1 -1
- package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
- package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
- package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
- package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
- package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
- package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
- package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
- package/src/modules/workspace/file-readers/document-reader.ts +64 -0
- package/src/modules/workspace/file-readers/email-reader.ts +21 -0
- package/src/modules/workspace/file-readers/email-text.ts +193 -0
- package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
- package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
- package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
- package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
- package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
- package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
- package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
- package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
- package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
- package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
- package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
- package/src/modules/workspace/file-readers/file-reader.ts +104 -0
- package/src/modules/workspace/file-readers/image-read.ts +122 -0
- package/src/modules/workspace/file-readers/image-reader.ts +39 -0
- package/src/modules/workspace/file-readers/odf-text.ts +123 -0
- package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
- package/src/modules/workspace/file-readers/text-reader.ts +131 -0
- package/src/modules/workspace/startup/__tests__/kb-startup-runner.test.ts +141 -0
- package/src/modules/workspace/startup/kb-git.ts +20 -7
- package/src/modules/workspace/workspace.service.ts +132 -25
- package/src/modules/workspace/workspace.tools.ts +174 -12
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import AdmZip from 'adm-zip';
|
|
2
|
+
import { localBlocks, localElementBlocks, localName, paragraphRunText, zipEntryOversize } from './ooxml-text.js';
|
|
3
|
+
/**
|
|
4
|
+
* Extract the BODY text of a `.docx` (Word) document.
|
|
5
|
+
*
|
|
6
|
+
* A docx is a zip whose main part is `word/document.xml`. v1 extracts the body
|
|
7
|
+
* only — headers/footers are skipped, and the marker summary says so.
|
|
8
|
+
*
|
|
9
|
+
* - Paragraphs become lines. `<w:t>` runs are concatenated with NO separator
|
|
10
|
+
* (Word splits runs mid-word on formatting boundaries).
|
|
11
|
+
* - Tables become lines with cell text tab-separated, one line per row.
|
|
12
|
+
*/
|
|
13
|
+
export function extractDocx(bytes) {
|
|
14
|
+
let xml;
|
|
15
|
+
try {
|
|
16
|
+
const zip = new AdmZip(bytes);
|
|
17
|
+
const entry = zip.getEntry('word/document.xml');
|
|
18
|
+
if (!entry) {
|
|
19
|
+
return { ok: false, message: 'could not be parsed as a .docx (no word/document.xml inside the archive)' };
|
|
20
|
+
}
|
|
21
|
+
// Declared-uncompressed-size bound BEFORE inflation — see zipEntryOversize.
|
|
22
|
+
const oversize = zipEntryOversize(entry);
|
|
23
|
+
if (oversize)
|
|
24
|
+
return { ok: false, message: `could not be extracted as a .docx (${oversize})` };
|
|
25
|
+
xml = entry.getData().toString('utf8');
|
|
26
|
+
}
|
|
27
|
+
catch (err) {
|
|
28
|
+
return { ok: false, message: `could not be parsed as a .docx (${err.message})` };
|
|
29
|
+
}
|
|
30
|
+
// The BODY element read by the parser rather than matched by a regex: a
|
|
31
|
+
// comment or CDATA section mentioning `<w:body>` could answer as the
|
|
32
|
+
// document body and hand back its text instead of the real one.
|
|
33
|
+
const body = localBlocks(xml, 'body')[0] ?? xml;
|
|
34
|
+
const lines = [];
|
|
35
|
+
let paragraphs = 0;
|
|
36
|
+
let tables = 0;
|
|
37
|
+
// Tables and paragraphs in DOCUMENT order, straight from the parser: a match
|
|
38
|
+
// is never descended into, so a table's own paragraphs stay inside it and are
|
|
39
|
+
// rendered once, as its rows. The hand-rolled splitter this replaces counted
|
|
40
|
+
// `<w:tbl>` opens and closes with a regex, which a comment or CDATA section
|
|
41
|
+
// mentioning either could throw off — splitting a paragraph in half and
|
|
42
|
+
// dropping its text.
|
|
43
|
+
for (const block of localElementBlocks(body, ['tbl', 'p'])) {
|
|
44
|
+
if (localName(block.name) === 'tbl') {
|
|
45
|
+
tables++;
|
|
46
|
+
for (const row of localBlocks(block.body ?? '', 'tr')) {
|
|
47
|
+
const cells = localBlocks(row, 'tc').map((cell) => paragraphRunText(cell, 't'));
|
|
48
|
+
lines.push(cells.join(' '));
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
else {
|
|
52
|
+
lines.push(paragraphRunText(block.body ?? '', 't'));
|
|
53
|
+
paragraphs++;
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
while (lines.length > 0 && lines[lines.length - 1].trim() === '')
|
|
57
|
+
lines.pop();
|
|
58
|
+
const parts = [`${paragraphs} paragraph${paragraphs === 1 ? '' : 's'}`];
|
|
59
|
+
if (tables > 0)
|
|
60
|
+
parts.push(`${tables} table${tables === 1 ? '' : 's'}`);
|
|
61
|
+
return {
|
|
62
|
+
ok: true,
|
|
63
|
+
summary: `${parts.join(' + ')}; body only (headers/footers skipped); layout, images and formatting omitted`,
|
|
64
|
+
text: lines.join('\n'),
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
//# sourceMappingURL=extract-docx.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-docx.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-docx.ts"],"names":[],"mappings":"AAAA,OAAO,MAAM,MAAM,SAAS,CAAC;AAE7B,OAAO,EAAE,WAAW,EAAE,kBAAkB,EAAE,SAAS,EAAE,gBAAgB,EAAE,gBAAgB,EAAE,MAAM,iBAAiB,CAAC;AAEjH;;;;;;;;;GASG;AACH,MAAM,UAAU,WAAW,CAAC,KAAa;IACvC,IAAI,GAAW,CAAC;IAChB,IAAI,CAAC;QACH,MAAM,GAAG,GAAG,IAAI,MAAM,CAAC,KAAK,CAAC,CAAC;QAC9B,MAAM,KAAK,GAAG,GAAG,CAAC,QAAQ,CAAC,mBAAmB,CAAC,CAAC;QAChD,IAAI,CAAC,KAAK,EAAE,CAAC;YACX,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,0EAA0E,EAAE,CAAC;QAC5G,CAAC;QACD,4EAA4E;QAC5E,MAAM,QAAQ,GAAG,gBAAgB,CAAC,KAAK,CAAC,CAAC;QACzC,IAAI,QAAQ;YAAE,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,sCAAsC,QAAQ,GAAG,EAAE,CAAC;QAC/F,GAAG,GAAG,KAAK,CAAC,OAAO,EAAE,CAAC,QAAQ,CAAC,MAAM,CAAC,CAAC;IACzC,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,mCAAoC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAC9F,CAAC;IAED,wEAAwE;IACxE,qEAAqE;IACrE,gEAAgE;IAChE,MAAM,IAAI,GAAG,WAAW,CAAC,GAAG,EAAE,MAAM,CAAC,CAAC,CAAC,CAAC,IAAI,GAAG,CAAC;IAEhD,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,UAAU,GAAG,CAAC,CAAC;IACnB,IAAI,MAAM,GAAG,CAAC,CAAC;IACf,6EAA6E;IAC7E,8EAA8E;IAC9E,6EAA6E;IAC7E,4EAA4E;IAC5E,wEAAwE;IACxE,qBAAqB;IACrB,KAAK,MAAM,KAAK,IAAI,kBAAkB,CAAC,IAAI,EAAE,CAAC,KAAK,EAAE,GAAG,CAAC,CAAC,EAAE,CAAC;QAC3D,IAAI,SAAS,CAAC,KAAK,CAAC,IAAI,CAAC,KAAK,KAAK,EAAE,CAAC;YACpC,MAAM,EAAE,CAAC;YACT,KAAK,MAAM,GAAG,IAAI,WAAW,CAAC,KAAK,CAAC,IAAI,IAAI,EAAE,EAAE,IAAI,CAAC,EAAE,CAAC;gBACtD,MAAM,KAAK,GAAG,WAAW,CAAC,GAAG,EAAE,IAAI,CAAC,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,gBAAgB,CAAC,IAAI,EAAE,GAAG,CAAC,CAAC,CAAC;gBAChF,KAAK,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC,CAAC;YAC9B,CAAC;QACH,CAAC;aAAM,CAAC;YACN,KAAK,CAAC,IAAI,CAAC,gBAAgB,CAAC,KAAK,CAAC,IAAI,IAAI,EAAE,EAAE,GAAG,CAAC,CAAC,CAAC;YACpD,UAAU,EAAE,CAAC;QACf,CAAC;IACH,CAAC;IACD,OAAO,KAAK,CAAC,MAAM,GAAG,CAAC,IAAI,KAAK,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC,IAAI,EAAE,KAAK,EAAE;QAAE,KAAK,CAAC,GAAG,EAAE,CAAC;IAE9E,MAAM,KAAK,GAAG,CAAC,GAAG,UAAU,aAAa,UAAU,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC,CAAC;IACxE,IAAI,MAAM,GAAG,CAAC;QAAE,KAAK,CAAC,IAAI,CAAC,GAAG,MAAM,SAAS,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,EAAE,CAAC,CAAC;IACxE,OAAO;QACL,EAAE,EAAE,IAAI;QACR,OAAO,EAAE,GAAG,KAAK,CAAC,IAAI,CAAC,KAAK,CAAC,8EAA8E;QAC3G,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC;KACvB,CAAC;AACJ,CAAC"}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Extract a `.eml` (RFC 822 / MIME) email into the shared email text shape
|
|
4
|
+
* (see `email-text.ts`).
|
|
5
|
+
*
|
|
6
|
+
* Parsing is `postal-mime` (postalsys, MIT-0): small, dependency-free, ESM,
|
|
7
|
+
* and the same code runs in the browser — which keeps the frontend viewer's
|
|
8
|
+
* story identical to the agent's. It normalizes encoded-word headers and
|
|
9
|
+
* multipart bodies, and hands the `Date:` header over as ISO when it parses.
|
|
10
|
+
*
|
|
11
|
+
* postal-mime is LENIENT — random bytes "parse" to an empty message rather
|
|
12
|
+
* than throwing — so recognizability is checked here: a file yielding NO
|
|
13
|
+
* email headers at all (no From/To/Cc/Bcc/Subject/Date/Message-ID) and no
|
|
14
|
+
* attachments is not an email, and gets the typed could-not-be-parsed
|
|
15
|
+
* failure instead of an empty extraction.
|
|
16
|
+
*/
|
|
17
|
+
export declare function extractEml(bytes: Buffer): Promise<ExtractResult>;
|
|
18
|
+
//# sourceMappingURL=extract-eml.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-eml.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-eml.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAI5D;;;;;;;;;;;;;;GAcG;AACH,wBAAsB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,OAAO,CAAC,aAAa,CAAC,CA4BtE"}
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import PostalMime from 'postal-mime';
|
|
2
|
+
import { emailExtraction, htmlToEmailText } from './email-text.js';
|
|
3
|
+
import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
|
|
4
|
+
/**
|
|
5
|
+
* Extract a `.eml` (RFC 822 / MIME) email into the shared email text shape
|
|
6
|
+
* (see `email-text.ts`).
|
|
7
|
+
*
|
|
8
|
+
* Parsing is `postal-mime` (postalsys, MIT-0): small, dependency-free, ESM,
|
|
9
|
+
* and the same code runs in the browser — which keeps the frontend viewer's
|
|
10
|
+
* story identical to the agent's. It normalizes encoded-word headers and
|
|
11
|
+
* multipart bodies, and hands the `Date:` header over as ISO when it parses.
|
|
12
|
+
*
|
|
13
|
+
* postal-mime is LENIENT — random bytes "parse" to an empty message rather
|
|
14
|
+
* than throwing — so recognizability is checked here: a file yielding NO
|
|
15
|
+
* email headers at all (no From/To/Cc/Bcc/Subject/Date/Message-ID) and no
|
|
16
|
+
* attachments is not an email, and gets the typed could-not-be-parsed
|
|
17
|
+
* failure instead of an empty extraction.
|
|
18
|
+
*/
|
|
19
|
+
export async function extractEml(bytes) {
|
|
20
|
+
// The same raw-size bound the PDF extractor applies: no container to
|
|
21
|
+
// pre-scan, so the bound is the file's size, checked before parsing.
|
|
22
|
+
if (bytes.length > MAX_DOC_PART_BYTES) {
|
|
23
|
+
return {
|
|
24
|
+
ok: false,
|
|
25
|
+
message: `could not be extracted as a .eml (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
let email;
|
|
29
|
+
try {
|
|
30
|
+
email = await PostalMime.parse(bytes);
|
|
31
|
+
}
|
|
32
|
+
catch (err) {
|
|
33
|
+
return { ok: false, message: `could not be parsed as a .eml (${err.message})` };
|
|
34
|
+
}
|
|
35
|
+
if (email.from === undefined &&
|
|
36
|
+
email.to === undefined &&
|
|
37
|
+
email.cc === undefined &&
|
|
38
|
+
email.bcc === undefined &&
|
|
39
|
+
email.subject === undefined &&
|
|
40
|
+
email.date === undefined &&
|
|
41
|
+
email.messageId === undefined &&
|
|
42
|
+
email.attachments.length === 0) {
|
|
43
|
+
return { ok: false, message: 'could not be parsed as a .eml (no email headers found)' };
|
|
44
|
+
}
|
|
45
|
+
return { ok: true, ...emailExtraction(emlModel(email)) };
|
|
46
|
+
}
|
|
47
|
+
/** postal-mime's parse, shaped into the format-independent email model. */
|
|
48
|
+
function emlModel(email) {
|
|
49
|
+
const text = email.text !== undefined && email.text.trim() !== '' ? email.text : undefined;
|
|
50
|
+
const html = email.html !== undefined && email.html.trim() !== '' ? email.html : undefined;
|
|
51
|
+
const body = text ?? (html !== undefined ? htmlToEmailText(html) : '');
|
|
52
|
+
return {
|
|
53
|
+
from: email.from && addressListText([email.from]),
|
|
54
|
+
to: email.to && addressListText(email.to),
|
|
55
|
+
cc: email.cc && addressListText(email.cc),
|
|
56
|
+
bcc: email.bcc && addressListText(email.bcc),
|
|
57
|
+
subject: email.subject,
|
|
58
|
+
date: email.date !== undefined ? isoDate(email.date) : undefined,
|
|
59
|
+
body: body.replace(/\s+$/, ''),
|
|
60
|
+
bodySource: text !== undefined ? 'text' : html !== undefined ? 'html' : 'none',
|
|
61
|
+
attachments: email.attachments.map((a) => ({
|
|
62
|
+
name: a.filename ?? 'unnamed attachment',
|
|
63
|
+
mimeType: a.mimeType,
|
|
64
|
+
sizeBytes: typeof a.content === 'string' ? Buffer.byteLength(a.content) : a.content.byteLength,
|
|
65
|
+
})),
|
|
66
|
+
};
|
|
67
|
+
}
|
|
68
|
+
/** `Name <addr>, addr2, Group: member, member` — groups flattened inline. */
|
|
69
|
+
function addressListText(list) {
|
|
70
|
+
const s = list
|
|
71
|
+
.map(function one(a) {
|
|
72
|
+
if (a.group !== undefined)
|
|
73
|
+
return `${a.name}: ${a.group.map(one).join(', ')}`;
|
|
74
|
+
if (a.address === undefined || a.address === '')
|
|
75
|
+
return a.name;
|
|
76
|
+
return a.name !== '' && a.name !== a.address ? `${a.name} <${a.address}>` : a.address;
|
|
77
|
+
})
|
|
78
|
+
.filter((t) => t !== '')
|
|
79
|
+
.join(', ');
|
|
80
|
+
return s === '' ? undefined : s;
|
|
81
|
+
}
|
|
82
|
+
/** postal-mime already normalizes parseable dates to ISO; keep the raw value when it could not. */
|
|
83
|
+
function isoDate(value) {
|
|
84
|
+
const d = new Date(value);
|
|
85
|
+
return Number.isNaN(d.getTime()) ? value : d.toISOString();
|
|
86
|
+
}
|
|
87
|
+
//# sourceMappingURL=extract-eml.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-eml.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-eml.ts"],"names":[],"mappings":"AAAA,OAAO,UAAwC,MAAM,aAAa,CAAC;AAEnE,OAAO,EAAE,eAAe,EAAE,eAAe,EAAyC,MAAM,iBAAiB,CAAC;AAC1G,OAAO,EAAE,kBAAkB,EAAE,MAAM,iBAAiB,CAAC;AAErD;;;;;;;;;;;;;;GAcG;AACH,MAAM,CAAC,KAAK,UAAU,UAAU,CAAC,KAAa;IAC5C,qEAAqE;IACrE,qEAAqE;IACrE,IAAI,KAAK,CAAC,MAAM,GAAG,kBAAkB,EAAE,CAAC;QACtC,OAAO;YACL,EAAE,EAAE,KAAK;YACT,OAAO,EAAE,iDAAiD,KAAK,CAAC,MAAM,qBAAqB,kBAAkB,iCAAiC;SAC/I,CAAC;IACJ,CAAC;IACD,IAAI,KAAY,CAAC;IACjB,IAAI,CAAC;QACH,KAAK,GAAG,MAAM,UAAU,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC;IACxC,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,kCAAmC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAC7F,CAAC;IACD,IACE,KAAK,CAAC,IAAI,KAAK,SAAS;QACxB,KAAK,CAAC,EAAE,KAAK,SAAS;QACtB,KAAK,CAAC,EAAE,KAAK,SAAS;QACtB,KAAK,CAAC,GAAG,KAAK,SAAS;QACvB,KAAK,CAAC,OAAO,KAAK,SAAS;QAC3B,KAAK,CAAC,IAAI,KAAK,SAAS;QACxB,KAAK,CAAC,SAAS,KAAK,SAAS;QAC7B,KAAK,CAAC,WAAW,CAAC,MAAM,KAAK,CAAC,EAC9B,CAAC;QACD,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,wDAAwD,EAAE,CAAC;IAC1F,CAAC;IACD,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,GAAG,eAAe,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC;AAC3D,CAAC;AAED,2EAA2E;AAC3E,SAAS,QAAQ,CAAC,KAAY;IAC5B,MAAM,IAAI,GAAG,KAAK,CAAC,IAAI,KAAK,SAAS,IAAI,KAAK,CAAC,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE,CAAC,CAAC,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,SAAS,CAAC;IAC3F,MAAM,IAAI,GAAG,KAAK,CAAC,IAAI,KAAK,SAAS,IAAI,KAAK,CAAC,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE,CAAC,CAAC,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,SAAS,CAAC;IAC3F,MAAM,IAAI,GAAG,IAAI,IAAI,CAAC,IAAI,KAAK,SAAS,CAAC,CAAC,CAAC,eAAe,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC;IACvE,OAAO;QACL,IAAI,EAAE,KAAK,CAAC,IAAI,IAAI,eAAe,CAAC,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC;QACjD,EAAE,EAAE,KAAK,CAAC,EAAE,IAAI,eAAe,CAAC,KAAK,CAAC,EAAE,CAAC;QACzC,EAAE,EAAE,KAAK,CAAC,EAAE,IAAI,eAAe,CAAC,KAAK,CAAC,EAAE,CAAC;QACzC,GAAG,EAAE,KAAK,CAAC,GAAG,IAAI,eAAe,CAAC,KAAK,CAAC,GAAG,CAAC;QAC5C,OAAO,EAAE,KAAK,CAAC,OAAO;QACtB,IAAI,EAAE,KAAK,CAAC,IAAI,KAAK,SAAS,CAAC,CAAC,CAAC,OAAO,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,SAAS;QAChE,IAAI,EAAE,IAAI,CAAC,OAAO,CAAC,MAAM,EAAE,EAAE,CAAC;QAC9B,UAAU,EAAE,IAAI,KAAK,SAAS,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,IAAI,KAAK,SAAS,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,MAAM;QAC9E,WAAW,EAAE,KAAK,CAAC,WAAW,CAAC,GAAG,CAChC,CAAC,CAAC,EAAmB,EAAE,CAAC,CAAC;YACvB,IAAI,EAAE,CAAC,CAAC,QAAQ,IAAI,oBAAoB;YACxC,QAAQ,EAAE,CAAC,CAAC,QAAQ;YACpB,SAAS,EAAE,OAAO,CAAC,CAAC,OAAO,KAAK,QAAQ,CAAC,CAAC,CAAC,MAAM,CAAC,UAAU,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,UAAU;SAC/F,CAAC,CACH;KACF,CAAC;AACJ,CAAC;AAED,6EAA6E;AAC7E,SAAS,eAAe,CAAC,IAAe;IACtC,MAAM,CAAC,GAAG,IAAI;SACX,GAAG,CAAC,SAAS,GAAG,CAAC,CAAU;QAC1B,IAAI,CAAC,CAAC,KAAK,KAAK,SAAS;YAAE,OAAO,GAAG,CAAC,CAAC,IAAI,KAAK,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;QAC9E,IAAI,CAAC,CAAC,OAAO,KAAK,SAAS,IAAI,CAAC,CAAC,OAAO,KAAK,EAAE;YAAE,OAAO,CAAC,CAAC,IAAI,CAAC;QAC/D,OAAO,CAAC,CAAC,IAAI,KAAK,EAAE,IAAI,CAAC,CAAC,IAAI,KAAK,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,IAAI,KAAK,CAAC,CAAC,OAAO,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC;IACxF,CAAC,CAAC;SACD,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,KAAK,EAAE,CAAC;SACvB,IAAI,CAAC,IAAI,CAAC,CAAC;IACd,OAAO,CAAC,KAAK,EAAE,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC;AAClC,CAAC;AAED,mGAAmG;AACnG,SAAS,OAAO,CAAC,KAAa;IAC5B,MAAM,CAAC,GAAG,IAAI,IAAI,CAAC,KAAK,CAAC,CAAC;IAC1B,OAAO,MAAM,CAAC,KAAK,CAAC,CAAC,CAAC,OAAO,EAAE,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC;AAC7D,CAAC"}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Extract a `.msg` (Outlook item, CFB container) email into the shared email
|
|
4
|
+
* text shape (see `email-text.ts`).
|
|
5
|
+
*
|
|
6
|
+
* Parsing is `@kenjiuno/msgreader` (HiraokaHyperTools, Apache-2.0) — the
|
|
7
|
+
* maintained MAPI/CFB reader. It never throws for bad content of its own
|
|
8
|
+
* accord: unparseable bytes come back as `{ dataType: null, error }`, which
|
|
9
|
+
* maps onto the typed could-not-be-parsed failure here.
|
|
10
|
+
*
|
|
11
|
+
* Body preference mirrors `.eml`: the plain-text `PidTagBody` first, an HTML
|
|
12
|
+
* body stripped to text second. An Outlook item whose body exists ONLY as
|
|
13
|
+
* compressed RTF is degraded honestly — the extraction says
|
|
14
|
+
* "[body is RTF; no plain-text part]" instead of pretending to decode RTF.
|
|
15
|
+
*/
|
|
16
|
+
export declare function extractMsg(bytes: Buffer): ExtractResult;
|
|
17
|
+
//# sourceMappingURL=extract-msg.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-msg.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-msg.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAI5D;;;;;;;;;;;;;GAaG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CAsBvD"}
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
import MsgReaderImport from '@kenjiuno/msgreader';
|
|
2
|
+
import { emailExtraction, htmlToEmailText } from './email-text.js';
|
|
3
|
+
import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
|
|
4
|
+
/**
|
|
5
|
+
* Extract a `.msg` (Outlook item, CFB container) email into the shared email
|
|
6
|
+
* text shape (see `email-text.ts`).
|
|
7
|
+
*
|
|
8
|
+
* Parsing is `@kenjiuno/msgreader` (HiraokaHyperTools, Apache-2.0) — the
|
|
9
|
+
* maintained MAPI/CFB reader. It never throws for bad content of its own
|
|
10
|
+
* accord: unparseable bytes come back as `{ dataType: null, error }`, which
|
|
11
|
+
* maps onto the typed could-not-be-parsed failure here.
|
|
12
|
+
*
|
|
13
|
+
* Body preference mirrors `.eml`: the plain-text `PidTagBody` first, an HTML
|
|
14
|
+
* body stripped to text second. An Outlook item whose body exists ONLY as
|
|
15
|
+
* compressed RTF is degraded honestly — the extraction says
|
|
16
|
+
* "[body is RTF; no plain-text part]" instead of pretending to decode RTF.
|
|
17
|
+
*/
|
|
18
|
+
export function extractMsg(bytes) {
|
|
19
|
+
if (bytes.length > MAX_DOC_PART_BYTES) {
|
|
20
|
+
return {
|
|
21
|
+
ok: false,
|
|
22
|
+
message: `could not be extracted as a .msg (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
|
|
23
|
+
};
|
|
24
|
+
}
|
|
25
|
+
let fields;
|
|
26
|
+
try {
|
|
27
|
+
// DataView over the Buffer's exact region — no copy, and msgreader never
|
|
28
|
+
// sees bytes outside the file.
|
|
29
|
+
fields = new MsgReader(new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength)).getFileData();
|
|
30
|
+
}
|
|
31
|
+
catch (err) {
|
|
32
|
+
return { ok: false, message: `could not be parsed as a .msg (${err.message})` };
|
|
33
|
+
}
|
|
34
|
+
if (fields.dataType !== 'msg') {
|
|
35
|
+
return {
|
|
36
|
+
ok: false,
|
|
37
|
+
message: `could not be parsed as a .msg (${fields.error ?? 'not an Outlook message file'})`,
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
return { ok: true, ...emailExtraction(msgModel(fields)) };
|
|
41
|
+
}
|
|
42
|
+
/** msgreader's field data, shaped into the format-independent email model. */
|
|
43
|
+
function msgModel(fields) {
|
|
44
|
+
const recipients = fields.recipients ?? [];
|
|
45
|
+
const text = fields.body !== undefined && fields.body.trim() !== '' ? fields.body : undefined;
|
|
46
|
+
const html = htmlBody(fields);
|
|
47
|
+
const body = text ?? (html !== undefined ? htmlToEmailText(html) : '');
|
|
48
|
+
const bodySource = text !== undefined ? 'text' : html !== undefined ? 'html' : fields.compressedRtf !== undefined ? 'rtf-only' : 'none';
|
|
49
|
+
const date = fields.clientSubmitTime ?? fields.messageDeliveryTime;
|
|
50
|
+
return {
|
|
51
|
+
from: mailboxText(fields.senderName, fields.senderSmtpAddress ?? fields.senderEmail),
|
|
52
|
+
to: recipientList(recipients, 'to'),
|
|
53
|
+
cc: recipientList(recipients, 'cc'),
|
|
54
|
+
bcc: recipientList(recipients, 'bcc'),
|
|
55
|
+
subject: fields.subject,
|
|
56
|
+
date: date !== undefined ? isoDate(date) : undefined,
|
|
57
|
+
body: body.replace(/\s+$/, ''),
|
|
58
|
+
bodySource,
|
|
59
|
+
attachments: (fields.attachments ?? []).map((a) => ({
|
|
60
|
+
name: a.fileName ?? a.fileNameShort ?? a.name ?? 'unnamed attachment',
|
|
61
|
+
mimeType: a.attachMimeTag,
|
|
62
|
+
sizeBytes: a.contentLength,
|
|
63
|
+
})),
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
/** The HTML body, whichever MAPI property carries it (string, or utf-8 bytes). */
|
|
67
|
+
function htmlBody(fields) {
|
|
68
|
+
if (fields.bodyHtml !== undefined && fields.bodyHtml.trim() !== '')
|
|
69
|
+
return fields.bodyHtml;
|
|
70
|
+
if (fields.html instanceof Uint8Array && fields.html.length > 0) {
|
|
71
|
+
return Buffer.from(fields.html).toString('utf8');
|
|
72
|
+
}
|
|
73
|
+
return undefined;
|
|
74
|
+
}
|
|
75
|
+
/** `Name <addr>` / `Name` / `addr` — whatever the message carries. */
|
|
76
|
+
function mailboxText(name, address) {
|
|
77
|
+
const n = name?.trim() ?? '';
|
|
78
|
+
const a = address?.trim() ?? '';
|
|
79
|
+
if (n !== '' && a !== '' && n !== a)
|
|
80
|
+
return `${n} <${a}>`;
|
|
81
|
+
if (a !== '')
|
|
82
|
+
return a;
|
|
83
|
+
return n !== '' ? n : undefined;
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* The bucket a recipient belongs to. msgreader maps the MAPI `PidTagRecipientType`
|
|
87
|
+
* values 1/2/3 to these strings itself (lib/MsgReader.js, the `recipType` case)
|
|
88
|
+
* — but ONLY those three: any other raw PT_LONG value, e.g. `MAPI_TO | MAPI_P1`
|
|
89
|
+
* (0x10000001) on a resubmitted message, leaks through as a NUMBER despite the
|
|
90
|
+
* `'to' | 'cc' | 'bcc'` typing. Mask the resubmit/submitted flag bits and remap
|
|
91
|
+
* so such a recipient keeps its line instead of vanishing; anything else
|
|
92
|
+
* (including untyped) counts as `to`, matching the frontend's `msgMessage.ts`.
|
|
93
|
+
*/
|
|
94
|
+
function recipientBucket(recipType) {
|
|
95
|
+
if (recipType === 'to' || recipType === 'cc' || recipType === 'bcc')
|
|
96
|
+
return recipType;
|
|
97
|
+
if (typeof recipType === 'number') {
|
|
98
|
+
const base = recipType & 0x0fffffff; // strip MAPI_SUBMITTED (0x80000000) / MAPI_P1 (0x10000000)
|
|
99
|
+
if (base === 2)
|
|
100
|
+
return 'cc';
|
|
101
|
+
if (base === 3)
|
|
102
|
+
return 'bcc';
|
|
103
|
+
}
|
|
104
|
+
return 'to';
|
|
105
|
+
}
|
|
106
|
+
/** The comma-joined mailboxes of one recipient type. Untyped recipients count as `to`. */
|
|
107
|
+
function recipientList(recipients, type) {
|
|
108
|
+
const s = recipients
|
|
109
|
+
.filter((r) => recipientBucket(r.recipType) === type)
|
|
110
|
+
.map((r) => mailboxText(r.name, r.smtpAddress ?? r.email))
|
|
111
|
+
.filter((t) => t !== undefined)
|
|
112
|
+
.join(', ');
|
|
113
|
+
return s === '' ? undefined : s;
|
|
114
|
+
}
|
|
115
|
+
/** msgreader emits RFC-1123 GMT strings; normalize to ISO, keep raw when unparseable. */
|
|
116
|
+
function isoDate(value) {
|
|
117
|
+
const d = new Date(value);
|
|
118
|
+
return Number.isNaN(d.getTime()) ? value : d.toISOString();
|
|
119
|
+
}
|
|
120
|
+
const MsgReader = MsgReaderImport.default ?? MsgReaderImport;
|
|
121
|
+
//# sourceMappingURL=extract-msg.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-msg.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-msg.ts"],"names":[],"mappings":"AAAA,OAAO,eAAe,MAAM,qBAAqB,CAAC;AAElD,OAAO,EAAE,eAAe,EAAE,eAAe,EAAyC,MAAM,iBAAiB,CAAC;AAC1G,OAAO,EAAE,kBAAkB,EAAE,MAAM,iBAAiB,CAAC;AAErD;;;;;;;;;;;;;GAaG;AACH,MAAM,UAAU,UAAU,CAAC,KAAa;IACtC,IAAI,KAAK,CAAC,MAAM,GAAG,kBAAkB,EAAE,CAAC;QACtC,OAAO;YACL,EAAE,EAAE,KAAK;YACT,OAAO,EAAE,iDAAiD,KAAK,CAAC,MAAM,qBAAqB,kBAAkB,iCAAiC;SAC/I,CAAC;IACJ,CAAC;IACD,IAAI,MAAkB,CAAC;IACvB,IAAI,CAAC;QACH,yEAAyE;QACzE,+BAA+B;QAC/B,MAAM,GAAG,IAAI,SAAS,CAAC,IAAI,QAAQ,CAAC,KAAK,CAAC,MAAM,EAAE,KAAK,CAAC,UAAU,EAAE,KAAK,CAAC,UAAU,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC;IACvG,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,kCAAmC,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAC7F,CAAC;IACD,IAAI,MAAM,CAAC,QAAQ,KAAK,KAAK,EAAE,CAAC;QAC9B,OAAO;YACL,EAAE,EAAE,KAAK;YACT,OAAO,EAAE,kCAAkC,MAAM,CAAC,KAAK,IAAI,6BAA6B,GAAG;SAC5F,CAAC;IACJ,CAAC;IACD,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,GAAG,eAAe,CAAC,QAAQ,CAAC,MAAM,CAAC,CAAC,EAAE,CAAC;AAC5D,CAAC;AAED,8EAA8E;AAC9E,SAAS,QAAQ,CAAC,MAAkB;IAClC,MAAM,UAAU,GAAG,MAAM,CAAC,UAAU,IAAI,EAAE,CAAC;IAC3C,MAAM,IAAI,GAAG,MAAM,CAAC,IAAI,KAAK,SAAS,IAAI,MAAM,CAAC,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE,CAAC,CAAC,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC,CAAC,SAAS,CAAC;IAC9F,MAAM,IAAI,GAAG,QAAQ,CAAC,MAAM,CAAC,CAAC;IAC9B,MAAM,IAAI,GAAG,IAAI,IAAI,CAAC,IAAI,KAAK,SAAS,CAAC,CAAC,CAAC,eAAe,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC;IACvE,MAAM,UAAU,GACd,IAAI,KAAK,SAAS,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,IAAI,KAAK,SAAS,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,MAAM,CAAC,aAAa,KAAK,SAAS,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,MAAM,CAAC;IACvH,MAAM,IAAI,GAAG,MAAM,CAAC,gBAAgB,IAAI,MAAM,CAAC,mBAAmB,CAAC;IACnE,OAAO;QACL,IAAI,EAAE,WAAW,CAAC,MAAM,CAAC,UAAU,EAAE,MAAM,CAAC,iBAAiB,IAAI,MAAM,CAAC,WAAW,CAAC;QACpF,EAAE,EAAE,aAAa,CAAC,UAAU,EAAE,IAAI,CAAC;QACnC,EAAE,EAAE,aAAa,CAAC,UAAU,EAAE,IAAI,CAAC;QACnC,GAAG,EAAE,aAAa,CAAC,UAAU,EAAE,KAAK,CAAC;QACrC,OAAO,EAAE,MAAM,CAAC,OAAO;QACvB,IAAI,EAAE,IAAI,KAAK,SAAS,CAAC,CAAC,CAAC,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,SAAS;QACpD,IAAI,EAAE,IAAI,CAAC,OAAO,CAAC,MAAM,EAAE,EAAE,CAAC;QAC9B,UAAU;QACV,WAAW,EAAE,CAAC,MAAM,CAAC,WAAW,IAAI,EAAE,CAAC,CAAC,GAAG,CACzC,CAAC,CAAC,EAAmB,EAAE,CAAC,CAAC;YACvB,IAAI,EAAE,CAAC,CAAC,QAAQ,IAAI,CAAC,CAAC,aAAa,IAAI,CAAC,CAAC,IAAI,IAAI,oBAAoB;YACrE,QAAQ,EAAE,CAAC,CAAC,aAAa;YACzB,SAAS,EAAE,CAAC,CAAC,aAAa;SAC3B,CAAC,CACH;KACF,CAAC;AACJ,CAAC;AAED,kFAAkF;AAClF,SAAS,QAAQ,CAAC,MAAkB;IAClC,IAAI,MAAM,CAAC,QAAQ,KAAK,SAAS,IAAI,MAAM,CAAC,QAAQ,CAAC,IAAI,EAAE,KAAK,EAAE;QAAE,OAAO,MAAM,CAAC,QAAQ,CAAC;IAC3F,IAAI,MAAM,CAAC,IAAI,YAAY,UAAU,IAAI,MAAM,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAChE,OAAO,MAAM,CAAC,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC,QAAQ,CAAC,MAAM,CAAC,CAAC;IACnD,CAAC;IACD,OAAO,SAAS,CAAC;AACnB,CAAC;AAED,sEAAsE;AACtE,SAAS,WAAW,CAAC,IAAwB,EAAE,OAA2B;IACxE,MAAM,CAAC,GAAG,IAAI,EAAE,IAAI,EAAE,IAAI,EAAE,CAAC;IAC7B,MAAM,CAAC,GAAG,OAAO,EAAE,IAAI,EAAE,IAAI,EAAE,CAAC;IAChC,IAAI,CAAC,KAAK,EAAE,IAAI,CAAC,KAAK,EAAE,IAAI,CAAC,KAAK,CAAC;QAAE,OAAO,GAAG,CAAC,KAAK,CAAC,GAAG,CAAC;IAC1D,IAAI,CAAC,KAAK,EAAE;QAAE,OAAO,CAAC,CAAC;IACvB,OAAO,CAAC,KAAK,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC;AAClC,CAAC;AAED;;;;;;;;GAQG;AACH,SAAS,eAAe,CAAC,SAAkB;IACzC,IAAI,SAAS,KAAK,IAAI,IAAI,SAAS,KAAK,IAAI,IAAI,SAAS,KAAK,KAAK;QAAE,OAAO,SAAS,CAAC;IACtF,IAAI,OAAO,SAAS,KAAK,QAAQ,EAAE,CAAC;QAClC,MAAM,IAAI,GAAG,SAAS,GAAG,UAAU,CAAC,CAAC,2DAA2D;QAChG,IAAI,IAAI,KAAK,CAAC;YAAE,OAAO,IAAI,CAAC;QAC5B,IAAI,IAAI,KAAK,CAAC;YAAE,OAAO,KAAK,CAAC;IAC/B,CAAC;IACD,OAAO,IAAI,CAAC;AACd,CAAC;AAED,0FAA0F;AAC1F,SAAS,aAAa,CAAC,UAAiC,EAAE,IAAyB;IACjF,MAAM,CAAC,GAAG,UAAU;SACjB,MAAM,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,eAAe,CAAC,CAAC,CAAC,SAAS,CAAC,KAAK,IAAI,CAAC;SACpD,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,WAAW,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC,WAAW,IAAI,CAAC,CAAC,KAAK,CAAC,CAAC;SACzD,MAAM,CAAC,CAAC,CAAC,EAAe,EAAE,CAAC,CAAC,KAAK,SAAS,CAAC;SAC3C,IAAI,CAAC,IAAI,CAAC,CAAC;IACd,OAAO,CAAC,KAAK,EAAE,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC;AAClC,CAAC;AAED,yFAAyF;AACzF,SAAS,OAAO,CAAC,KAAa;IAC5B,MAAM,CAAC,GAAG,IAAI,IAAI,CAAC,KAAK,CAAC,CAAC;IAC1B,OAAO,MAAM,CAAC,KAAK,CAAC,CAAC,CAAC,OAAO,EAAE,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC;AAC7D,CAAC;AASD,MAAM,SAAS,GACZ,eAA2D,CAAC,OAAO,IAAI,eAAe,CAAC"}
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Extract the text of a `.odp` (OpenDocument Presentation) deck.
|
|
4
|
+
*
|
|
5
|
+
* Slides are the `<draw:page>` elements of `content.xml`, in DOCUMENT order —
|
|
6
|
+
* ODF orders slides in the file itself, so unlike pptx there is no numeric
|
|
7
|
+
* filename sort. Each slide is emitted under a `[slide N]` marker (N = 1-based
|
|
8
|
+
* position); speaker notes (`<presentation:notes>` inside the page) follow
|
|
9
|
+
* under `[slide N notes]` when non-empty. Within a slide, each `<text:p>` in
|
|
10
|
+
* its frames is a line; spans concatenate with no separator.
|
|
11
|
+
*/
|
|
12
|
+
export declare function extractOdp(bytes: Buffer): ExtractResult;
|
|
13
|
+
//# sourceMappingURL=extract-odp.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-odp.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-odp.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAI5D;;;;;;;;;GASG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CAmCvD"}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import { odfParagraphLines, readOdfContentXml } from './odf-text.js';
|
|
2
|
+
import { localBlocks, localElementBlocks, removeLocalElements } from './ooxml-text.js';
|
|
3
|
+
/**
|
|
4
|
+
* Extract the text of a `.odp` (OpenDocument Presentation) deck.
|
|
5
|
+
*
|
|
6
|
+
* Slides are the `<draw:page>` elements of `content.xml`, in DOCUMENT order —
|
|
7
|
+
* ODF orders slides in the file itself, so unlike pptx there is no numeric
|
|
8
|
+
* filename sort. Each slide is emitted under a `[slide N]` marker (N = 1-based
|
|
9
|
+
* position); speaker notes (`<presentation:notes>` inside the page) follow
|
|
10
|
+
* under `[slide N notes]` when non-empty. Within a slide, each `<text:p>` in
|
|
11
|
+
* its frames is a line; spans concatenate with no separator.
|
|
12
|
+
*/
|
|
13
|
+
export function extractOdp(bytes) {
|
|
14
|
+
const content = readOdfContentXml(bytes, '.odp');
|
|
15
|
+
if (!content.ok)
|
|
16
|
+
return content;
|
|
17
|
+
const pages = drawPageBlocks(content.xml);
|
|
18
|
+
if (pages.length === 0) {
|
|
19
|
+
return { ok: false, message: 'could not be parsed as a .odp (no draw:page elements in content.xml)' };
|
|
20
|
+
}
|
|
21
|
+
const lines = [];
|
|
22
|
+
let anyNotes = false;
|
|
23
|
+
pages.forEach((page, i) => {
|
|
24
|
+
// Split the notes part out FIRST so its paragraphs don't render as slide text.
|
|
25
|
+
// Notes read by the parser and matched on their LOCAL name: a comment that
|
|
26
|
+
// resembled `<presentation:notes>` used to be emitted as real speaker notes,
|
|
27
|
+
// and a deck binding the presentation namespace to another prefix had none.
|
|
28
|
+
const notesXml = localBlocks(page, 'notes')[0] ?? '';
|
|
29
|
+
// The slide's own text is the page with the notes ELEMENTS removed by
|
|
30
|
+
// their parsed boundaries — global string replacement of the notes BODY
|
|
31
|
+
// also deleted slide text that happened to serialize identically to it.
|
|
32
|
+
const slideXml = removeLocalElements(page, ['notes']);
|
|
33
|
+
lines.push(`[slide ${i + 1}]`);
|
|
34
|
+
lines.push(...odfParagraphLines(slideXml));
|
|
35
|
+
const noteLines = odfParagraphLines(notesXml);
|
|
36
|
+
if (noteLines.length > 0) {
|
|
37
|
+
anyNotes = true;
|
|
38
|
+
lines.push(`[slide ${i + 1} notes]`);
|
|
39
|
+
lines.push(...noteLines);
|
|
40
|
+
}
|
|
41
|
+
});
|
|
42
|
+
return {
|
|
43
|
+
ok: true,
|
|
44
|
+
summary: `${pages.length} slide${pages.length === 1 ? '' : 's'}${anyNotes ? ' + notes' : ''}; layout, images and formatting omitted`,
|
|
45
|
+
text: lines.join('\n'),
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
/**
|
|
49
|
+
* The `<draw:page>…</draw:page>` bodies in document order (pages never nest).
|
|
50
|
+
* A SELF-CLOSING `<draw:page/>` is a legal, fully blank slide — it yields ''
|
|
51
|
+
* so the deck's numbering (and a deliberately blank deck) stays correct.
|
|
52
|
+
*/
|
|
53
|
+
function drawPageBlocks(xml) {
|
|
54
|
+
// The shared quote-aware scanner: a `/>` INSIDE a quoted attribute value (a
|
|
55
|
+
// page name like `a/>b`) is part of the value, never the self-closing
|
|
56
|
+
// delimiter, and a page whose close tag is missing costs one scan of the
|
|
57
|
+
// document rather than one per opener (see `xmlElementBlocks`).
|
|
58
|
+
return localElementBlocks(xml, ['page']).map((e) => e.body ?? '');
|
|
59
|
+
}
|
|
60
|
+
//# sourceMappingURL=extract-odp.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-odp.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-odp.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,iBAAiB,EAAE,iBAAiB,EAAE,MAAM,eAAe,CAAC;AACrE,OAAO,EAAE,WAAW,EAAE,kBAAkB,EAAE,mBAAmB,EAAE,MAAM,iBAAiB,CAAC;AAEvF;;;;;;;;;GASG;AACH,MAAM,UAAU,UAAU,CAAC,KAAa;IACtC,MAAM,OAAO,GAAG,iBAAiB,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC;IACjD,IAAI,CAAC,OAAO,CAAC,EAAE;QAAE,OAAO,OAAO,CAAC;IAEhC,MAAM,KAAK,GAAG,cAAc,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC;IAC1C,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACvB,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,sEAAsE,EAAE,CAAC;IACxG,CAAC;IAED,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,QAAQ,GAAG,KAAK,CAAC;IACrB,KAAK,CAAC,OAAO,CAAC,CAAC,IAAI,EAAE,CAAC,EAAE,EAAE;QACxB,+EAA+E;QAC/E,2EAA2E;QAC3E,6EAA6E;QAC7E,4EAA4E;QAC5E,MAAM,QAAQ,GAAG,WAAW,CAAC,IAAI,EAAE,OAAO,CAAC,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;QACrD,sEAAsE;QACtE,wEAAwE;QACxE,wEAAwE;QACxE,MAAM,QAAQ,GAAG,mBAAmB,CAAC,IAAI,EAAE,CAAC,OAAO,CAAC,CAAC,CAAC;QACtD,KAAK,CAAC,IAAI,CAAC,UAAU,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;QAC/B,KAAK,CAAC,IAAI,CAAC,GAAG,iBAAiB,CAAC,QAAQ,CAAC,CAAC,CAAC;QAC3C,MAAM,SAAS,GAAG,iBAAiB,CAAC,QAAQ,CAAC,CAAC;QAC9C,IAAI,SAAS,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YACzB,QAAQ,GAAG,IAAI,CAAC;YAChB,KAAK,CAAC,IAAI,CAAC,UAAU,CAAC,GAAG,CAAC,SAAS,CAAC,CAAC;YACrC,KAAK,CAAC,IAAI,CAAC,GAAG,SAAS,CAAC,CAAC;QAC3B,CAAC;IACH,CAAC,CAAC,CAAC;IACH,OAAO;QACL,EAAE,EAAE,IAAI;QACR,OAAO,EAAE,GAAG,KAAK,CAAC,MAAM,SAAS,KAAK,CAAC,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,GAAG,QAAQ,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,EAAE,yCAAyC;QACpI,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC;KACvB,CAAC;AACJ,CAAC;AAED;;;;GAIG;AACH,SAAS,cAAc,CAAC,GAAW;IACjC,4EAA4E;IAC5E,sEAAsE;IACtE,yEAAyE;IACzE,gEAAgE;IAChE,OAAO,kBAAkB,CAAC,GAAG,EAAE,CAAC,MAAM,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,IAAI,EAAE,CAAC,CAAC;AACpE,CAAC"}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Extract a `.ods` (OpenDocument Spreadsheet) workbook: per `<table:table>`
|
|
4
|
+
* (sheet) a `[sheet: Name]` marker (the `table:name` attribute), then the rows
|
|
5
|
+
* as tab-separated cell text. A cell's text is its `<text:p>` content
|
|
6
|
+
* (multiple paragraphs join with a space — a newline would break the row
|
|
7
|
+
* line); covered cells (under a merge) render empty.
|
|
8
|
+
*/
|
|
9
|
+
export declare function extractOds(bytes: Buffer): ExtractResult;
|
|
10
|
+
//# sourceMappingURL=extract-ods.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-ods.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-ods.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAyB5D;;;;;;GAMG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CAqBvD"}
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
import { attrByLocalName, decodeXmlEntities, localElementBlocks, localName, walkLocalElementBlocks, } from './ooxml-text.js';
|
|
2
|
+
import { odfParagraphBlocks, odfParagraphText, readOdfContentXml, } from './odf-text.js';
|
|
3
|
+
/**
|
|
4
|
+
* Per-sheet extraction caps — the SAME bounds as the xlsx extractor. ODF is
|
|
5
|
+
* fond of `table:number-columns-repeated="16384"` (or a million empty trailing
|
|
6
|
+
* rows) to pad a sheet to the grid, so repeats are expanded BOUNDED and the
|
|
7
|
+
* extraction says when it truncated (a `[sheet truncated …]` line right under
|
|
8
|
+
* the sheet marker). Trailing EMPTY cells/rows are trimmed before their
|
|
9
|
+
* repeats are applied at all, so grid padding never counts as truncation.
|
|
10
|
+
*/
|
|
11
|
+
const MAX_ROWS_PER_SHEET = 10_000;
|
|
12
|
+
const MAX_COLS_PER_SHEET = 200;
|
|
13
|
+
/**
|
|
14
|
+
* Extract a `.ods` (OpenDocument Spreadsheet) workbook: per `<table:table>`
|
|
15
|
+
* (sheet) a `[sheet: Name]` marker (the `table:name` attribute), then the rows
|
|
16
|
+
* as tab-separated cell text. A cell's text is its `<text:p>` content
|
|
17
|
+
* (multiple paragraphs join with a space — a newline would break the row
|
|
18
|
+
* line); covered cells (under a merge) render empty.
|
|
19
|
+
*/
|
|
20
|
+
export function extractOds(bytes) {
|
|
21
|
+
const content = readOdfContentXml(bytes, '.ods');
|
|
22
|
+
if (!content.ok)
|
|
23
|
+
return content;
|
|
24
|
+
const tables = tableBlocks(content.xml);
|
|
25
|
+
if (tables.length === 0) {
|
|
26
|
+
return { ok: false, message: 'could not be parsed as a .ods (no table:table elements in content.xml)' };
|
|
27
|
+
}
|
|
28
|
+
const lines = [];
|
|
29
|
+
for (const table of tables) {
|
|
30
|
+
lines.push(`[sheet: ${table.name}]`);
|
|
31
|
+
const { rows, truncated } = expandRows(table.xml);
|
|
32
|
+
if (truncated.length > 0)
|
|
33
|
+
lines.push(`[sheet truncated to the ${truncated.join(' and ')}]`);
|
|
34
|
+
lines.push(...rows.map((cells) => cells.join('\t')));
|
|
35
|
+
}
|
|
36
|
+
return {
|
|
37
|
+
ok: true,
|
|
38
|
+
summary: `${tables.length} sheet${tables.length === 1 ? '' : 's'}, rows as tab-separated values; formulas, formatting and charts omitted`,
|
|
39
|
+
text: lines.join('\n'),
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* The `<table:table>` blocks with their decoded `table:name`, in document
|
|
44
|
+
* order. Non-greedy close — a nested table (legal in ODF text documents, not
|
|
45
|
+
* produced by spreadsheets) would end the outer block early, degrading
|
|
46
|
+
* grouping but never crashing.
|
|
47
|
+
*/
|
|
48
|
+
function tableBlocks(xml) {
|
|
49
|
+
// Read by the parser and matched on the LOCAL name: a comment or CDATA
|
|
50
|
+
// section holding a table-looking fragment used to answer as a real sheet,
|
|
51
|
+
// and a document binding the table namespace to another prefix had none.
|
|
52
|
+
const out = [];
|
|
53
|
+
for (const table of localElementBlocks(xml, ['table'])) {
|
|
54
|
+
const raw = attrByLocalName(table.attributes, 'name');
|
|
55
|
+
out.push({
|
|
56
|
+
// Control separators become spaces: a name holding an encoded newline
|
|
57
|
+
// or tab (` `) would corrupt the `[sheet: …]` marker's own line and
|
|
58
|
+
// the TSV structure under it.
|
|
59
|
+
name: raw ? decodeXmlEntities(raw).replace(/[\t\n\r]+/g, ' ') : `Sheet${out.length + 1}`,
|
|
60
|
+
xml: table.body ?? '',
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
return out;
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* Expand a sheet's rows with BOUNDED repeat handling, INCREMENTALLY — a row's
|
|
67
|
+
* expansion lands in the capped output as it parses, so the caps bound memory
|
|
68
|
+
* as well as output (materializing every row's cells before consulting the
|
|
69
|
+
* cap let an accepted ODS allocate its whole expansion first):
|
|
70
|
+
*
|
|
71
|
+
* - all-empty rows are buffered as a COUNT (with their
|
|
72
|
+
* `table:number-rows-repeated` applied) and flushed only when a non-empty
|
|
73
|
+
* row follows, so a million-row empty tail simply disappears,
|
|
74
|
+
* - once the row cap is hit, the remaining rows are never parsed at all.
|
|
75
|
+
*
|
|
76
|
+
* `truncated` lists what the caps cut (mirrors the xlsx extractor's note).
|
|
77
|
+
*/
|
|
78
|
+
function expandRows(tableXml) {
|
|
79
|
+
// The shared quote-aware scanner as a WALK, not an array: materializing
|
|
80
|
+
// every row block before consulting the cap let a sheet of >10k explicit
|
|
81
|
+
// rows allocate them all first. Each row lands here as it parses, and the
|
|
82
|
+
// visitor's `true` stops the scan at the cap — a self-closing row WITH
|
|
83
|
+
// attributes is still a row, a `/>` inside a quoted attribute value is not
|
|
84
|
+
// a delimiter, and an UNCLOSED row costs one scan of the sheet rather than
|
|
85
|
+
// one per opener (see `xmlElementBlocks`).
|
|
86
|
+
const rows = [];
|
|
87
|
+
let pendingEmpty = 0;
|
|
88
|
+
let rowsTruncated = false;
|
|
89
|
+
let colsTruncated = false;
|
|
90
|
+
walkLocalElementBlocks(tableXml, ['table-row'], (row) => {
|
|
91
|
+
const repeat = repeatCount(attrByLocalName(row.attributes, 'number-rows-repeated'));
|
|
92
|
+
const cells = expandCells(row.body ?? '');
|
|
93
|
+
if (cells.cells.length === 0) {
|
|
94
|
+
// Empty rows are interior padding until a non-empty row proves it —
|
|
95
|
+
// trailing ones are dropped with their repeats (grid padding, not data).
|
|
96
|
+
pendingEmpty += repeat;
|
|
97
|
+
return false;
|
|
98
|
+
}
|
|
99
|
+
if (cells.truncated)
|
|
100
|
+
colsTruncated = true;
|
|
101
|
+
for (; pendingEmpty > 0 && rows.length < MAX_ROWS_PER_SHEET; pendingEmpty--)
|
|
102
|
+
rows.push([]);
|
|
103
|
+
let i = 0;
|
|
104
|
+
for (; i < repeat && rows.length < MAX_ROWS_PER_SHEET; i++)
|
|
105
|
+
rows.push(cells.cells);
|
|
106
|
+
if (pendingEmpty > 0 || i < repeat) {
|
|
107
|
+
// The cap cut real content (a sheet that merely FILLS it is not truncated).
|
|
108
|
+
rowsTruncated = true;
|
|
109
|
+
return true;
|
|
110
|
+
}
|
|
111
|
+
return false;
|
|
112
|
+
});
|
|
113
|
+
const truncated = [];
|
|
114
|
+
if (rowsTruncated)
|
|
115
|
+
truncated.push(`first ${MAX_ROWS_PER_SHEET} rows`);
|
|
116
|
+
if (colsTruncated)
|
|
117
|
+
truncated.push(`first ${MAX_COLS_PER_SHEET} columns`);
|
|
118
|
+
return { rows, truncated };
|
|
119
|
+
}
|
|
120
|
+
/**
|
|
121
|
+
* One row's cell texts: `<table:table-cell>` / `<table:covered-table-cell>`
|
|
122
|
+
* in order, expanded INCREMENTALLY like the rows above — trailing EMPTY cells
|
|
123
|
+
* are buffered as a count (their `table:number-columns-repeated` never
|
|
124
|
+
* expands) and the parse stops at the column cap.
|
|
125
|
+
*/
|
|
126
|
+
function expandCells(rowXml) {
|
|
127
|
+
// Same walking scanner as the rows above — a row spelling a million
|
|
128
|
+
// explicit cells stops parsing at the column cap too.
|
|
129
|
+
const cells = [];
|
|
130
|
+
let pendingEmpty = 0;
|
|
131
|
+
let truncated = false;
|
|
132
|
+
walkLocalElementBlocks(rowXml, ['table-cell', 'covered-table-cell'], (cell) => {
|
|
133
|
+
const repeat = repeatCount(attrByLocalName(cell.attributes, 'number-columns-repeated'));
|
|
134
|
+
// Covered cells carry no own text anyway. Element-produced newlines/tabs
|
|
135
|
+
// INSIDE a cell (<text:line-break/>, <text:tab/>) become single spaces:
|
|
136
|
+
// the extraction's contract is one row per line with tab-separated cells,
|
|
137
|
+
// and a literal \n or \t inside a cell's text would silently break both.
|
|
138
|
+
// A COVERED cell is the hidden half of a merge: the visible cell carries
|
|
139
|
+
// the text. Such a cell may still hold stale content, and emitting it put
|
|
140
|
+
// a value in the grid where the sheet shows none.
|
|
141
|
+
const text = localName(cell.name) === 'covered-table-cell'
|
|
142
|
+
? ''
|
|
143
|
+
: odfParagraphBlocks(cell.body ?? '')
|
|
144
|
+
.map(odfParagraphText)
|
|
145
|
+
.join(' ')
|
|
146
|
+
.replace(/[\t\n\r]+/g, ' ');
|
|
147
|
+
if (text === '') {
|
|
148
|
+
pendingEmpty += repeat;
|
|
149
|
+
return false;
|
|
150
|
+
}
|
|
151
|
+
for (; pendingEmpty > 0 && cells.length < MAX_COLS_PER_SHEET; pendingEmpty--)
|
|
152
|
+
cells.push('');
|
|
153
|
+
let i = 0;
|
|
154
|
+
for (; i < repeat && cells.length < MAX_COLS_PER_SHEET; i++)
|
|
155
|
+
cells.push(text);
|
|
156
|
+
if (pendingEmpty > 0 || i < repeat) {
|
|
157
|
+
// The cap cut real content (a row that merely FILLS it is not truncated).
|
|
158
|
+
truncated = true;
|
|
159
|
+
return true;
|
|
160
|
+
}
|
|
161
|
+
return false;
|
|
162
|
+
});
|
|
163
|
+
return { cells, truncated };
|
|
164
|
+
}
|
|
165
|
+
/** A `…-repeated="N"` attribute value, clamped to a sane positive integer. */
|
|
166
|
+
function repeatCount(raw) {
|
|
167
|
+
// Decoded first: the block scanner hands attribute values RAW, and a repeat
|
|
168
|
+
// legally written with character references (`10`) must count as
|
|
169
|
+
// 10, not silently fall back to 1.
|
|
170
|
+
const n = raw !== undefined ? parseInt(decodeXmlEntities(raw), 10) : 1;
|
|
171
|
+
return Number.isFinite(n) && n >= 1 ? n : 1;
|
|
172
|
+
}
|
|
173
|
+
//# sourceMappingURL=extract-ods.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract-ods.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-ods.ts"],"names":[],"mappings":"AACA,OAAO,EACL,eAAe,EACf,iBAAiB,EACjB,kBAAkB,EAClB,SAAS,EACT,sBAAsB,GACvB,MAAM,iBAAiB,CAAC;AACzB,OAAO,EACL,kBAAkB,EAClB,gBAAgB,EAChB,iBAAiB,GAClB,MAAM,eAAe,CAAC;AAEvB;;;;;;;GAOG;AACH,MAAM,kBAAkB,GAAG,MAAM,CAAC;AAClC,MAAM,kBAAkB,GAAG,GAAG,CAAC;AAE/B;;;;;;GAMG;AACH,MAAM,UAAU,UAAU,CAAC,KAAa;IACtC,MAAM,OAAO,GAAG,iBAAiB,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC;IACjD,IAAI,CAAC,OAAO,CAAC,EAAE;QAAE,OAAO,OAAO,CAAC;IAEhC,MAAM,MAAM,GAAG,WAAW,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC;IACxC,IAAI,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QACxB,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,wEAAwE,EAAE,CAAC;IAC1G,CAAC;IAED,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,KAAK,MAAM,KAAK,IAAI,MAAM,EAAE,CAAC;QAC3B,KAAK,CAAC,IAAI,CAAC,WAAW,KAAK,CAAC,IAAI,GAAG,CAAC,CAAC;QACrC,MAAM,EAAE,IAAI,EAAE,SAAS,EAAE,GAAG,UAAU,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC;QAClD,IAAI,SAAS,CAAC,MAAM,GAAG,CAAC;YAAE,KAAK,CAAC,IAAI,CAAC,2BAA2B,SAAS,CAAC,IAAI,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC;QAC5F,KAAK,CAAC,IAAI,CAAC,GAAG,IAAI,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;IACvD,CAAC;IACD,OAAO;QACL,EAAE,EAAE,IAAI;QACR,OAAO,EAAE,GAAG,MAAM,CAAC,MAAM,SAAS,MAAM,CAAC,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,yEAAyE;QACzI,IAAI,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC;KACvB,CAAC;AACJ,CAAC;AAED;;;;;GAKG;AACH,SAAS,WAAW,CAAC,GAAW;IAC9B,uEAAuE;IACvE,2EAA2E;IAC3E,yEAAyE;IACzE,MAAM,GAAG,GAAyC,EAAE,CAAC;IACrD,KAAK,MAAM,KAAK,IAAI,kBAAkB,CAAC,GAAG,EAAE,CAAC,OAAO,CAAC,CAAC,EAAE,CAAC;QACvD,MAAM,GAAG,GAAG,eAAe,CAAC,KAAK,CAAC,UAAU,EAAE,MAAM,CAAC,CAAC;QACtD,GAAG,CAAC,IAAI,CAAC;YACP,sEAAsE;YACtE,wEAAwE;YACxE,8BAA8B;YAC9B,IAAI,EAAE,GAAG,CAAC,CAAC,CAAC,iBAAiB,CAAC,GAAG,CAAC,CAAC,OAAO,CAAC,YAAY,EAAE,GAAG,CAAC,CAAC,CAAC,CAAC,QAAQ,GAAG,CAAC,MAAM,GAAG,CAAC,EAAE;YACxF,GAAG,EAAE,KAAK,CAAC,IAAI,IAAI,EAAE;SACtB,CAAC,CAAC;IACL,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;;;;;;;;GAYG;AACH,SAAS,UAAU,CAAC,QAAgB;IAClC,wEAAwE;IACxE,yEAAyE;IACzE,0EAA0E;IAC1E,uEAAuE;IACvE,2EAA2E;IAC3E,2EAA2E;IAC3E,2CAA2C;IAC3C,MAAM,IAAI,GAAe,EAAE,CAAC;IAC5B,IAAI,YAAY,GAAG,CAAC,CAAC;IACrB,IAAI,aAAa,GAAG,KAAK,CAAC;IAC1B,IAAI,aAAa,GAAG,KAAK,CAAC;IAC1B,sBAAsB,CAAC,QAAQ,EAAE,CAAC,WAAW,CAAC,EAAE,CAAC,GAAG,EAAE,EAAE;QACtD,MAAM,MAAM,GAAG,WAAW,CAAC,eAAe,CAAC,GAAG,CAAC,UAAU,EAAE,sBAAsB,CAAC,CAAC,CAAC;QACpF,MAAM,KAAK,GAAG,WAAW,CAAC,GAAG,CAAC,IAAI,IAAI,EAAE,CAAC,CAAC;QAC1C,IAAI,KAAK,CAAC,KAAK,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;YAC7B,oEAAoE;YACpE,yEAAyE;YACzE,YAAY,IAAI,MAAM,CAAC;YACvB,OAAO,KAAK,CAAC;QACf,CAAC;QACD,IAAI,KAAK,CAAC,SAAS;YAAE,aAAa,GAAG,IAAI,CAAC;QAC1C,OAAO,YAAY,GAAG,CAAC,IAAI,IAAI,CAAC,MAAM,GAAG,kBAAkB,EAAE,YAAY,EAAE;YAAE,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;QAC3F,IAAI,CAAC,GAAG,CAAC,CAAC;QACV,OAAO,CAAC,GAAG,MAAM,IAAI,IAAI,CAAC,MAAM,GAAG,kBAAkB,EAAE,CAAC,EAAE;YAAE,IAAI,CAAC,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC;QACnF,IAAI,YAAY,GAAG,CAAC,IAAI,CAAC,GAAG,MAAM,EAAE,CAAC;YACnC,4EAA4E;YAC5E,aAAa,GAAG,IAAI,CAAC;YACrB,OAAO,IAAI,CAAC;QACd,CAAC;QACD,OAAO,KAAK,CAAC;IACf,CAAC,CAAC,CAAC;IACH,MAAM,SAAS,GAAa,EAAE,CAAC;IAC/B,IAAI,aAAa;QAAE,SAAS,CAAC,IAAI,CAAC,SAAS,kBAAkB,OAAO,CAAC,CAAC;IACtE,IAAI,aAAa;QAAE,SAAS,CAAC,IAAI,CAAC,SAAS,kBAAkB,UAAU,CAAC,CAAC;IACzE,OAAO,EAAE,IAAI,EAAE,SAAS,EAAE,CAAC;AAC7B,CAAC;AAED;;;;;GAKG;AACH,SAAS,WAAW,CAAC,MAAc;IACjC,oEAAoE;IACpE,sDAAsD;IACtD,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,YAAY,GAAG,CAAC,CAAC;IACrB,IAAI,SAAS,GAAG,KAAK,CAAC;IACtB,sBAAsB,CAAC,MAAM,EAAE,CAAC,YAAY,EAAE,oBAAoB,CAAC,EAAE,CAAC,IAAI,EAAE,EAAE;QAC5E,MAAM,MAAM,GAAG,WAAW,CAAC,eAAe,CAAC,IAAI,CAAC,UAAU,EAAE,yBAAyB,CAAC,CAAC,CAAC;QACxF,yEAAyE;QACzE,wEAAwE;QACxE,0EAA0E;QAC1E,yEAAyE;QACzE,yEAAyE;QACzE,0EAA0E;QAC1E,kDAAkD;QAClD,MAAM,IAAI,GAAG,SAAS,CAAC,IAAI,CAAC,IAAI,CAAC,KAAK,oBAAoB;YACxD,CAAC,CAAC,EAAE;YACJ,CAAC,CAAC,kBAAkB,CAAC,IAAI,CAAC,IAAI,IAAI,EAAE,CAAC;iBAChC,GAAG,CAAC,gBAAgB,CAAC;iBACrB,IAAI,CAAC,GAAG,CAAC;iBACT,OAAO,CAAC,YAAY,EAAE,GAAG,CAAC,CAAC;QAClC,IAAI,IAAI,KAAK,EAAE,EAAE,CAAC;YAChB,YAAY,IAAI,MAAM,CAAC;YACvB,OAAO,KAAK,CAAC;QACf,CAAC;QACD,OAAO,YAAY,GAAG,CAAC,IAAI,KAAK,CAAC,MAAM,GAAG,kBAAkB,EAAE,YAAY,EAAE;YAAE,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;QAC7F,IAAI,CAAC,GAAG,CAAC,CAAC;QACV,OAAO,CAAC,GAAG,MAAM,IAAI,KAAK,CAAC,MAAM,GAAG,kBAAkB,EAAE,CAAC,EAAE;YAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QAC9E,IAAI,YAAY,GAAG,CAAC,IAAI,CAAC,GAAG,MAAM,EAAE,CAAC;YACnC,0EAA0E;YAC1E,SAAS,GAAG,IAAI,CAAC;YACjB,OAAO,IAAI,CAAC;QACd,CAAC;QACD,OAAO,KAAK,CAAC;IACf,CAAC,CAAC,CAAC;IACH,OAAO,EAAE,KAAK,EAAE,SAAS,EAAE,CAAC;AAC9B,CAAC;AAED,8EAA8E;AAC9E,SAAS,WAAW,CAAC,GAAuB;IAC1C,4EAA4E;IAC5E,yEAAyE;IACzE,mCAAmC;IACnC,MAAM,CAAC,GAAG,GAAG,KAAK,SAAS,CAAC,CAAC,CAAC,QAAQ,CAAC,iBAAiB,CAAC,GAAG,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;IACvE,OAAO,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;AAC9C,CAAC"}
|