@bevel-software/platform-core-backend 0.11.2 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +1163 -425
- package/dist/core/create-core-server.js +1 -1
- package/dist/core/create-core-server.js.map +1 -1
- package/dist/core/create-core-services.d.ts +2 -0
- package/dist/core/create-core-services.d.ts.map +1 -1
- package/dist/core/create-core-services.js +5 -0
- package/dist/core/create-core-services.js.map +1 -1
- package/dist/core-config.d.ts +7 -0
- package/dist/core-config.d.ts.map +1 -1
- package/dist/core-config.js +9 -0
- package/dist/core-config.js.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.js +7 -1
- package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.js +59 -0
- package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.js +19 -0
- package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.js +151 -0
- package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
- package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
- package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
- package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
- package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
- package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
- package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.js +55 -0
- package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.js +108 -0
- package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.js +30 -0
- package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.js +116 -0
- package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.js +117 -0
- package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
- package/dist/modules/workspace/workspace.tools.d.ts +2 -1
- package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.tools.js +158 -15
- package/dist/modules/workspace/workspace.tools.js.map +1 -1
- package/package.json +9 -4
- package/src/core/create-core-server.ts +1 -1
- package/src/core/create-core-services.ts +6 -0
- package/src/core-config.ts +9 -0
- package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
- package/src/modules/code-mode/code-mode.tool.ts +7 -1
- package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
- package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
- package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
- package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
- package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
- package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
- package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
- package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
- package/src/modules/workspace/file-readers/document-reader.ts +64 -0
- package/src/modules/workspace/file-readers/email-reader.ts +21 -0
- package/src/modules/workspace/file-readers/email-text.ts +193 -0
- package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
- package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
- package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
- package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
- package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
- package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
- package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
- package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
- package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
- package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
- package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
- package/src/modules/workspace/file-readers/file-reader.ts +104 -0
- package/src/modules/workspace/file-readers/image-read.ts +122 -0
- package/src/modules/workspace/file-readers/image-reader.ts +39 -0
- package/src/modules/workspace/file-readers/odf-text.ts +123 -0
- package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
- package/src/modules/workspace/file-readers/text-reader.ts +131 -0
- package/src/modules/workspace/workspace.tools.ts +174 -12
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
export declare function odfParagraphText(paragraphXml: string): string;
|
|
2
|
+
/**
|
|
3
|
+
* The `<text:p>` / `<text:h>` paragraph bodies of an XML fragment, in DOCUMENT
|
|
4
|
+
* order (headings interleaved with paragraphs, as written). Self-closing
|
|
5
|
+
* elements (an empty paragraph) yield ''. `<text:page-number>` and friends do
|
|
6
|
+
* not count as paragraphs.
|
|
7
|
+
*/
|
|
8
|
+
export declare function odfParagraphBlocks(xml: string): string[];
|
|
9
|
+
/** Non-empty paragraph texts of an ODF fragment — what odp slides/notes render. */
|
|
10
|
+
export declare function odfParagraphLines(xml: string): string[];
|
|
11
|
+
/**
|
|
12
|
+
* `content.xml` of an ODF package or a typed could-not-parse failure message
|
|
13
|
+
* (`kind` names the extension for the message, e.g. '.odt').
|
|
14
|
+
*
|
|
15
|
+
* Bounded: the entry's DECLARED uncompressed size is checked against
|
|
16
|
+
* `MAX_DOC_PART_BYTES` (50 MB) before anything inflates, so a zip bomb is a
|
|
17
|
+
* typed refusal, never an allocation.
|
|
18
|
+
*/
|
|
19
|
+
export declare function readOdfContentXml(bytes: Buffer, kind: '.odt' | '.odp' | '.ods'): {
|
|
20
|
+
ok: true;
|
|
21
|
+
xml: string;
|
|
22
|
+
} | {
|
|
23
|
+
ok: false;
|
|
24
|
+
message: string;
|
|
25
|
+
};
|
|
26
|
+
//# sourceMappingURL=odf-text.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"odf-text.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/odf-text.ts"],"names":[],"mappings":"AAuCA,wBAAgB,gBAAgB,CAAC,YAAY,EAAE,MAAM,GAAG,MAAM,CAoC7D;AAED;;;;;GAKG;AACH,wBAAgB,kBAAkB,CAAC,GAAG,EAAE,MAAM,GAAG,MAAM,EAAE,CAExD;AAGD,mFAAmF;AACnF,wBAAgB,iBAAiB,CAAC,GAAG,EAAE,MAAM,GAAG,MAAM,EAAE,CAOvD;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAC/B,KAAK,EAAE,MAAM,EACb,IAAI,EAAE,MAAM,GAAG,MAAM,GAAG,MAAM,GAC7B;IAAE,EAAE,EAAE,IAAI,CAAC;IAAC,GAAG,EAAE,MAAM,CAAA;CAAE,GAAG;IAAE,EAAE,EAAE,KAAK,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAa5D"}
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
import AdmZip from 'adm-zip';
|
|
2
|
+
import { Parser } from 'htmlparser2';
|
|
3
|
+
import { attrByLocalName, localElementBlocks, localName, zipEntryOversize } from './ooxml-text.js';
|
|
4
|
+
/**
|
|
5
|
+
* ODF (OpenDocument) text helpers shared by the odt/odp/ods extractors.
|
|
6
|
+
*
|
|
7
|
+
* Elements are matched by LOCAL name — `p`, not `text:p`. A prefix is the
|
|
8
|
+
* document's own choice, and naming `text:` literally meant an ODF file that
|
|
9
|
+
* defaulted the namespace, or bound it to any other prefix, extracted as
|
|
10
|
+
* empty. This module used to answer that by REWRITING every non-conventional
|
|
11
|
+
* prefix across the whole of `content.xml` before scanning it, a pass that had
|
|
12
|
+
* to be bounded against crafted alias lists and could still corrupt ordinary
|
|
13
|
+
* paragraph text that happened to look like an alias. Matching local names
|
|
14
|
+
* needs none of it.
|
|
15
|
+
*/
|
|
16
|
+
/**
|
|
17
|
+
* The text of one ODF paragraph (`<text:p>` / `<text:h>` content). Character
|
|
18
|
+
* data is concatenated with NO separator — formatting runs (`<text:span>`)
|
|
19
|
+
* split words exactly like OOXML runs do, so ignoring the span tags joins them
|
|
20
|
+
* back. Three ODF whitespace elements are REAL characters and render as such:
|
|
21
|
+
*
|
|
22
|
+
* - `<text:tab/>` → a tab
|
|
23
|
+
* - `<text:line-break/>` → a newline
|
|
24
|
+
* - `<text:s text:c="N"/>` → N spaces (no `text:c` attribute = 1)
|
|
25
|
+
*
|
|
26
|
+
* Read through the parser, so entities decode, CDATA sections are the text
|
|
27
|
+
* they hold rather than markup, and a comment is not paragraph content.
|
|
28
|
+
*/
|
|
29
|
+
/**
|
|
30
|
+
* Cap on the characters the whitespace ELEMENTS may add to ONE paragraph.
|
|
31
|
+
* Each `<text:s text:c="N"/>` is clamped on its own below, but nothing else
|
|
32
|
+
* bounds how many such elements a paragraph may hold — a content.xml well
|
|
33
|
+
* under the 50 MB part limit could still expand to gigabytes of spaces. The
|
|
34
|
+
* budget caps the SUM per paragraph; real layout whitespace is nowhere near it.
|
|
35
|
+
*/
|
|
36
|
+
const MAX_PARAGRAPH_WHITESPACE_CHARS = 10_000;
|
|
37
|
+
export function odfParagraphText(paragraphXml) {
|
|
38
|
+
let out = '';
|
|
39
|
+
let whitespaceBudget = MAX_PARAGRAPH_WHITESPACE_CHARS;
|
|
40
|
+
const parser = new Parser({
|
|
41
|
+
onopentag(name, attributes) {
|
|
42
|
+
const local = localName(name);
|
|
43
|
+
if (local === 'tab') {
|
|
44
|
+
if (whitespaceBudget > 0) {
|
|
45
|
+
out += '\t';
|
|
46
|
+
whitespaceBudget--;
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
else if (local === 'line-break') {
|
|
50
|
+
if (whitespaceBudget > 0) {
|
|
51
|
+
out += '\n';
|
|
52
|
+
whitespaceBudget--;
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
else if (local === 's') {
|
|
56
|
+
const raw = attrByLocalName(attributes, 'c');
|
|
57
|
+
const count = raw !== undefined ? parseInt(raw, 10) : 1;
|
|
58
|
+
// Bounded defensively — a corrupt attribute must not balloon the
|
|
59
|
+
// extraction — and again by the per-paragraph budget above.
|
|
60
|
+
const spaces = Math.min(Number.isFinite(count) ? Math.min(Math.max(count, 0), 1000) : 1, whitespaceBudget);
|
|
61
|
+
out += ' '.repeat(spaces);
|
|
62
|
+
whitespaceBudget -= spaces;
|
|
63
|
+
}
|
|
64
|
+
},
|
|
65
|
+
ontext(text) {
|
|
66
|
+
out += text;
|
|
67
|
+
},
|
|
68
|
+
}, { xmlMode: true, decodeEntities: true });
|
|
69
|
+
parser.write(paragraphXml);
|
|
70
|
+
parser.end();
|
|
71
|
+
return out;
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* The `<text:p>` / `<text:h>` paragraph bodies of an XML fragment, in DOCUMENT
|
|
75
|
+
* order (headings interleaved with paragraphs, as written). Self-closing
|
|
76
|
+
* elements (an empty paragraph) yield ''. `<text:page-number>` and friends do
|
|
77
|
+
* not count as paragraphs.
|
|
78
|
+
*/
|
|
79
|
+
export function odfParagraphBlocks(xml) {
|
|
80
|
+
return localElementBlocks(xml, ['p', 'h']).map((e) => e.body ?? '');
|
|
81
|
+
}
|
|
82
|
+
/** Non-empty paragraph texts of an ODF fragment — what odp slides/notes render. */
|
|
83
|
+
export function odfParagraphLines(xml) {
|
|
84
|
+
const out = [];
|
|
85
|
+
for (const p of odfParagraphBlocks(xml)) {
|
|
86
|
+
const text = odfParagraphText(p);
|
|
87
|
+
if (text.trim() !== '')
|
|
88
|
+
out.push(text);
|
|
89
|
+
}
|
|
90
|
+
return out;
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* `content.xml` of an ODF package or a typed could-not-parse failure message
|
|
94
|
+
* (`kind` names the extension for the message, e.g. '.odt').
|
|
95
|
+
*
|
|
96
|
+
* Bounded: the entry's DECLARED uncompressed size is checked against
|
|
97
|
+
* `MAX_DOC_PART_BYTES` (50 MB) before anything inflates, so a zip bomb is a
|
|
98
|
+
* typed refusal, never an allocation.
|
|
99
|
+
*/
|
|
100
|
+
export function readOdfContentXml(bytes, kind) {
|
|
101
|
+
try {
|
|
102
|
+
const zip = new AdmZip(bytes);
|
|
103
|
+
const entry = zip.getEntry('content.xml');
|
|
104
|
+
if (!entry) {
|
|
105
|
+
return { ok: false, message: `could not be parsed as a ${kind} (no content.xml inside the archive)` };
|
|
106
|
+
}
|
|
107
|
+
const oversize = zipEntryOversize(entry);
|
|
108
|
+
if (oversize)
|
|
109
|
+
return { ok: false, message: `could not be extracted as a ${kind} (${oversize})` };
|
|
110
|
+
return { ok: true, xml: entry.getData().toString('utf8') };
|
|
111
|
+
}
|
|
112
|
+
catch (err) {
|
|
113
|
+
return { ok: false, message: `could not be parsed as a ${kind} (${err.message})` };
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
//# sourceMappingURL=odf-text.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"odf-text.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/odf-text.ts"],"names":[],"mappings":"AAAA,OAAO,MAAM,MAAM,SAAS,CAAC;AAC7B,OAAO,EAAE,MAAM,EAAE,MAAM,aAAa,CAAC;AACrC,OAAO,EAAE,eAAe,EAAE,kBAAkB,EAAE,SAAS,EAAE,gBAAgB,EAAE,MAAM,iBAAiB,CAAC;AAEnG;;;;;;;;;;;GAWG;AAEH;;;;;;;;;;;;GAYG;AACH;;;;;;GAMG;AACH,MAAM,8BAA8B,GAAG,MAAM,CAAC;AAE9C,MAAM,UAAU,gBAAgB,CAAC,YAAoB;IACnD,IAAI,GAAG,GAAG,EAAE,CAAC;IACb,IAAI,gBAAgB,GAAG,8BAA8B,CAAC;IACtD,MAAM,MAAM,GAAG,IAAI,MAAM,CACvB;QACE,SAAS,CAAC,IAAI,EAAE,UAAU;YACxB,MAAM,KAAK,GAAG,SAAS,CAAC,IAAI,CAAC,CAAC;YAC9B,IAAI,KAAK,KAAK,KAAK,EAAE,CAAC;gBACpB,IAAI,gBAAgB,GAAG,CAAC,EAAE,CAAC;oBACzB,GAAG,IAAI,IAAI,CAAC;oBACZ,gBAAgB,EAAE,CAAC;gBACrB,CAAC;YACH,CAAC;iBAAM,IAAI,KAAK,KAAK,YAAY,EAAE,CAAC;gBAClC,IAAI,gBAAgB,GAAG,CAAC,EAAE,CAAC;oBACzB,GAAG,IAAI,IAAI,CAAC;oBACZ,gBAAgB,EAAE,CAAC;gBACrB,CAAC;YACH,CAAC;iBAAM,IAAI,KAAK,KAAK,GAAG,EAAE,CAAC;gBACzB,MAAM,GAAG,GAAG,eAAe,CAAC,UAAU,EAAE,GAAG,CAAC,CAAC;gBAC7C,MAAM,KAAK,GAAG,GAAG,KAAK,SAAS,CAAC,CAAC,CAAC,QAAQ,CAAC,GAAG,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC;gBACxD,iEAAiE;gBACjE,4DAA4D;gBAC5D,MAAM,MAAM,GAAG,IAAI,CAAC,GAAG,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,GAAG,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,EAAE,CAAC,CAAC,EAAE,IAAI,CAAC,CAAC,CAAC,CAAC,CAAC,EAAE,gBAAgB,CAAC,CAAC;gBAC3G,GAAG,IAAI,GAAG,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC;gBAC1B,gBAAgB,IAAI,MAAM,CAAC;YAC7B,CAAC;QACH,CAAC;QACD,MAAM,CAAC,IAAI;YACT,GAAG,IAAI,IAAI,CAAC;QACd,CAAC;KACF,EACD,EAAE,OAAO,EAAE,IAAI,EAAE,cAAc,EAAE,IAAI,EAAE,CACxC,CAAC;IACF,MAAM,CAAC,KAAK,CAAC,YAAY,CAAC,CAAC;IAC3B,MAAM,CAAC,GAAG,EAAE,CAAC;IACb,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,kBAAkB,CAAC,GAAW;IAC5C,OAAO,kBAAkB,CAAC,GAAG,EAAE,CAAC,GAAG,EAAE,GAAG,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,IAAI,IAAI,EAAE,CAAC,CAAC;AACtE,CAAC;AAGD,mFAAmF;AACnF,MAAM,UAAU,iBAAiB,CAAC,GAAW;IAC3C,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,MAAM,CAAC,IAAI,kBAAkB,CAAC,GAAG,CAAC,EAAE,CAAC;QACxC,MAAM,IAAI,GAAG,gBAAgB,CAAC,CAAC,CAAC,CAAC;QACjC,IAAI,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE;YAAE,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IACzC,CAAC;IACD,OAAO,GAAG,CAAC;AACb,CAAC;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,iBAAiB,CAC/B,KAAa,EACb,IAA8B;IAE9B,IAAI,CAAC;QACH,MAAM,GAAG,GAAG,IAAI,MAAM,CAAC,KAAK,CAAC,CAAC;QAC9B,MAAM,KAAK,GAAG,GAAG,CAAC,QAAQ,CAAC,aAAa,CAAC,CAAC;QAC1C,IAAI,CAAC,KAAK,EAAE,CAAC;YACX,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,4BAA4B,IAAI,sCAAsC,EAAE,CAAC;QACxG,CAAC;QACD,MAAM,QAAQ,GAAG,gBAAgB,CAAC,KAAK,CAAC,CAAC;QACzC,IAAI,QAAQ;YAAE,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,+BAA+B,IAAI,KAAK,QAAQ,GAAG,EAAE,CAAC;QACjG,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,GAAG,EAAE,KAAK,CAAC,OAAO,EAAE,CAAC,QAAQ,CAAC,MAAM,CAAC,EAAE,CAAC;IAC7D,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,OAAO,EAAE,4BAA4B,IAAI,KAAM,GAAa,CAAC,OAAO,GAAG,EAAE,CAAC;IAChG,CAAC;AACH,CAAC"}
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
import type AdmZip from 'adm-zip';
|
|
2
|
+
/**
|
|
3
|
+
* Decompression bounds for the document extractors (OOXML, ODF and — as a
|
|
4
|
+
* plain byte cap — PDF). A zip's central directory declares each entry's
|
|
5
|
+
* UNCOMPRESSED size, so a zip bomb (a few KB that inflate to gigabytes) is
|
|
6
|
+
* detectable BEFORE any inflation happens; 50 MB of XML is far beyond any
|
|
7
|
+
* real office document part (a huge deck's slide parts run to single-digit
|
|
8
|
+
* MB) while staying well inside what a server can afford to decode.
|
|
9
|
+
* `MAX_DOC_TOTAL_BYTES` additionally bounds the SUM of the parts a multi-part
|
|
10
|
+
* extraction reads (pptx slides/notes, xlsx sheet parts) at 200 MB.
|
|
11
|
+
*/
|
|
12
|
+
export declare const MAX_DOC_PART_BYTES: number;
|
|
13
|
+
export declare const MAX_DOC_TOTAL_BYTES: number;
|
|
14
|
+
/**
|
|
15
|
+
* The typed-failure fragment for a zip entry whose DECLARED uncompressed size
|
|
16
|
+
* exceeds {@link MAX_DOC_PART_BYTES}, or null when the entry is within bounds.
|
|
17
|
+
* Checked against the central-directory header BEFORE `getData()` inflates
|
|
18
|
+
* anything, so an oversized (or bomb) entry costs nothing.
|
|
19
|
+
*/
|
|
20
|
+
export declare function zipEntryOversize(entry: AdmZip.IZipEntry): string | null;
|
|
21
|
+
export declare const XML_NCNAME: string;
|
|
22
|
+
/**
|
|
23
|
+
* One tag's attributes as `name → raw value` tokens, in document order. A
|
|
24
|
+
* real left-to-right tokenizer, not a regex probe: quoted values (either
|
|
25
|
+
* quote style, whitespace around `=` tolerated) are skipped over WHOLE, so a
|
|
26
|
+
* `target='…'`-looking sequence INSIDE another attribute's value can never
|
|
27
|
+
* be mistaken for an attribute of its own. Values are RAW (entities not
|
|
28
|
+
* decoded); a malformed tail (unterminated quote) simply ends the scan.
|
|
29
|
+
*/
|
|
30
|
+
export declare function xmlAttrTokens(tagXml: string): Array<{
|
|
31
|
+
name: string;
|
|
32
|
+
value: string;
|
|
33
|
+
}>;
|
|
34
|
+
/**
|
|
35
|
+
* The value of attribute `name` inside one tag's text, or undefined. Exact
|
|
36
|
+
* (prefix-included) name match over the {@link xmlAttrTokens} scan — see there
|
|
37
|
+
* for the quoting guarantees. The value is returned RAW (entities not
|
|
38
|
+
* decoded); callers decode where display matters.
|
|
39
|
+
*/
|
|
40
|
+
export declare function xmlAttrValue(tagXml: string, name: string): string | undefined;
|
|
41
|
+
/**
|
|
42
|
+
* Like {@link xmlAttrValue}, but matching the attribute's LOCAL name — the
|
|
43
|
+
* part after any namespace prefix. For parsers that scan by local element
|
|
44
|
+
* name (OPC `.rels` parts, whose producer is free to prefix the relationship
|
|
45
|
+
* namespace) and must accept `r:Target` wherever `Target` is meant.
|
|
46
|
+
*/
|
|
47
|
+
export declare function xmlAttrValueByLocalName(tagXml: string, localName: string): string | undefined;
|
|
48
|
+
/**
|
|
49
|
+
* Decode the five XML named entities plus numeric (`A` / `A`)
|
|
50
|
+
* references. Decimal references admit ONLY decimal digits and hex digits only
|
|
51
|
+
* after `#x` — a malformed `A;` must stay literal text, not be consumed
|
|
52
|
+
* with `parseInt` silently stopping at the `A` and emitting U+000C.
|
|
53
|
+
*/
|
|
54
|
+
export declare function decodeXmlEntities(s: string): string;
|
|
55
|
+
/** One element found by {@link xmlElementBlocks}. */
|
|
56
|
+
/** One element found by {@link xmlElementBlocks}. */
|
|
57
|
+
export interface XmlElementBlock {
|
|
58
|
+
/** The qualified name as written, e.g. `w:p` — which of `names` matched. */
|
|
59
|
+
name: string;
|
|
60
|
+
/** The element's attributes. Values are RAW (entities not decoded). */
|
|
61
|
+
attributes: Record<string, string>;
|
|
62
|
+
/** The body between `>` and the matching close tag; undefined when self-closing. */
|
|
63
|
+
body: string | undefined;
|
|
64
|
+
/** Index of the element's opening `<` in the scanned string. */
|
|
65
|
+
start: number;
|
|
66
|
+
/** One past the element's final `>` (as far as the parse got, for a block cut short by the depth cap). */
|
|
67
|
+
end: number;
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* How deep the element stack may go before a part is given up on.
|
|
71
|
+
*
|
|
72
|
+
* Real office XML nests a few dozen levels. A crafted part can nest as deep as
|
|
73
|
+
* it has bytes, and the parser's own cost climbs faster than linearly once the
|
|
74
|
+
* stack is enormous: measured on unclosed `<a:p>` openers, 40 k deep took
|
|
75
|
+
* 136 ms and 80 k took 2.7 s. A 50 MB part could spell millions. So the depth
|
|
76
|
+
* is bounded far above any real document and far below where that curve bites.
|
|
77
|
+
*
|
|
78
|
+
* On reaching it the parse stops and what was found is returned — including
|
|
79
|
+
* the element still open, whose body is taken as far as the parse got, so a
|
|
80
|
+
* paragraph holding real text before the crafted tail still yields that text.
|
|
81
|
+
*/
|
|
82
|
+
export declare const MAX_ELEMENT_DEPTH = 1000;
|
|
83
|
+
/**
|
|
84
|
+
* Thrown to stop the parse at {@link MAX_ELEMENT_DEPTH}; never escapes an
|
|
85
|
+
* extractor (also shared by `email-text.ts`'s HTML strip).
|
|
86
|
+
*/
|
|
87
|
+
export declare const TOO_DEEP: unique symbol;
|
|
88
|
+
/**
|
|
89
|
+
* The `<name …>…</name>` and self-closing `<name …/>` elements named by
|
|
90
|
+
* `names`, in document order, at ANY depth — but never descending into a
|
|
91
|
+
* match, since a match nested inside another is part of that one's body
|
|
92
|
+
* rather than a block of its own.
|
|
93
|
+
*
|
|
94
|
+
* Parsing is `htmlparser2` in XML mode, which is the point of this function.
|
|
95
|
+
* What stood here before was a hand-rolled scanner, and the lexical rules it
|
|
96
|
+
* had to know kept turning out to be one rule short: a `>` inside a quoted
|
|
97
|
+
* attribute value, a `/` that only ends a name as part of `/>`, a `</w:p>`
|
|
98
|
+
* written inside a comment or a CDATA section, a namespace prefix outside
|
|
99
|
+
* ASCII. Each gap was a real defect, several were reachable from an uploaded
|
|
100
|
+
* file, and the fixes needed their own bookkeeping — a tag-end memo, a section
|
|
101
|
+
* index, caps on both — which then had defects of their own. All of that is
|
|
102
|
+
* the parser's job here, and it is code with far more mileage than ours.
|
|
103
|
+
*
|
|
104
|
+
* Bodies are RAW slices of `xml` (entities not decoded), because the callers
|
|
105
|
+
* re-scan them for nested elements and decode only the text they keep.
|
|
106
|
+
*/
|
|
107
|
+
/** The part of a qualified XML name after its namespace prefix. */
|
|
108
|
+
export declare function localName(qualified: string): string;
|
|
109
|
+
/**
|
|
110
|
+
* The value of the attribute whose LOCAL name is `want`, or undefined.
|
|
111
|
+
* Namespace DECLARATIONS are not attributes and never answer for one.
|
|
112
|
+
*/
|
|
113
|
+
export declare function attrByLocalName(attributes: Record<string, string>, want: string): string | undefined;
|
|
114
|
+
/**
|
|
115
|
+
* {@link xmlElementBlocks}, matching each element's LOCAL name instead of the
|
|
116
|
+
* qualified one — `p` finds `<w:p>`, `<a:p>` and an unprefixed `<p>` alike.
|
|
117
|
+
*
|
|
118
|
+
* Prefixes are a document's own choice: XML binds them to namespace URIs, and
|
|
119
|
+
* a producer may bind any prefix it likes or default the namespace and use
|
|
120
|
+
* none. Naming `a:p` or `text:p` literally therefore read only the documents
|
|
121
|
+
* whose authors happened to pick the usual prefix — a valid deck using `d:p`
|
|
122
|
+
* for DrawingML extracted as EMPTY, and an ODT that defaulted the text
|
|
123
|
+
* namespace found no paragraphs at all. Matching the local name reads both,
|
|
124
|
+
* and replaces the prefix-rewriting pass the ODF readers used to run over
|
|
125
|
+
* every document to paper over the same problem.
|
|
126
|
+
*/
|
|
127
|
+
export declare function localElementBlocks(xml: string, localNames: readonly string[]): XmlElementBlock[];
|
|
128
|
+
/**
|
|
129
|
+
* {@link localElementBlocks} as a WALK: `visit` receives each element as its
|
|
130
|
+
* close tag is reached, and returning true STOPS the scan — the input past
|
|
131
|
+
* that element is never parsed and no block is materialized beyond it. For
|
|
132
|
+
* callers with a cap (the ods row walk): collecting every block into an array
|
|
133
|
+
* before consulting the cap let an accepted document allocate its whole
|
|
134
|
+
* expansion first.
|
|
135
|
+
*/
|
|
136
|
+
export declare function walkLocalElementBlocks(xml: string, localNames: readonly string[], visit: (block: XmlElementBlock) => boolean | void): void;
|
|
137
|
+
/**
|
|
138
|
+
* `xml` with every element named by `localNames` (matched on its LOCAL name)
|
|
139
|
+
* removed WHOLE — open tag through matching close tag — by the parsed block
|
|
140
|
+
* boundaries. The structural counterpart to string replacement, which deleted
|
|
141
|
+
* every occurrence of a block's serialized BODY: a slide whose visible text
|
|
142
|
+
* happened to serialize identically to its notes lost that text too.
|
|
143
|
+
*/
|
|
144
|
+
export declare function removeLocalElements(xml: string, localNames: readonly string[]): string;
|
|
145
|
+
export declare function xmlElementBlocks(xml: string, names: Iterable<string>,
|
|
146
|
+
/** Overrides name matching — {@link localElementBlocks} matches local names with it. */
|
|
147
|
+
matches?: (name: string) => boolean,
|
|
148
|
+
/**
|
|
149
|
+
* Streaming hook — see {@link walkLocalElementBlocks}. When given, each
|
|
150
|
+
* block is handed to it INSTEAD of being accumulated (the return value is
|
|
151
|
+
* then an empty array), and returning true stops the scan.
|
|
152
|
+
*/
|
|
153
|
+
visit?: (block: XmlElementBlock) => boolean | void): XmlElementBlock[];
|
|
154
|
+
/**
|
|
155
|
+
* The text of one OOXML paragraph: every `<w:t>`/`<a:t>` run's character
|
|
156
|
+
* content, concatenated with NO separator — Word/PowerPoint split runs
|
|
157
|
+
* mid-word on formatting boundaries, so any separator would break words apart.
|
|
158
|
+
* `tag` is the run tag ('w:t' for docx, 'a:t' for pptx). A self-closing
|
|
159
|
+
* `<w:t/>` is an empty run and contributes nothing, exactly as it did when
|
|
160
|
+
* the pattern simply failed to match it.
|
|
161
|
+
*/
|
|
162
|
+
export declare function paragraphRunText(paragraphXml: string, localTag: string): string;
|
|
163
|
+
/**
|
|
164
|
+
* Split an XML fragment into its `<{tag}>…</{tag}>` blocks, in document order.
|
|
165
|
+
* A self-closing `<{tag}/>` yields '' — an EMPTY block, which is what an empty
|
|
166
|
+
* `<w:p/>` paragraph or `<w:tc/>` cell means. See {@link xmlElementBlocks} for
|
|
167
|
+
* the non-nesting and quoting guarantees.
|
|
168
|
+
*/
|
|
169
|
+
/** {@link xmlBlocks} by LOCAL name — see {@link localElementBlocks} for why. */
|
|
170
|
+
export declare function localBlocks(xml: string, local: string): string[];
|
|
171
|
+
export declare function xmlBlocks(xml: string, tag: string): string[];
|
|
172
|
+
//# sourceMappingURL=ooxml-text.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"ooxml-text.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/ooxml-text.ts"],"names":[],"mappings":"AAgBA,OAAO,KAAK,MAAM,MAAM,SAAS,CAAC;AAElC;;;;;;;;;GASG;AACH,eAAO,MAAM,kBAAkB,QAAmB,CAAC;AACnD,eAAO,MAAM,mBAAmB,QAAoB,CAAC;AAErD;;;;;GAKG;AACH,wBAAgB,gBAAgB,CAAC,KAAK,EAAE,MAAM,CAAC,SAAS,GAAG,MAAM,GAAG,IAAI,CAKvE;AAsBD,eAAO,MAAM,UAAU,QAEyC,CAAC;AAEjE;;;;;;;GAOG;AACH,wBAAgB,aAAa,CAAC,MAAM,EAAE,MAAM,GAAG,KAAK,CAAC;IAAE,IAAI,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,MAAM,CAAA;CAAE,CAAC,CA4BpF;AAED;;;;;GAKG;AACH,wBAAgB,YAAY,CAAC,MAAM,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAK7E;AAED;;;;;GAKG;AACH,wBAAgB,uBAAuB,CAAC,MAAM,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAS7F;AAuBD;;;;;GAKG;AACH,wBAAgB,iBAAiB,CAAC,CAAC,EAAE,MAAM,GAAG,MAAM,CAmBnD;AAED,qDAAqD;AACrD,qDAAqD;AACrD,MAAM,WAAW,eAAe;IAC9B,4EAA4E;IAC5E,IAAI,EAAE,MAAM,CAAC;IACb,uEAAuE;IACvE,UAAU,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACnC,oFAAoF;IACpF,IAAI,EAAE,MAAM,GAAG,SAAS,CAAC;IACzB,gEAAgE;IAChE,KAAK,EAAE,MAAM,CAAC;IACd,0GAA0G;IAC1G,GAAG,EAAE,MAAM,CAAC;CACb;AAED;;;;;;;;;;;;GAYG;AACH,eAAO,MAAM,iBAAiB,OAAQ,CAAC;AAEvC;;;GAGG;AACH,eAAO,MAAM,QAAQ,eAAqB,CAAC;AAgB3C;;;;;;;;;;;;;;;;;;GAkBG;AACH,mEAAmE;AACnE,wBAAgB,SAAS,CAAC,SAAS,EAAE,MAAM,GAAG,MAAM,CAEnD;AAED;;;GAGG;AACH,wBAAgB,eAAe,CAC7B,UAAU,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,EAClC,IAAI,EAAE,MAAM,GACX,MAAM,GAAG,SAAS,CAMpB;AAED;;;;;;;;;;;;GAYG;AACH,wBAAgB,kBAAkB,CAAC,GAAG,EAAE,MAAM,EAAE,UAAU,EAAE,SAAS,MAAM,EAAE,GAAG,eAAe,EAAE,CAGhG;AAED;;;;;;;GAOG;AACH,wBAAgB,sBAAsB,CACpC,GAAG,EAAE,MAAM,EACX,UAAU,EAAE,SAAS,MAAM,EAAE,EAC7B,KAAK,EAAE,CAAC,KAAK,EAAE,eAAe,KAAK,OAAO,GAAG,IAAI,GAChD,IAAI,CAGN;AAED;;;;;;GAMG;AACH,wBAAgB,mBAAmB,CAAC,GAAG,EAAE,MAAM,EAAE,UAAU,EAAE,SAAS,MAAM,EAAE,GAAG,MAAM,CAYtF;AAED,wBAAgB,gBAAgB,CAC9B,GAAG,EAAE,MAAM,EACX,KAAK,EAAE,QAAQ,CAAC,MAAM,CAAC;AACvB,wFAAwF;AACxF,OAAO,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO;AACnC;;;;GAIG;AACH,KAAK,CAAC,EAAE,CAAC,KAAK,EAAE,eAAe,KAAK,OAAO,GAAG,IAAI,GACjD,eAAe,EAAE,CAkEnB;AAED;;;;;;;GAOG;AACH,wBAAgB,gBAAgB,CAAC,YAAY,EAAE,MAAM,EAAE,QAAQ,EAAE,MAAM,GAAG,MAAM,CA6C/E;AAED;;;;;GAKG;AACH,gFAAgF;AAChF,wBAAgB,WAAW,CAAC,GAAG,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,GAAG,MAAM,EAAE,CAEhE;AAED,wBAAgB,SAAS,CAAC,GAAG,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,MAAM,EAAE,CAE5D"}
|