@bevel-software/platform-core-backend 0.11.2 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +1165 -427
- package/dist/core/create-core-server.js +1 -1
- package/dist/core/create-core-server.js.map +1 -1
- package/dist/core/create-core-services.d.ts +2 -0
- package/dist/core/create-core-services.d.ts.map +1 -1
- package/dist/core/create-core-services.js +5 -0
- package/dist/core/create-core-services.js.map +1 -1
- package/dist/core-config.d.ts +7 -0
- package/dist/core-config.d.ts.map +1 -1
- package/dist/core-config.js +9 -0
- package/dist/core-config.js.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.js +7 -1
- package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
- package/dist/modules/kb-fs/clone-config.d.ts +40 -2
- package/dist/modules/kb-fs/clone-config.d.ts.map +1 -1
- package/dist/modules/kb-fs/clone-config.js +94 -2
- package/dist/modules/kb-fs/clone-config.js.map +1 -1
- package/dist/modules/workflow/workflow.service.d.ts +38 -0
- package/dist/modules/workflow/workflow.service.d.ts.map +1 -1
- package/dist/modules/workflow/workflow.service.js +112 -6
- package/dist/modules/workflow/workflow.service.js.map +1 -1
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.js +59 -0
- package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.js +19 -0
- package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.js +151 -0
- package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
- package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
- package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
- package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
- package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
- package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
- package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.js +55 -0
- package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.js +108 -0
- package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.js +30 -0
- package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.js +116 -0
- package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.js +117 -0
- package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
- package/dist/modules/workspace/startup/kb-git.d.ts.map +1 -1
- package/dist/modules/workspace/startup/kb-git.js +21 -4
- package/dist/modules/workspace/startup/kb-git.js.map +1 -1
- package/dist/modules/workspace/workspace.service.d.ts +52 -8
- package/dist/modules/workspace/workspace.service.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.service.js +121 -23
- package/dist/modules/workspace/workspace.service.js.map +1 -1
- package/dist/modules/workspace/workspace.tools.d.ts +2 -1
- package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.tools.js +158 -15
- package/dist/modules/workspace/workspace.tools.js.map +1 -1
- package/package.json +11 -6
- package/src/core/create-core-server.ts +1 -1
- package/src/core/create-core-services.ts +6 -0
- package/src/core-config.ts +9 -0
- package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
- package/src/modules/code-mode/code-mode.tool.ts +7 -1
- package/src/modules/kb-fs/__tests__/clone-config.test.ts +63 -2
- package/src/modules/kb-fs/clone-config.ts +97 -2
- package/src/modules/secrets-vault/secrets-vault.routes.ts +582 -582
- package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
- package/src/modules/workflow/__tests__/workflow.service.commitFileWhileLocked.test.ts +11 -5
- package/src/modules/workflow/__tests__/workflow.service.releaseLock.test.ts +172 -7
- package/src/modules/workflow/workflow.service.ts +118 -6
- package/src/modules/workspace/__tests__/workspace.service.test.ts +1 -1
- package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
- package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
- package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
- package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
- package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
- package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
- package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
- package/src/modules/workspace/file-readers/document-reader.ts +64 -0
- package/src/modules/workspace/file-readers/email-reader.ts +21 -0
- package/src/modules/workspace/file-readers/email-text.ts +193 -0
- package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
- package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
- package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
- package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
- package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
- package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
- package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
- package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
- package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
- package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
- package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
- package/src/modules/workspace/file-readers/file-reader.ts +104 -0
- package/src/modules/workspace/file-readers/image-read.ts +122 -0
- package/src/modules/workspace/file-readers/image-reader.ts +39 -0
- package/src/modules/workspace/file-readers/odf-text.ts +123 -0
- package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
- package/src/modules/workspace/file-readers/text-reader.ts +131 -0
- package/src/modules/workspace/startup/__tests__/kb-startup-runner.test.ts +141 -0
- package/src/modules/workspace/startup/kb-git.ts +20 -7
- package/src/modules/workspace/workspace.service.ts +132 -25
- package/src/modules/workspace/workspace.tools.ts +174 -12
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
import MsgReaderImport from '@kenjiuno/msgreader';
|
|
2
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
3
|
+
import { emailExtraction, htmlToEmailText, type EmailAttachment, type EmailModel } from './email-text.js';
|
|
4
|
+
import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Extract a `.msg` (Outlook item, CFB container) email into the shared email
|
|
8
|
+
* text shape (see `email-text.ts`).
|
|
9
|
+
*
|
|
10
|
+
* Parsing is `@kenjiuno/msgreader` (HiraokaHyperTools, Apache-2.0) — the
|
|
11
|
+
* maintained MAPI/CFB reader. It never throws for bad content of its own
|
|
12
|
+
* accord: unparseable bytes come back as `{ dataType: null, error }`, which
|
|
13
|
+
* maps onto the typed could-not-be-parsed failure here.
|
|
14
|
+
*
|
|
15
|
+
* Body preference mirrors `.eml`: the plain-text `PidTagBody` first, an HTML
|
|
16
|
+
* body stripped to text second. An Outlook item whose body exists ONLY as
|
|
17
|
+
* compressed RTF is degraded honestly — the extraction says
|
|
18
|
+
* "[body is RTF; no plain-text part]" instead of pretending to decode RTF.
|
|
19
|
+
*/
|
|
20
|
+
export function extractMsg(bytes: Buffer): ExtractResult {
|
|
21
|
+
if (bytes.length > MAX_DOC_PART_BYTES) {
|
|
22
|
+
return {
|
|
23
|
+
ok: false,
|
|
24
|
+
message: `could not be extracted as a .msg (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
|
|
25
|
+
};
|
|
26
|
+
}
|
|
27
|
+
let fields: FieldsData;
|
|
28
|
+
try {
|
|
29
|
+
// DataView over the Buffer's exact region — no copy, and msgreader never
|
|
30
|
+
// sees bytes outside the file.
|
|
31
|
+
fields = new MsgReader(new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength)).getFileData();
|
|
32
|
+
} catch (err) {
|
|
33
|
+
return { ok: false, message: `could not be parsed as a .msg (${(err as Error).message})` };
|
|
34
|
+
}
|
|
35
|
+
if (fields.dataType !== 'msg') {
|
|
36
|
+
return {
|
|
37
|
+
ok: false,
|
|
38
|
+
message: `could not be parsed as a .msg (${fields.error ?? 'not an Outlook message file'})`,
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
return { ok: true, ...emailExtraction(msgModel(fields)) };
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** msgreader's field data, shaped into the format-independent email model. */
|
|
45
|
+
function msgModel(fields: FieldsData): EmailModel {
|
|
46
|
+
const recipients = fields.recipients ?? [];
|
|
47
|
+
const text = fields.body !== undefined && fields.body.trim() !== '' ? fields.body : undefined;
|
|
48
|
+
const html = htmlBody(fields);
|
|
49
|
+
const body = text ?? (html !== undefined ? htmlToEmailText(html) : '');
|
|
50
|
+
const bodySource: EmailModel['bodySource'] =
|
|
51
|
+
text !== undefined ? 'text' : html !== undefined ? 'html' : fields.compressedRtf !== undefined ? 'rtf-only' : 'none';
|
|
52
|
+
const date = fields.clientSubmitTime ?? fields.messageDeliveryTime;
|
|
53
|
+
return {
|
|
54
|
+
from: mailboxText(fields.senderName, fields.senderSmtpAddress ?? fields.senderEmail),
|
|
55
|
+
to: recipientList(recipients, 'to'),
|
|
56
|
+
cc: recipientList(recipients, 'cc'),
|
|
57
|
+
bcc: recipientList(recipients, 'bcc'),
|
|
58
|
+
subject: fields.subject,
|
|
59
|
+
date: date !== undefined ? isoDate(date) : undefined,
|
|
60
|
+
body: body.replace(/\s+$/, ''),
|
|
61
|
+
bodySource,
|
|
62
|
+
attachments: (fields.attachments ?? []).map(
|
|
63
|
+
(a): EmailAttachment => ({
|
|
64
|
+
name: a.fileName ?? a.fileNameShort ?? a.name ?? 'unnamed attachment',
|
|
65
|
+
mimeType: a.attachMimeTag,
|
|
66
|
+
sizeBytes: a.contentLength,
|
|
67
|
+
}),
|
|
68
|
+
),
|
|
69
|
+
};
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** The HTML body, whichever MAPI property carries it (string, or utf-8 bytes). */
|
|
73
|
+
function htmlBody(fields: FieldsData): string | undefined {
|
|
74
|
+
if (fields.bodyHtml !== undefined && fields.bodyHtml.trim() !== '') return fields.bodyHtml;
|
|
75
|
+
if (fields.html instanceof Uint8Array && fields.html.length > 0) {
|
|
76
|
+
return Buffer.from(fields.html).toString('utf8');
|
|
77
|
+
}
|
|
78
|
+
return undefined;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** `Name <addr>` / `Name` / `addr` — whatever the message carries. */
|
|
82
|
+
function mailboxText(name: string | undefined, address: string | undefined): string | undefined {
|
|
83
|
+
const n = name?.trim() ?? '';
|
|
84
|
+
const a = address?.trim() ?? '';
|
|
85
|
+
if (n !== '' && a !== '' && n !== a) return `${n} <${a}>`;
|
|
86
|
+
if (a !== '') return a;
|
|
87
|
+
return n !== '' ? n : undefined;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* The bucket a recipient belongs to. msgreader maps the MAPI `PidTagRecipientType`
|
|
92
|
+
* values 1/2/3 to these strings itself (lib/MsgReader.js, the `recipType` case)
|
|
93
|
+
* — but ONLY those three: any other raw PT_LONG value, e.g. `MAPI_TO | MAPI_P1`
|
|
94
|
+
* (0x10000001) on a resubmitted message, leaks through as a NUMBER despite the
|
|
95
|
+
* `'to' | 'cc' | 'bcc'` typing. Mask the resubmit/submitted flag bits and remap
|
|
96
|
+
* so such a recipient keeps its line instead of vanishing; anything else
|
|
97
|
+
* (including untyped) counts as `to`, matching the frontend's `msgMessage.ts`.
|
|
98
|
+
*/
|
|
99
|
+
function recipientBucket(recipType: unknown): 'to' | 'cc' | 'bcc' {
|
|
100
|
+
if (recipType === 'to' || recipType === 'cc' || recipType === 'bcc') return recipType;
|
|
101
|
+
if (typeof recipType === 'number') {
|
|
102
|
+
const base = recipType & 0x0fffffff; // strip MAPI_SUBMITTED (0x80000000) / MAPI_P1 (0x10000000)
|
|
103
|
+
if (base === 2) return 'cc';
|
|
104
|
+
if (base === 3) return 'bcc';
|
|
105
|
+
}
|
|
106
|
+
return 'to';
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** The comma-joined mailboxes of one recipient type. Untyped recipients count as `to`. */
|
|
110
|
+
function recipientList(recipients: readonly FieldsData[], type: 'to' | 'cc' | 'bcc'): string | undefined {
|
|
111
|
+
const s = recipients
|
|
112
|
+
.filter((r) => recipientBucket(r.recipType) === type)
|
|
113
|
+
.map((r) => mailboxText(r.name, r.smtpAddress ?? r.email))
|
|
114
|
+
.filter((t): t is string => t !== undefined)
|
|
115
|
+
.join(', ');
|
|
116
|
+
return s === '' ? undefined : s;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/** msgreader emits RFC-1123 GMT strings; normalize to ISO, keep raw when unparseable. */
|
|
120
|
+
function isoDate(value: string): string {
|
|
121
|
+
const d = new Date(value);
|
|
122
|
+
return Number.isNaN(d.getTime()) ? value : d.toISOString();
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* CJS/ESM interop: msgreader is CJS with a transpiled `exports.default`.
|
|
127
|
+
* Vitest's transform hands the class straight through the default import, but
|
|
128
|
+
* NATIVE Node ESM (the built `dist/`) hands the exports OBJECT — so unwrap
|
|
129
|
+
* `.default` when it is there.
|
|
130
|
+
*/
|
|
131
|
+
type MsgReaderClass = typeof MsgReaderImport;
|
|
132
|
+
const MsgReader: MsgReaderClass =
|
|
133
|
+
(MsgReaderImport as unknown as { default?: MsgReaderClass }).default ?? MsgReaderImport;
|
|
134
|
+
type FieldsData = ReturnType<InstanceType<MsgReaderClass>['getFileData']>;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
import { odfParagraphLines, readOdfContentXml } from './odf-text.js';
|
|
3
|
+
import { localBlocks, localElementBlocks, removeLocalElements } from './ooxml-text.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Extract the text of a `.odp` (OpenDocument Presentation) deck.
|
|
7
|
+
*
|
|
8
|
+
* Slides are the `<draw:page>` elements of `content.xml`, in DOCUMENT order —
|
|
9
|
+
* ODF orders slides in the file itself, so unlike pptx there is no numeric
|
|
10
|
+
* filename sort. Each slide is emitted under a `[slide N]` marker (N = 1-based
|
|
11
|
+
* position); speaker notes (`<presentation:notes>` inside the page) follow
|
|
12
|
+
* under `[slide N notes]` when non-empty. Within a slide, each `<text:p>` in
|
|
13
|
+
* its frames is a line; spans concatenate with no separator.
|
|
14
|
+
*/
|
|
15
|
+
export function extractOdp(bytes: Buffer): ExtractResult {
|
|
16
|
+
const content = readOdfContentXml(bytes, '.odp');
|
|
17
|
+
if (!content.ok) return content;
|
|
18
|
+
|
|
19
|
+
const pages = drawPageBlocks(content.xml);
|
|
20
|
+
if (pages.length === 0) {
|
|
21
|
+
return { ok: false, message: 'could not be parsed as a .odp (no draw:page elements in content.xml)' };
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
const lines: string[] = [];
|
|
25
|
+
let anyNotes = false;
|
|
26
|
+
pages.forEach((page, i) => {
|
|
27
|
+
// Split the notes part out FIRST so its paragraphs don't render as slide text.
|
|
28
|
+
// Notes read by the parser and matched on their LOCAL name: a comment that
|
|
29
|
+
// resembled `<presentation:notes>` used to be emitted as real speaker notes,
|
|
30
|
+
// and a deck binding the presentation namespace to another prefix had none.
|
|
31
|
+
const notesXml = localBlocks(page, 'notes')[0] ?? '';
|
|
32
|
+
// The slide's own text is the page with the notes ELEMENTS removed by
|
|
33
|
+
// their parsed boundaries — global string replacement of the notes BODY
|
|
34
|
+
// also deleted slide text that happened to serialize identically to it.
|
|
35
|
+
const slideXml = removeLocalElements(page, ['notes']);
|
|
36
|
+
lines.push(`[slide ${i + 1}]`);
|
|
37
|
+
lines.push(...odfParagraphLines(slideXml));
|
|
38
|
+
const noteLines = odfParagraphLines(notesXml);
|
|
39
|
+
if (noteLines.length > 0) {
|
|
40
|
+
anyNotes = true;
|
|
41
|
+
lines.push(`[slide ${i + 1} notes]`);
|
|
42
|
+
lines.push(...noteLines);
|
|
43
|
+
}
|
|
44
|
+
});
|
|
45
|
+
return {
|
|
46
|
+
ok: true,
|
|
47
|
+
summary: `${pages.length} slide${pages.length === 1 ? '' : 's'}${anyNotes ? ' + notes' : ''}; layout, images and formatting omitted`,
|
|
48
|
+
text: lines.join('\n'),
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* The `<draw:page>…</draw:page>` bodies in document order (pages never nest).
|
|
54
|
+
* A SELF-CLOSING `<draw:page/>` is a legal, fully blank slide — it yields ''
|
|
55
|
+
* so the deck's numbering (and a deliberately blank deck) stays correct.
|
|
56
|
+
*/
|
|
57
|
+
function drawPageBlocks(xml: string): string[] {
|
|
58
|
+
// The shared quote-aware scanner: a `/>` INSIDE a quoted attribute value (a
|
|
59
|
+
// page name like `a/>b`) is part of the value, never the self-closing
|
|
60
|
+
// delimiter, and a page whose close tag is missing costs one scan of the
|
|
61
|
+
// document rather than one per opener (see `xmlElementBlocks`).
|
|
62
|
+
return localElementBlocks(xml, ['page']).map((e) => e.body ?? '');
|
|
63
|
+
}
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
import {
|
|
3
|
+
attrByLocalName,
|
|
4
|
+
decodeXmlEntities,
|
|
5
|
+
localElementBlocks,
|
|
6
|
+
localName,
|
|
7
|
+
walkLocalElementBlocks,
|
|
8
|
+
} from './ooxml-text.js';
|
|
9
|
+
import {
|
|
10
|
+
odfParagraphBlocks,
|
|
11
|
+
odfParagraphText,
|
|
12
|
+
readOdfContentXml,
|
|
13
|
+
} from './odf-text.js';
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Per-sheet extraction caps — the SAME bounds as the xlsx extractor. ODF is
|
|
17
|
+
* fond of `table:number-columns-repeated="16384"` (or a million empty trailing
|
|
18
|
+
* rows) to pad a sheet to the grid, so repeats are expanded BOUNDED and the
|
|
19
|
+
* extraction says when it truncated (a `[sheet truncated …]` line right under
|
|
20
|
+
* the sheet marker). Trailing EMPTY cells/rows are trimmed before their
|
|
21
|
+
* repeats are applied at all, so grid padding never counts as truncation.
|
|
22
|
+
*/
|
|
23
|
+
const MAX_ROWS_PER_SHEET = 10_000;
|
|
24
|
+
const MAX_COLS_PER_SHEET = 200;
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Extract a `.ods` (OpenDocument Spreadsheet) workbook: per `<table:table>`
|
|
28
|
+
* (sheet) a `[sheet: Name]` marker (the `table:name` attribute), then the rows
|
|
29
|
+
* as tab-separated cell text. A cell's text is its `<text:p>` content
|
|
30
|
+
* (multiple paragraphs join with a space — a newline would break the row
|
|
31
|
+
* line); covered cells (under a merge) render empty.
|
|
32
|
+
*/
|
|
33
|
+
export function extractOds(bytes: Buffer): ExtractResult {
|
|
34
|
+
const content = readOdfContentXml(bytes, '.ods');
|
|
35
|
+
if (!content.ok) return content;
|
|
36
|
+
|
|
37
|
+
const tables = tableBlocks(content.xml);
|
|
38
|
+
if (tables.length === 0) {
|
|
39
|
+
return { ok: false, message: 'could not be parsed as a .ods (no table:table elements in content.xml)' };
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const lines: string[] = [];
|
|
43
|
+
for (const table of tables) {
|
|
44
|
+
lines.push(`[sheet: ${table.name}]`);
|
|
45
|
+
const { rows, truncated } = expandRows(table.xml);
|
|
46
|
+
if (truncated.length > 0) lines.push(`[sheet truncated to the ${truncated.join(' and ')}]`);
|
|
47
|
+
lines.push(...rows.map((cells) => cells.join('\t')));
|
|
48
|
+
}
|
|
49
|
+
return {
|
|
50
|
+
ok: true,
|
|
51
|
+
summary: `${tables.length} sheet${tables.length === 1 ? '' : 's'}, rows as tab-separated values; formulas, formatting and charts omitted`,
|
|
52
|
+
text: lines.join('\n'),
|
|
53
|
+
};
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* The `<table:table>` blocks with their decoded `table:name`, in document
|
|
58
|
+
* order. Non-greedy close — a nested table (legal in ODF text documents, not
|
|
59
|
+
* produced by spreadsheets) would end the outer block early, degrading
|
|
60
|
+
* grouping but never crashing.
|
|
61
|
+
*/
|
|
62
|
+
function tableBlocks(xml: string): Array<{ name: string; xml: string }> {
|
|
63
|
+
// Read by the parser and matched on the LOCAL name: a comment or CDATA
|
|
64
|
+
// section holding a table-looking fragment used to answer as a real sheet,
|
|
65
|
+
// and a document binding the table namespace to another prefix had none.
|
|
66
|
+
const out: Array<{ name: string; xml: string }> = [];
|
|
67
|
+
for (const table of localElementBlocks(xml, ['table'])) {
|
|
68
|
+
const raw = attrByLocalName(table.attributes, 'name');
|
|
69
|
+
out.push({
|
|
70
|
+
// Control separators become spaces: a name holding an encoded newline
|
|
71
|
+
// or tab (` `) would corrupt the `[sheet: …]` marker's own line and
|
|
72
|
+
// the TSV structure under it.
|
|
73
|
+
name: raw ? decodeXmlEntities(raw).replace(/[\t\n\r]+/g, ' ') : `Sheet${out.length + 1}`,
|
|
74
|
+
xml: table.body ?? '',
|
|
75
|
+
});
|
|
76
|
+
}
|
|
77
|
+
return out;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Expand a sheet's rows with BOUNDED repeat handling, INCREMENTALLY — a row's
|
|
82
|
+
* expansion lands in the capped output as it parses, so the caps bound memory
|
|
83
|
+
* as well as output (materializing every row's cells before consulting the
|
|
84
|
+
* cap let an accepted ODS allocate its whole expansion first):
|
|
85
|
+
*
|
|
86
|
+
* - all-empty rows are buffered as a COUNT (with their
|
|
87
|
+
* `table:number-rows-repeated` applied) and flushed only when a non-empty
|
|
88
|
+
* row follows, so a million-row empty tail simply disappears,
|
|
89
|
+
* - once the row cap is hit, the remaining rows are never parsed at all.
|
|
90
|
+
*
|
|
91
|
+
* `truncated` lists what the caps cut (mirrors the xlsx extractor's note).
|
|
92
|
+
*/
|
|
93
|
+
function expandRows(tableXml: string): { rows: string[][]; truncated: string[] } {
|
|
94
|
+
// The shared quote-aware scanner as a WALK, not an array: materializing
|
|
95
|
+
// every row block before consulting the cap let a sheet of >10k explicit
|
|
96
|
+
// rows allocate them all first. Each row lands here as it parses, and the
|
|
97
|
+
// visitor's `true` stops the scan at the cap — a self-closing row WITH
|
|
98
|
+
// attributes is still a row, a `/>` inside a quoted attribute value is not
|
|
99
|
+
// a delimiter, and an UNCLOSED row costs one scan of the sheet rather than
|
|
100
|
+
// one per opener (see `xmlElementBlocks`).
|
|
101
|
+
const rows: string[][] = [];
|
|
102
|
+
let pendingEmpty = 0;
|
|
103
|
+
let rowsTruncated = false;
|
|
104
|
+
let colsTruncated = false;
|
|
105
|
+
walkLocalElementBlocks(tableXml, ['table-row'], (row) => {
|
|
106
|
+
const repeat = repeatCount(attrByLocalName(row.attributes, 'number-rows-repeated'));
|
|
107
|
+
const cells = expandCells(row.body ?? '');
|
|
108
|
+
if (cells.cells.length === 0) {
|
|
109
|
+
// Empty rows are interior padding until a non-empty row proves it —
|
|
110
|
+
// trailing ones are dropped with their repeats (grid padding, not data).
|
|
111
|
+
pendingEmpty += repeat;
|
|
112
|
+
return false;
|
|
113
|
+
}
|
|
114
|
+
if (cells.truncated) colsTruncated = true;
|
|
115
|
+
for (; pendingEmpty > 0 && rows.length < MAX_ROWS_PER_SHEET; pendingEmpty--) rows.push([]);
|
|
116
|
+
let i = 0;
|
|
117
|
+
for (; i < repeat && rows.length < MAX_ROWS_PER_SHEET; i++) rows.push(cells.cells);
|
|
118
|
+
if (pendingEmpty > 0 || i < repeat) {
|
|
119
|
+
// The cap cut real content (a sheet that merely FILLS it is not truncated).
|
|
120
|
+
rowsTruncated = true;
|
|
121
|
+
return true;
|
|
122
|
+
}
|
|
123
|
+
return false;
|
|
124
|
+
});
|
|
125
|
+
const truncated: string[] = [];
|
|
126
|
+
if (rowsTruncated) truncated.push(`first ${MAX_ROWS_PER_SHEET} rows`);
|
|
127
|
+
if (colsTruncated) truncated.push(`first ${MAX_COLS_PER_SHEET} columns`);
|
|
128
|
+
return { rows, truncated };
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* One row's cell texts: `<table:table-cell>` / `<table:covered-table-cell>`
|
|
133
|
+
* in order, expanded INCREMENTALLY like the rows above — trailing EMPTY cells
|
|
134
|
+
* are buffered as a count (their `table:number-columns-repeated` never
|
|
135
|
+
* expands) and the parse stops at the column cap.
|
|
136
|
+
*/
|
|
137
|
+
function expandCells(rowXml: string): { cells: string[]; truncated: boolean } {
|
|
138
|
+
// Same walking scanner as the rows above — a row spelling a million
|
|
139
|
+
// explicit cells stops parsing at the column cap too.
|
|
140
|
+
const cells: string[] = [];
|
|
141
|
+
let pendingEmpty = 0;
|
|
142
|
+
let truncated = false;
|
|
143
|
+
walkLocalElementBlocks(rowXml, ['table-cell', 'covered-table-cell'], (cell) => {
|
|
144
|
+
const repeat = repeatCount(attrByLocalName(cell.attributes, 'number-columns-repeated'));
|
|
145
|
+
// Covered cells carry no own text anyway. Element-produced newlines/tabs
|
|
146
|
+
// INSIDE a cell (<text:line-break/>, <text:tab/>) become single spaces:
|
|
147
|
+
// the extraction's contract is one row per line with tab-separated cells,
|
|
148
|
+
// and a literal \n or \t inside a cell's text would silently break both.
|
|
149
|
+
// A COVERED cell is the hidden half of a merge: the visible cell carries
|
|
150
|
+
// the text. Such a cell may still hold stale content, and emitting it put
|
|
151
|
+
// a value in the grid where the sheet shows none.
|
|
152
|
+
const text = localName(cell.name) === 'covered-table-cell'
|
|
153
|
+
? ''
|
|
154
|
+
: odfParagraphBlocks(cell.body ?? '')
|
|
155
|
+
.map(odfParagraphText)
|
|
156
|
+
.join(' ')
|
|
157
|
+
.replace(/[\t\n\r]+/g, ' ');
|
|
158
|
+
if (text === '') {
|
|
159
|
+
pendingEmpty += repeat;
|
|
160
|
+
return false;
|
|
161
|
+
}
|
|
162
|
+
for (; pendingEmpty > 0 && cells.length < MAX_COLS_PER_SHEET; pendingEmpty--) cells.push('');
|
|
163
|
+
let i = 0;
|
|
164
|
+
for (; i < repeat && cells.length < MAX_COLS_PER_SHEET; i++) cells.push(text);
|
|
165
|
+
if (pendingEmpty > 0 || i < repeat) {
|
|
166
|
+
// The cap cut real content (a row that merely FILLS it is not truncated).
|
|
167
|
+
truncated = true;
|
|
168
|
+
return true;
|
|
169
|
+
}
|
|
170
|
+
return false;
|
|
171
|
+
});
|
|
172
|
+
return { cells, truncated };
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/** A `…-repeated="N"` attribute value, clamped to a sane positive integer. */
|
|
176
|
+
function repeatCount(raw: string | undefined): number {
|
|
177
|
+
// Decoded first: the block scanner hands attribute values RAW, and a repeat
|
|
178
|
+
// legally written with character references (`10`) must count as
|
|
179
|
+
// 10, not silently fall back to 1.
|
|
180
|
+
const n = raw !== undefined ? parseInt(decodeXmlEntities(raw), 10) : 1;
|
|
181
|
+
return Number.isFinite(n) && n >= 1 ? n : 1;
|
|
182
|
+
}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
import { localBlocks, removeLocalElements } from './ooxml-text.js';
|
|
3
|
+
import { odfParagraphBlocks, odfParagraphText, readOdfContentXml } from './odf-text.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Extract the BODY text of a `.odt` (OpenDocument Text) document.
|
|
7
|
+
*
|
|
8
|
+
* An odt is a zip whose main part is `content.xml`; the body lives under
|
|
9
|
+
* `<office:text>`. Headers/footers are skipped like docx — in ODF they live in
|
|
10
|
+
* `styles.xml`, which is never opened, so reading `content.xml` alone IS the
|
|
11
|
+
* body-only extraction.
|
|
12
|
+
*
|
|
13
|
+
* Paragraphs (`<text:p>`) and headings (`<text:h>`, heading text as its own
|
|
14
|
+
* line) become lines in document order; `<text:span>` runs inside concatenate
|
|
15
|
+
* with NO separator, and the ODF whitespace elements (`<text:tab/>`,
|
|
16
|
+
* `<text:line-break/>`, `<text:s text:c="N"/>`) render as real characters —
|
|
17
|
+
* see `odfParagraphText`.
|
|
18
|
+
*/
|
|
19
|
+
export function extractOdt(bytes: Buffer): ExtractResult {
|
|
20
|
+
const content = readOdfContentXml(bytes, '.odt');
|
|
21
|
+
if (!content.ok) return content;
|
|
22
|
+
|
|
23
|
+
// Table cells contain their own <text:p>, so the flat paragraph scan renders
|
|
24
|
+
// table text too (one line per cell paragraph, like the raw document order).
|
|
25
|
+
// The text BODY, read by the parser and matched on its LOCAL name: a
|
|
26
|
+
// comment mentioning `</office:text>` used to terminate the body early and
|
|
27
|
+
// drop every paragraph after it, and an ODT binding the office namespace to
|
|
28
|
+
// another prefix had no body at all.
|
|
29
|
+
const body = localBlocks(content.xml, 'text')[0] ?? content.xml;
|
|
30
|
+
|
|
31
|
+
// Tracked-change bookkeeping is not body text: `<text:tracked-changes>`
|
|
32
|
+
// stores every DELETION's content as ordinary paragraphs, so the flat scan
|
|
33
|
+
// below would read deleted text back in as document lines. Removed by its
|
|
34
|
+
// parsed element boundaries before the paragraph walk.
|
|
35
|
+
// `<office:annotation>` is a COMMENT on the document, stored as ordinary
|
|
36
|
+
// paragraphs: read flat, a reviewer's note came back as a document line.
|
|
37
|
+
const visible = removeLocalElements(body, ['tracked-changes', 'annotation']);
|
|
38
|
+
|
|
39
|
+
const lines = odfParagraphBlocks(visible).map(odfParagraphText);
|
|
40
|
+
const paragraphs = lines.length;
|
|
41
|
+
while (lines.length > 0 && lines[lines.length - 1].trim() === '') lines.pop();
|
|
42
|
+
|
|
43
|
+
return {
|
|
44
|
+
ok: true,
|
|
45
|
+
summary: `${paragraphs} paragraph${paragraphs === 1 ? '' : 's'}; layout, images and formatting omitted`,
|
|
46
|
+
text: lines.join('\n'),
|
|
47
|
+
};
|
|
48
|
+
}
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Extract a PDF's TEXT LAYER page by page with pdf.js (`pdfjs-dist`,
|
|
6
|
+
* Mozilla's maintained renderer — chosen over the unmaintained thin wrappers
|
|
7
|
+
* around it). The LEGACY build is the one supported under Node; it is loaded
|
|
8
|
+
* lazily (and once) because it is a heavyweight module most deployments only
|
|
9
|
+
* need after the first PDF read.
|
|
10
|
+
*
|
|
11
|
+
* Layout heuristic: text items on one line are joined with single spaces; a
|
|
12
|
+
* new line starts when pdf.js flags an EOL or the item's Y position jumps.
|
|
13
|
+
* A PDF with NO text layer (a scan) extracts to just the `[page N]` markers,
|
|
14
|
+
* and the summary says "no text layer (scanned document?)" — no OCR in v1.
|
|
15
|
+
*/
|
|
16
|
+
/**
|
|
17
|
+
* How much DECODED text one PDF may yield before extraction gives up.
|
|
18
|
+
*
|
|
19
|
+
* The raw-size cap below bounds what arrives; it does not bound what comes
|
|
20
|
+
* out. PDF text lives in compressed streams, so a file comfortably under
|
|
21
|
+
* 50 MB can decode to far more than that, and every character of it is held
|
|
22
|
+
* in `lines` until the extraction returns. This bound is the decoded
|
|
23
|
+
* counterpart, checked as the text accumulates rather than after.
|
|
24
|
+
*/
|
|
25
|
+
const MAX_PDF_TEXT_CHARS = 20 * 1024 * 1024; // 20M chars of extracted text
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* How many pages one PDF may have before extraction gives up.
|
|
29
|
+
*
|
|
30
|
+
* The decoded-text bound does not cover a document whose cost is its PAGE
|
|
31
|
+
* COUNT rather than its prose: every page costs a `getPage`, a
|
|
32
|
+
* `getTextContent` and a retained `[page N]` marker even when it holds no
|
|
33
|
+
* text at all, so a file declaring hundreds of thousands of empty pages spends
|
|
34
|
+
* minutes and megabytes without ever tripping a character budget. Real
|
|
35
|
+
* documents do not come close — a 2,000-page manual is an outlier.
|
|
36
|
+
*/
|
|
37
|
+
const MAX_PDF_PAGES = 10_000;
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* How many text items one PAGE may hold. Items arrive through
|
|
41
|
+
* `streamTextContent` in small chunks (~100 items each), so this bound — like
|
|
42
|
+
* the character budget — fires while the page is still streaming, not after
|
|
43
|
+
* it has materialized. It exists because item COUNT is its own cost: each
|
|
44
|
+
* item is a retained heap object, and a page of empty-string items would
|
|
45
|
+
* never trip the character budget.
|
|
46
|
+
*/
|
|
47
|
+
const MAX_PDF_ITEMS_PER_PAGE = 200_000;
|
|
48
|
+
|
|
49
|
+
/** The typed failure both decoded-text bounds return. */
|
|
50
|
+
function overBudget(): ExtractResult {
|
|
51
|
+
return {
|
|
52
|
+
ok: false,
|
|
53
|
+
message: `could not be extracted as a PDF (its text decodes to over ${MAX_PDF_TEXT_CHARS} characters — over the extraction limit)`,
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export async function extractPdf(bytes: Buffer): Promise<ExtractResult> {
|
|
58
|
+
// The same bounded-read guard the zip-based extractors apply per part: a
|
|
59
|
+
// PDF has no compressed container to pre-scan, so the bound is simply the
|
|
60
|
+
// file's raw size, checked before pdf.js parses anything.
|
|
61
|
+
if (bytes.length > MAX_DOC_PART_BYTES) {
|
|
62
|
+
return {
|
|
63
|
+
ok: false,
|
|
64
|
+
message: `could not be extracted as a PDF (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
let doc: Awaited<ReturnType<typeof openPdf>>;
|
|
68
|
+
try {
|
|
69
|
+
doc = await openPdf(bytes);
|
|
70
|
+
} catch (err) {
|
|
71
|
+
return { ok: false, message: `could not be parsed as a PDF (${(err as Error).message})` };
|
|
72
|
+
}
|
|
73
|
+
try {
|
|
74
|
+
if (doc.numPages > MAX_PDF_PAGES) {
|
|
75
|
+
return {
|
|
76
|
+
ok: false,
|
|
77
|
+
message: `could not be extracted as a PDF (it declares ${doc.numPages} pages — over the ${MAX_PDF_PAGES}-page extraction limit)`,
|
|
78
|
+
};
|
|
79
|
+
}
|
|
80
|
+
const lines: string[] = [];
|
|
81
|
+
let textChars = 0;
|
|
82
|
+
let anyText = false;
|
|
83
|
+
for (let n = 1; n <= doc.numPages; n++) {
|
|
84
|
+
lines.push(`[page ${n}]`);
|
|
85
|
+
// The marker is retained text like any other line: a document whose cost
|
|
86
|
+
// is its page count must reach the same bound as one whose cost is prose.
|
|
87
|
+
textChars += n.toString().length + 8;
|
|
88
|
+
if (textChars > MAX_PDF_TEXT_CHARS) return overBudget();
|
|
89
|
+
const page = await doc.getPage(n);
|
|
90
|
+
// `streamTextContent` delivers the page's items in small chunks (~100
|
|
91
|
+
// items each, `getTextContent` is just this stream materialized), so
|
|
92
|
+
// both budgets fire WHILE the page streams: a crafted single page can
|
|
93
|
+
// no longer build its whole item array before a bound trips. Once one
|
|
94
|
+
// does, the reader is cancelled and pdf.js stops producing.
|
|
95
|
+
const reader = (
|
|
96
|
+
page.streamTextContent() as ReadableStream<Awaited<ReturnType<typeof page.getTextContent>>>
|
|
97
|
+
).getReader();
|
|
98
|
+
let pageItems = 0;
|
|
99
|
+
let line = '';
|
|
100
|
+
let lastY: number | undefined;
|
|
101
|
+
const flush = (): void => {
|
|
102
|
+
if (line.trim() !== '') {
|
|
103
|
+
lines.push(line);
|
|
104
|
+
anyText = true;
|
|
105
|
+
// The '\n' the final join emits for this line is retained text too.
|
|
106
|
+
textChars += 1;
|
|
107
|
+
}
|
|
108
|
+
line = '';
|
|
109
|
+
};
|
|
110
|
+
for (;;) {
|
|
111
|
+
const { done, value: chunk } = await reader.read();
|
|
112
|
+
if (done) break;
|
|
113
|
+
pageItems += chunk.items.length;
|
|
114
|
+
if (pageItems > MAX_PDF_ITEMS_PER_PAGE) {
|
|
115
|
+
await reader.cancel().catch(() => undefined);
|
|
116
|
+
page.cleanup();
|
|
117
|
+
return {
|
|
118
|
+
ok: false,
|
|
119
|
+
message: `could not be extracted as a PDF (page ${n} holds more than ${MAX_PDF_ITEMS_PER_PAGE} text items — over the extraction limit)`,
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
for (const item of chunk.items) {
|
|
123
|
+
if (!('str' in item)) continue; // marked-content item — no text
|
|
124
|
+
const y = item.transform?.[5];
|
|
125
|
+
// Y-position jump = new visual line (1pt tolerance for kerning wobble).
|
|
126
|
+
if (typeof y === 'number') {
|
|
127
|
+
if (lastY !== undefined && Math.abs(y - lastY) > 1) flush();
|
|
128
|
+
lastY = y;
|
|
129
|
+
}
|
|
130
|
+
if (item.str !== '') {
|
|
131
|
+
// COUNTED before it is kept — the join space included: the bound
|
|
132
|
+
// exists to stop the decoded text from accumulating, so it must
|
|
133
|
+
// fire mid-page and cover every character the result will hold.
|
|
134
|
+
textChars += item.str.length + (line === '' ? 0 : 1);
|
|
135
|
+
if (textChars > MAX_PDF_TEXT_CHARS) {
|
|
136
|
+
await reader.cancel().catch(() => undefined);
|
|
137
|
+
return overBudget();
|
|
138
|
+
}
|
|
139
|
+
line += (line === '' ? '' : ' ') + item.str;
|
|
140
|
+
}
|
|
141
|
+
if (item.hasEOL) flush();
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
flush();
|
|
145
|
+
page.cleanup();
|
|
146
|
+
}
|
|
147
|
+
const pages = `${doc.numPages} page${doc.numPages === 1 ? '' : 's'}`;
|
|
148
|
+
return anyText
|
|
149
|
+
? { ok: true, summary: `${pages}; layout, images and formatting omitted`, text: lines.join('\n') }
|
|
150
|
+
: { ok: true, summary: `${pages}; no text layer (scanned document?)`, text: lines.join('\n') };
|
|
151
|
+
} catch (err) {
|
|
152
|
+
return { ok: false, message: `could not extract the PDF's text (${(err as Error).message})` };
|
|
153
|
+
} finally {
|
|
154
|
+
await doc.destroy();
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
type PdfJs = typeof import('pdfjs-dist/legacy/build/pdf.mjs');
|
|
159
|
+
let pdfjsPromise: Promise<PdfJs> | undefined;
|
|
160
|
+
|
|
161
|
+
async function openPdf(bytes: Buffer) {
|
|
162
|
+
pdfjsPromise ??= import('pdfjs-dist/legacy/build/pdf.mjs').catch((err: unknown) => {
|
|
163
|
+
// A FAILED load must not be memoized: left in place, the rejected promise
|
|
164
|
+
// would answer every later read and disable PDF extraction for the whole
|
|
165
|
+
// process. Reset so the next read retries the import.
|
|
166
|
+
pdfjsPromise = undefined;
|
|
167
|
+
throw err;
|
|
168
|
+
});
|
|
169
|
+
const { getDocument } = await pdfjsPromise;
|
|
170
|
+
return getDocument({
|
|
171
|
+
// Copy into a fresh Uint8Array: pdf.js TRANSFERS the buffer it is given
|
|
172
|
+
// (detaching it), and the caller's Buffer must stay usable for hashing.
|
|
173
|
+
data: new Uint8Array(bytes),
|
|
174
|
+
// Server side: no font rendering — text content is all we consume.
|
|
175
|
+
disableFontFace: true,
|
|
176
|
+
useSystemFonts: true,
|
|
177
|
+
}).promise;
|
|
178
|
+
}
|