@bevel-software/platform-core-backend 0.11.2 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +1163 -425
- package/dist/core/create-core-server.js +1 -1
- package/dist/core/create-core-server.js.map +1 -1
- package/dist/core/create-core-services.d.ts +2 -0
- package/dist/core/create-core-services.d.ts.map +1 -1
- package/dist/core/create-core-services.js +5 -0
- package/dist/core/create-core-services.js.map +1 -1
- package/dist/core-config.d.ts +7 -0
- package/dist/core-config.d.ts.map +1 -1
- package/dist/core-config.js +9 -0
- package/dist/core-config.js.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.js +7 -1
- package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.js +59 -0
- package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.js +19 -0
- package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.js +151 -0
- package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
- package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
- package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
- package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
- package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
- package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
- package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.js +55 -0
- package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.js +108 -0
- package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.js +30 -0
- package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.js +116 -0
- package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.js +117 -0
- package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
- package/dist/modules/workspace/workspace.tools.d.ts +2 -1
- package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.tools.js +158 -15
- package/dist/modules/workspace/workspace.tools.js.map +1 -1
- package/package.json +9 -4
- package/src/core/create-core-server.ts +1 -1
- package/src/core/create-core-services.ts +6 -0
- package/src/core-config.ts +9 -0
- package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
- package/src/modules/code-mode/code-mode.tool.ts +7 -1
- package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
- package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
- package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
- package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
- package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
- package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
- package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
- package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
- package/src/modules/workspace/file-readers/document-reader.ts +64 -0
- package/src/modules/workspace/file-readers/email-reader.ts +21 -0
- package/src/modules/workspace/file-readers/email-text.ts +193 -0
- package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
- package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
- package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
- package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
- package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
- package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
- package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
- package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
- package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
- package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
- package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
- package/src/modules/workspace/file-readers/file-reader.ts +104 -0
- package/src/modules/workspace/file-readers/image-read.ts +122 -0
- package/src/modules/workspace/file-readers/image-reader.ts +39 -0
- package/src/modules/workspace/file-readers/odf-text.ts +123 -0
- package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
- package/src/modules/workspace/file-readers/text-reader.ts +131 -0
- package/src/modules/workspace/workspace.tools.ts +174 -12
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The FileReader contract: ONE interface every read-path consumer goes
|
|
3
|
+
* through, with per-extension implementations picked by a single registry
|
|
4
|
+
* lookup — the same shape as the frontend's renderer registry.
|
|
5
|
+
*
|
|
6
|
+
* Adding a file format is ONE new reader entry in `createFileReaderRegistry`
|
|
7
|
+
* (file-reader.registry.ts). The three consumers — `read_file`, `grep` and the
|
|
8
|
+
* write-refusal in workspace.tools.ts — dispatch through `readerFor` and never
|
|
9
|
+
* branch on extensions themselves.
|
|
10
|
+
*/
|
|
11
|
+
import { fileExtension } from './doc-extract.types.js';
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* What reading a file produces, before the tool layer shapes it for MCP:
|
|
15
|
+
* text (the honest `[extracted text of …]` marker is ALREADY prepended for
|
|
16
|
+
* extractions, so line numbers match between read_file and grep), an image
|
|
17
|
+
* payload (the tool layer wraps it in the `McpImageResult` sentinel), or a
|
|
18
|
+
* refusal — legacy formats, oversized images, unreadable binary, corrupt
|
|
19
|
+
* documents — whose message is returned as the file's text content.
|
|
20
|
+
*/
|
|
21
|
+
export type ReadResult =
|
|
22
|
+
| { kind: 'text'; text: string }
|
|
23
|
+
| { kind: 'image'; data: string; mimeType: string; note: string }
|
|
24
|
+
| { kind: 'refusal'; message: string };
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* `path` as interpolated into a ONE-LINE notice or refusal: CR/LF are shown
|
|
28
|
+
* as escapes rather than obeyed (the same rule as `extractionMarker`), so a
|
|
29
|
+
* filename cannot forge extra output lines.
|
|
30
|
+
*/
|
|
31
|
+
export function displayPath(path: string): string {
|
|
32
|
+
return path.replace(/[\r\n]/g, (c) => (c === '\r' ? '\\r' : '\\n'));
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Any text interpolated into a ONE-LINE notice — a path, or an extractor's
|
|
37
|
+
* own failure message, which quotes bytes from the file and can therefore
|
|
38
|
+
* carry CR/LF of the document's choosing.
|
|
39
|
+
*/
|
|
40
|
+
export const oneLine = displayPath;
|
|
41
|
+
|
|
42
|
+
/** A per-format file reader. Register implementations in `createFileReaderRegistry`. */
|
|
43
|
+
export interface FileReader {
|
|
44
|
+
/** The extensions this reader owns — lowercase, with the dot. Empty for the default (fallback) reader. */
|
|
45
|
+
readonly extensions: readonly string[];
|
|
46
|
+
/** Read `bytes` (read at `path`) into what the read tools return. Must not throw for bad file content. */
|
|
47
|
+
read(bytes: Buffer, path: string): Promise<ReadResult>;
|
|
48
|
+
/**
|
|
49
|
+
* Text `grep` may search, or null when there is nothing (cheaply) searchable.
|
|
50
|
+
* Default readers: the decoded text content (TextReader; null on NUL bytes) /
|
|
51
|
+
* the CACHED extraction (DocumentReader — cold extraction is grep's call,
|
|
52
|
+
* under its per-walk budget) / absent = never greppable (ImageReader).
|
|
53
|
+
*/
|
|
54
|
+
greppableText?(bytes: Buffer, path: string): Promise<string | null>;
|
|
55
|
+
/** May the agent TEXT-editing tools (write_file/write_files/edit_file) touch this file? */
|
|
56
|
+
readonly textEditable: boolean;
|
|
57
|
+
/**
|
|
58
|
+
* Format-specific copy for the write-refusal thrown when `textEditable` is
|
|
59
|
+
* false (see `assertNotDocumentEdit` in workspace.tools.ts). Absent = the
|
|
60
|
+
* generic extracted-text/round-trip explanation.
|
|
61
|
+
*/
|
|
62
|
+
editRefusal?(path: string): string;
|
|
63
|
+
/**
|
|
64
|
+
* Refusal for OVERWRITING what this file ALREADY holds, or null to allow.
|
|
65
|
+
* Consulted only when `textEditable` is true: that answer covers the
|
|
66
|
+
* FORMAT, this one covers the CONTENT. The fallback reader needs it because
|
|
67
|
+
* an extensionless file may hold anything — `read_file` refuses binary
|
|
68
|
+
* content, and a write gate that did not ask would let an agent overwrite
|
|
69
|
+
* bytes it was never allowed to see.
|
|
70
|
+
*/
|
|
71
|
+
editRefusalForExisting?(bytes: Buffer, path: string): string | null;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* Extension → reader, with a default fallback (the text reader). Built once
|
|
76
|
+
* per deployment by `createFileReaderRegistry`; `readerFor` is the ONE lookup
|
|
77
|
+
* every consumer routes through.
|
|
78
|
+
*/
|
|
79
|
+
export class FileReaderRegistry {
|
|
80
|
+
private readonly byExtension = new Map<string, FileReader>();
|
|
81
|
+
|
|
82
|
+
constructor(
|
|
83
|
+
readers: readonly FileReader[],
|
|
84
|
+
private readonly fallback: FileReader,
|
|
85
|
+
) {
|
|
86
|
+
for (const reader of readers) {
|
|
87
|
+
for (const raw of reader.extensions) {
|
|
88
|
+
// Lookups arrive lowercased (`fileExtension`), so a key stored with any
|
|
89
|
+
// other casing could never match and its reader would silently fall back
|
|
90
|
+
// to the default — routing a .PDF-registered reader nowhere.
|
|
91
|
+
const ext = raw.toLowerCase();
|
|
92
|
+
// Duplicate claims fail LOUDLY: were the later registration to win
|
|
93
|
+
// silently, adding a format could disable an existing reader.
|
|
94
|
+
if (this.byExtension.has(ext)) throw new Error(`two file readers claim the extension "${ext}"`);
|
|
95
|
+
this.byExtension.set(ext, reader);
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** The reader owning `path`'s extension (lowercased by `fileExtension`), or the fallback. */
|
|
101
|
+
readerFor(path: string): FileReader {
|
|
102
|
+
return this.byExtension.get(fileExtension(path)) ?? this.fallback;
|
|
103
|
+
}
|
|
104
|
+
}
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `read_file` image support: which files come back as a native MCP image
|
|
3
|
+
* content block instead of text, the size cap that keeps the encoded payload
|
|
4
|
+
* under what Claude-family clients accept, and the self-describing note that
|
|
5
|
+
* rides beside the image so the transcript still says what was read.
|
|
6
|
+
*
|
|
7
|
+
* Deliberately dependency-free: dimensions are parsed straight from the
|
|
8
|
+
* container headers (PNG/GIF/JPEG) rather than through a native image library.
|
|
9
|
+
* No downscaling in this increment — `sharp` is a heavy native dependency, so
|
|
10
|
+
* an oversized image gets an honest refusal telling the caller to downscale
|
|
11
|
+
* locally or upload a smaller export.
|
|
12
|
+
*/
|
|
13
|
+
import { fileExtension } from './doc-extract.types.js';
|
|
14
|
+
|
|
15
|
+
/** The image types `read_file` returns as a native MCP image content block. `.svg` is TEXT and stays on the text path. */
|
|
16
|
+
const IMAGE_MIME_BY_EXT: Record<string, string> = {
|
|
17
|
+
'.png': 'image/png',
|
|
18
|
+
'.jpg': 'image/jpeg',
|
|
19
|
+
'.jpeg': 'image/jpeg',
|
|
20
|
+
// GIF passes through whole under the same cap; clients render the first frame.
|
|
21
|
+
'.gif': 'image/gif',
|
|
22
|
+
'.webp': 'image/webp',
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
/** The extensions the `ImageReader` registers for — exactly the map above. */
|
|
26
|
+
export const IMAGE_EXTENSIONS: readonly string[] = Object.keys(IMAGE_MIME_BY_EXT);
|
|
27
|
+
|
|
28
|
+
/** Is this a file `read_file` should return as an image? */
|
|
29
|
+
export function isImageFile(path: string): boolean {
|
|
30
|
+
return fileExtension(path) in IMAGE_MIME_BY_EXT;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** The MCP mime type for an image path — call only after `isImageFile`. */
|
|
34
|
+
export function imageMimeType(path: string): string {
|
|
35
|
+
return IMAGE_MIME_BY_EXT[fileExtension(path)]!;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Cap on the RAW bytes of an image `read_file` will return.
|
|
40
|
+
*
|
|
41
|
+
* The arithmetic: Claude-family clients reject a base64 image payload around
|
|
42
|
+
* 5 MB encoded (5 * 1024 * 1024 = 5,242,880 chars). Base64 inflates by 4/3,
|
|
43
|
+
* so the raw ceiling for that limit is 5,242,880 * 3/4 = 3,932,160 bytes. We
|
|
44
|
+
* cap at 3.5 MiB raw = 3,670,016 bytes, which encodes to
|
|
45
|
+
* ceil(3,670,016 / 3) * 4 = 4,893,356 base64 chars (~4.67 MiB) — under the
|
|
46
|
+
* reject line with headroom for the JSON envelope the sentinel travels in.
|
|
47
|
+
*/
|
|
48
|
+
export const IMAGE_MAX_RAW_BYTES = 3.5 * 1024 * 1024; // = 3,670,016
|
|
49
|
+
|
|
50
|
+
/** Width×height parsed from the container header, when the format makes that cheap. */
|
|
51
|
+
export interface ImageDimensions {
|
|
52
|
+
width: number;
|
|
53
|
+
height: number;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Best-effort dimensions straight from the header bytes: PNG (IHDR), GIF
|
|
58
|
+
* (logical screen descriptor), JPEG (SOFn scan). WebP is left dimensionless —
|
|
59
|
+
* its three sub-formats (VP8/VP8L/VP8X) each encode size differently, and the
|
|
60
|
+
* note is honest without it. Returns undefined on anything unexpected; never
|
|
61
|
+
* throws (a corrupt image must still be delivered or refused, not 500).
|
|
62
|
+
*/
|
|
63
|
+
export function imageDimensions(bytes: Buffer): ImageDimensions | undefined {
|
|
64
|
+
try {
|
|
65
|
+
// PNG: 8-byte signature, then the IHDR chunk — width/height at 16/20 (BE).
|
|
66
|
+
if (bytes.length >= 24 && bytes.readUInt32BE(0) === 0x89504e47 && bytes.toString('latin1', 12, 16) === 'IHDR') {
|
|
67
|
+
return { width: bytes.readUInt32BE(16), height: bytes.readUInt32BE(20) };
|
|
68
|
+
}
|
|
69
|
+
// GIF: the FULL "GIF87a"/"GIF89a" signature, then the logical screen
|
|
70
|
+
// size at 6/8 (LE) — a bare "GIF" prefix is not a GIF header.
|
|
71
|
+
if (bytes.length >= 10) {
|
|
72
|
+
const sig = bytes.toString('latin1', 0, 6);
|
|
73
|
+
if (sig === 'GIF87a' || sig === 'GIF89a') {
|
|
74
|
+
return { width: bytes.readUInt16LE(6), height: bytes.readUInt16LE(8) };
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
// JPEG: walk the marker segments to the first SOFn frame header.
|
|
78
|
+
if (bytes.length >= 4 && bytes[0] === 0xff && bytes[1] === 0xd8) {
|
|
79
|
+
let i = 2;
|
|
80
|
+
while (i + 9 < bytes.length) {
|
|
81
|
+
if (bytes[i] !== 0xff) {
|
|
82
|
+
i += 1; // stray fill byte — resync
|
|
83
|
+
continue;
|
|
84
|
+
}
|
|
85
|
+
const marker = bytes[i + 1]!;
|
|
86
|
+
// Standalone markers (no length field): padding, restarts, SOI/EOI.
|
|
87
|
+
if (marker === 0xff) {
|
|
88
|
+
i += 1;
|
|
89
|
+
continue;
|
|
90
|
+
}
|
|
91
|
+
if (marker === 0x01 || (marker >= 0xd0 && marker <= 0xd9)) {
|
|
92
|
+
i += 2;
|
|
93
|
+
continue;
|
|
94
|
+
}
|
|
95
|
+
// SOFn (C0–CF except the non-frame C4/C8/CC): height at +5, width at +7.
|
|
96
|
+
if (marker >= 0xc0 && marker <= 0xcf && marker !== 0xc4 && marker !== 0xc8 && marker !== 0xcc) {
|
|
97
|
+
return { height: bytes.readUInt16BE(i + 5), width: bytes.readUInt16BE(i + 7) };
|
|
98
|
+
}
|
|
99
|
+
i += 2 + bytes.readUInt16BE(i + 2);
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
} catch {
|
|
103
|
+
// Truncated/corrupt header — the note simply omits dimensions.
|
|
104
|
+
}
|
|
105
|
+
return undefined;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/** The one-line note emitted as a text block beside the image, so the transcript names what it shows. */
|
|
109
|
+
export function imageNote(path: string, mime: string, sizeBytes: number, dims: ImageDimensions | undefined): string {
|
|
110
|
+
const dimsPart = dims ? `, ${dims.width}×${dims.height} px` : '';
|
|
111
|
+
return `[image: ${path} — ${mime}, ${sizeBytes} bytes${dimsPart}]`;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/** The honest refusal for an image over {@link IMAGE_MAX_RAW_BYTES} — returned as the file's text content. */
|
|
115
|
+
export function oversizedImageNotice(path: string, mime: string, sizeBytes: number): string {
|
|
116
|
+
return (
|
|
117
|
+
`[${path} is a ${mime} image of ${sizeBytes} bytes — too large to return over MCP. The cap is ` +
|
|
118
|
+
`${IMAGE_MAX_RAW_BYTES} bytes (3.5 MiB) of raw image data, because base64 encoding inflates it by 4/3 and ` +
|
|
119
|
+
'Claude-family clients reject images near 5 MB encoded. Downscale the image locally or upload a smaller ' +
|
|
120
|
+
'export, then read that file instead.]'
|
|
121
|
+
);
|
|
122
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import type { FileReader, ReadResult } from './file-reader.js';
|
|
2
|
+
import {
|
|
3
|
+
IMAGE_EXTENSIONS,
|
|
4
|
+
IMAGE_MAX_RAW_BYTES,
|
|
5
|
+
imageDimensions,
|
|
6
|
+
imageMimeType,
|
|
7
|
+
imageNote,
|
|
8
|
+
oversizedImageNotice,
|
|
9
|
+
} from './image-read.js';
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* FileReader over the image types `read_file` returns as a native MCP image
|
|
13
|
+
* content block (`.svg` is text and stays on the text path). Within the raw
|
|
14
|
+
* cap the read yields the picture itself (base64 + mime + the self-describing
|
|
15
|
+
* note); over it, the honest downscale refusal — see image-read.ts for the
|
|
16
|
+
* cap arithmetic and header-parsing details.
|
|
17
|
+
*
|
|
18
|
+
* No `greppableText`: a picture is never text-searchable. And images stay
|
|
19
|
+
* `textEditable` — the write tools only refuse formats whose reads are lossy
|
|
20
|
+
* EXTRACTIONS (documents); an image read is the real bytes, and image writes
|
|
21
|
+
* were never gated.
|
|
22
|
+
*/
|
|
23
|
+
export class ImageReader implements FileReader {
|
|
24
|
+
readonly extensions: readonly string[] = IMAGE_EXTENSIONS;
|
|
25
|
+
readonly textEditable = true;
|
|
26
|
+
|
|
27
|
+
async read(bytes: Buffer, path: string): Promise<ReadResult> {
|
|
28
|
+
const mime = imageMimeType(path);
|
|
29
|
+
if (bytes.length > IMAGE_MAX_RAW_BYTES) {
|
|
30
|
+
return { kind: 'refusal', message: oversizedImageNotice(path, mime, bytes.length) };
|
|
31
|
+
}
|
|
32
|
+
return {
|
|
33
|
+
kind: 'image',
|
|
34
|
+
data: bytes.toString('base64'),
|
|
35
|
+
mimeType: mime,
|
|
36
|
+
note: imageNote(path, mime, bytes.length, imageDimensions(bytes)),
|
|
37
|
+
};
|
|
38
|
+
}
|
|
39
|
+
}
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
import AdmZip from 'adm-zip';
|
|
2
|
+
import { Parser } from 'htmlparser2';
|
|
3
|
+
import { attrByLocalName, localElementBlocks, localName, zipEntryOversize } from './ooxml-text.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* ODF (OpenDocument) text helpers shared by the odt/odp/ods extractors.
|
|
7
|
+
*
|
|
8
|
+
* Elements are matched by LOCAL name — `p`, not `text:p`. A prefix is the
|
|
9
|
+
* document's own choice, and naming `text:` literally meant an ODF file that
|
|
10
|
+
* defaulted the namespace, or bound it to any other prefix, extracted as
|
|
11
|
+
* empty. This module used to answer that by REWRITING every non-conventional
|
|
12
|
+
* prefix across the whole of `content.xml` before scanning it, a pass that had
|
|
13
|
+
* to be bounded against crafted alias lists and could still corrupt ordinary
|
|
14
|
+
* paragraph text that happened to look like an alias. Matching local names
|
|
15
|
+
* needs none of it.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* The text of one ODF paragraph (`<text:p>` / `<text:h>` content). Character
|
|
20
|
+
* data is concatenated with NO separator — formatting runs (`<text:span>`)
|
|
21
|
+
* split words exactly like OOXML runs do, so ignoring the span tags joins them
|
|
22
|
+
* back. Three ODF whitespace elements are REAL characters and render as such:
|
|
23
|
+
*
|
|
24
|
+
* - `<text:tab/>` → a tab
|
|
25
|
+
* - `<text:line-break/>` → a newline
|
|
26
|
+
* - `<text:s text:c="N"/>` → N spaces (no `text:c` attribute = 1)
|
|
27
|
+
*
|
|
28
|
+
* Read through the parser, so entities decode, CDATA sections are the text
|
|
29
|
+
* they hold rather than markup, and a comment is not paragraph content.
|
|
30
|
+
*/
|
|
31
|
+
/**
|
|
32
|
+
* Cap on the characters the whitespace ELEMENTS may add to ONE paragraph.
|
|
33
|
+
* Each `<text:s text:c="N"/>` is clamped on its own below, but nothing else
|
|
34
|
+
* bounds how many such elements a paragraph may hold — a content.xml well
|
|
35
|
+
* under the 50 MB part limit could still expand to gigabytes of spaces. The
|
|
36
|
+
* budget caps the SUM per paragraph; real layout whitespace is nowhere near it.
|
|
37
|
+
*/
|
|
38
|
+
const MAX_PARAGRAPH_WHITESPACE_CHARS = 10_000;
|
|
39
|
+
|
|
40
|
+
export function odfParagraphText(paragraphXml: string): string {
|
|
41
|
+
let out = '';
|
|
42
|
+
let whitespaceBudget = MAX_PARAGRAPH_WHITESPACE_CHARS;
|
|
43
|
+
const parser = new Parser(
|
|
44
|
+
{
|
|
45
|
+
onopentag(name, attributes) {
|
|
46
|
+
const local = localName(name);
|
|
47
|
+
if (local === 'tab') {
|
|
48
|
+
if (whitespaceBudget > 0) {
|
|
49
|
+
out += '\t';
|
|
50
|
+
whitespaceBudget--;
|
|
51
|
+
}
|
|
52
|
+
} else if (local === 'line-break') {
|
|
53
|
+
if (whitespaceBudget > 0) {
|
|
54
|
+
out += '\n';
|
|
55
|
+
whitespaceBudget--;
|
|
56
|
+
}
|
|
57
|
+
} else if (local === 's') {
|
|
58
|
+
const raw = attrByLocalName(attributes, 'c');
|
|
59
|
+
const count = raw !== undefined ? parseInt(raw, 10) : 1;
|
|
60
|
+
// Bounded defensively — a corrupt attribute must not balloon the
|
|
61
|
+
// extraction — and again by the per-paragraph budget above.
|
|
62
|
+
const spaces = Math.min(Number.isFinite(count) ? Math.min(Math.max(count, 0), 1000) : 1, whitespaceBudget);
|
|
63
|
+
out += ' '.repeat(spaces);
|
|
64
|
+
whitespaceBudget -= spaces;
|
|
65
|
+
}
|
|
66
|
+
},
|
|
67
|
+
ontext(text) {
|
|
68
|
+
out += text;
|
|
69
|
+
},
|
|
70
|
+
},
|
|
71
|
+
{ xmlMode: true, decodeEntities: true },
|
|
72
|
+
);
|
|
73
|
+
parser.write(paragraphXml);
|
|
74
|
+
parser.end();
|
|
75
|
+
return out;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* The `<text:p>` / `<text:h>` paragraph bodies of an XML fragment, in DOCUMENT
|
|
80
|
+
* order (headings interleaved with paragraphs, as written). Self-closing
|
|
81
|
+
* elements (an empty paragraph) yield ''. `<text:page-number>` and friends do
|
|
82
|
+
* not count as paragraphs.
|
|
83
|
+
*/
|
|
84
|
+
export function odfParagraphBlocks(xml: string): string[] {
|
|
85
|
+
return localElementBlocks(xml, ['p', 'h']).map((e) => e.body ?? '');
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
/** Non-empty paragraph texts of an ODF fragment — what odp slides/notes render. */
|
|
90
|
+
export function odfParagraphLines(xml: string): string[] {
|
|
91
|
+
const out: string[] = [];
|
|
92
|
+
for (const p of odfParagraphBlocks(xml)) {
|
|
93
|
+
const text = odfParagraphText(p);
|
|
94
|
+
if (text.trim() !== '') out.push(text);
|
|
95
|
+
}
|
|
96
|
+
return out;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* `content.xml` of an ODF package or a typed could-not-parse failure message
|
|
101
|
+
* (`kind` names the extension for the message, e.g. '.odt').
|
|
102
|
+
*
|
|
103
|
+
* Bounded: the entry's DECLARED uncompressed size is checked against
|
|
104
|
+
* `MAX_DOC_PART_BYTES` (50 MB) before anything inflates, so a zip bomb is a
|
|
105
|
+
* typed refusal, never an allocation.
|
|
106
|
+
*/
|
|
107
|
+
export function readOdfContentXml(
|
|
108
|
+
bytes: Buffer,
|
|
109
|
+
kind: '.odt' | '.odp' | '.ods',
|
|
110
|
+
): { ok: true; xml: string } | { ok: false; message: string } {
|
|
111
|
+
try {
|
|
112
|
+
const zip = new AdmZip(bytes);
|
|
113
|
+
const entry = zip.getEntry('content.xml');
|
|
114
|
+
if (!entry) {
|
|
115
|
+
return { ok: false, message: `could not be parsed as a ${kind} (no content.xml inside the archive)` };
|
|
116
|
+
}
|
|
117
|
+
const oversize = zipEntryOversize(entry);
|
|
118
|
+
if (oversize) return { ok: false, message: `could not be extracted as a ${kind} (${oversize})` };
|
|
119
|
+
return { ok: true, xml: entry.getData().toString('utf8') };
|
|
120
|
+
} catch (err) {
|
|
121
|
+
return { ok: false, message: `could not be parsed as a ${kind} (${(err as Error).message})` };
|
|
122
|
+
}
|
|
123
|
+
}
|