@bevel-software/platform-core-backend 0.11.2 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +1163 -425
- package/dist/core/create-core-server.js +1 -1
- package/dist/core/create-core-server.js.map +1 -1
- package/dist/core/create-core-services.d.ts +2 -0
- package/dist/core/create-core-services.d.ts.map +1 -1
- package/dist/core/create-core-services.js +5 -0
- package/dist/core/create-core-services.js.map +1 -1
- package/dist/core-config.d.ts +7 -0
- package/dist/core-config.d.ts.map +1 -1
- package/dist/core-config.js +9 -0
- package/dist/core-config.js.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.js +7 -1
- package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.js +59 -0
- package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.js +19 -0
- package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.js +151 -0
- package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
- package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
- package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
- package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
- package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
- package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
- package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.js +55 -0
- package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.js +108 -0
- package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.js +30 -0
- package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.js +116 -0
- package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.js +117 -0
- package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
- package/dist/modules/workspace/workspace.tools.d.ts +2 -1
- package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.tools.js +158 -15
- package/dist/modules/workspace/workspace.tools.js.map +1 -1
- package/package.json +9 -4
- package/src/core/create-core-server.ts +1 -1
- package/src/core/create-core-services.ts +6 -0
- package/src/core-config.ts +9 -0
- package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
- package/src/modules/code-mode/code-mode.tool.ts +7 -1
- package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
- package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
- package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
- package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
- package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
- package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
- package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
- package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
- package/src/modules/workspace/file-readers/document-reader.ts +64 -0
- package/src/modules/workspace/file-readers/email-reader.ts +21 -0
- package/src/modules/workspace/file-readers/email-text.ts +193 -0
- package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
- package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
- package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
- package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
- package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
- package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
- package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
- package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
- package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
- package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
- package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
- package/src/modules/workspace/file-readers/file-reader.ts +104 -0
- package/src/modules/workspace/file-readers/image-read.ts +122 -0
- package/src/modules/workspace/file-readers/image-reader.ts +39 -0
- package/src/modules/workspace/file-readers/odf-text.ts +123 -0
- package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
- package/src/modules/workspace/file-readers/text-reader.ts +131 -0
- package/src/modules/workspace/workspace.tools.ts +174 -12
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
import { extractionMarker, fileExtension, type ExtractFn, type ExtractResult } from './doc-extract.types.js';
|
|
2
|
+
import { DocExtractionCache, gitBlobSha } from './extraction-cache.js';
|
|
3
|
+
|
|
4
|
+
/** An extraction ready for a consumer: the honest marker header + the text. */
|
|
5
|
+
export interface DocExtraction {
|
|
6
|
+
/** e.g. `[extracted text of Plugins/GTM/deck.pptx — 14 slides + notes; layout, images and formatting omitted]` */
|
|
7
|
+
marker: string;
|
|
8
|
+
text: string;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
/** What `extract` returns: a marker+text, or a typed could-not-parse failure. */
|
|
12
|
+
export type DocExtractOutcome =
|
|
13
|
+
| ({ ok: true } & DocExtraction)
|
|
14
|
+
| { ok: false; message: string };
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* The content-hash CACHE wrap around document text extraction. The per-format
|
|
18
|
+
* `DocumentReader`s in the file-reader registry stay pure (each pairs one
|
|
19
|
+
* extension with one extract function); this service adds the caching
|
|
20
|
+
* generically — a reader hands its extract function in, and the service only
|
|
21
|
+
* runs it on a cache miss.
|
|
22
|
+
*
|
|
23
|
+
* Caching: keyed by the git BLOB sha of the bytes (see `gitBlobSha`) PLUS the
|
|
24
|
+
* path's extension, stored as `{ summary, text }` WITHOUT the path — the
|
|
25
|
+
* marker is assembled per call so the same content read under two paths
|
|
26
|
+
* (copies, branches) shares one entry yet each read's marker names the path
|
|
27
|
+
* that was read. The extension is part of the key because identical bytes
|
|
28
|
+
* extract DIFFERENTLY per format: a `.odt` renamed `.ods` must run the ods
|
|
29
|
+
* extractor, not return the odt extraction. Extraction is deterministic
|
|
30
|
+
* (stable slide/sheet/page ordering), which is what makes a content-keyed
|
|
31
|
+
* cache correct. Failures are NOT cached (rare, and fail fast). The key also carries
|
|
32
|
+
* `EXTRACTION_SCHEMA`, so an upgrade that changes what an extractor emits does
|
|
33
|
+
* not keep serving the previous release's text for unchanged bytes.
|
|
34
|
+
*/
|
|
35
|
+
/**
|
|
36
|
+
* Bumped whenever extraction OUTPUT changes — a new extractor, a fixed one, a
|
|
37
|
+
* reworded summary, a different marker.
|
|
38
|
+
*
|
|
39
|
+
* The rest of the key is content, and content-addressing is only correct while
|
|
40
|
+
* the same bytes mean the same text. An upgrade that changes what an extractor
|
|
41
|
+
* emits breaks that: every already-cached document would keep serving the OLD
|
|
42
|
+
* extraction forever, because its bytes never changed. Bumping this retires
|
|
43
|
+
* those entries (the cache evicts by age, so they cost nothing for long).
|
|
44
|
+
*
|
|
45
|
+
* v2: extraction moved from hand-rolled scanning to a real XML/HTML parser,
|
|
46
|
+
* which changed self-closing paragraphs, entity edge cases and recovery from
|
|
47
|
+
* malformed parts.
|
|
48
|
+
*/
|
|
49
|
+
export const EXTRACTION_SCHEMA = 'v2';
|
|
50
|
+
|
|
51
|
+
export class DocExtractService {
|
|
52
|
+
private readonly cache: DocExtractionCache;
|
|
53
|
+
/**
|
|
54
|
+
* Cold extractions currently running, by cache key. Two reads of the same
|
|
55
|
+
* document arriving together would otherwise BOTH parse it — the expensive
|
|
56
|
+
* half of this service, run twice for one answer — because neither had
|
|
57
|
+
* written the cache entry yet when the other looked.
|
|
58
|
+
*/
|
|
59
|
+
private readonly inFlight = new Map<string, Promise<ExtractResult>>();
|
|
60
|
+
|
|
61
|
+
constructor(cacheRoot: string) {
|
|
62
|
+
this.cache = new DocExtractionCache(cacheRoot);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* The CACHED extraction for these bytes, or undefined on a cache miss —
|
|
67
|
+
* never extracts. `grep` uses this (via `DocumentReader.greppableText`) to
|
|
68
|
+
* search already-extracted documents for free and apply its per-walk budget
|
|
69
|
+
* only to cold ones.
|
|
70
|
+
*/
|
|
71
|
+
async getCached(path: string, bytes: Buffer): Promise<DocExtraction | undefined> {
|
|
72
|
+
const hit = await this.cache.get(this.cacheKey(path, bytes));
|
|
73
|
+
return hit && { marker: extractionMarker(path, hit.summary), text: hit.text };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Extract via `extractFn` (cache hit or parse + store). `path` supplies the marker AND the key's format part. */
|
|
77
|
+
async extract(path: string, bytes: Buffer, extractFn: ExtractFn): Promise<DocExtractOutcome> {
|
|
78
|
+
const key = this.cacheKey(path, bytes);
|
|
79
|
+
const hit = await this.cache.get(key);
|
|
80
|
+
if (hit) return { ok: true, marker: extractionMarker(path, hit.summary), text: hit.text };
|
|
81
|
+
const running = this.inFlight.get(key);
|
|
82
|
+
const result = await (running ??
|
|
83
|
+
(() => {
|
|
84
|
+
// The cache write stays INSIDE the shared promise: were the entry
|
|
85
|
+
// dropped as soon as parsing settled, a read arriving during the
|
|
86
|
+
// write would miss both the cache and the in-flight map — and parse
|
|
87
|
+
// the same document again.
|
|
88
|
+
const started = (async (): Promise<ExtractResult> => {
|
|
89
|
+
const res = await extractFn(bytes);
|
|
90
|
+
if (res.ok) await this.cache.put(key, { summary: res.summary, text: res.text });
|
|
91
|
+
return res;
|
|
92
|
+
})().finally(() => this.inFlight.delete(key));
|
|
93
|
+
this.inFlight.set(key, started);
|
|
94
|
+
return started;
|
|
95
|
+
})());
|
|
96
|
+
if (!result.ok) return result;
|
|
97
|
+
return { ok: true, marker: extractionMarker(path, result.summary), text: result.text };
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** Content hash + lowercased extension — e.g. `…sha….odt` (see the class doc). */
|
|
101
|
+
private cacheKey(path: string, bytes: Buffer): string {
|
|
102
|
+
return `${gitBlobSha(bytes)}${fileExtension(path)}.${EXTRACTION_SCHEMA}`;
|
|
103
|
+
}
|
|
104
|
+
}
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Server-side document text extraction — shared types.
|
|
3
|
+
*
|
|
4
|
+
* The extractors turn an office document (`.docx` / `.pptx` / `.xlsx`), an
|
|
5
|
+
* OpenDocument file (`.odt` / `.odp` / `.ods`) or a PDF into plain text an
|
|
6
|
+
* agent can read and grep. They are HONEST about being
|
|
7
|
+
* lossy: every successful extraction carries a one-line `summary` the consumer
|
|
8
|
+
* turns into a marker header (`[extracted text of <path> — <summary>]`) so the
|
|
9
|
+
* reader knows it is looking at extracted text, not the file's bytes.
|
|
10
|
+
*
|
|
11
|
+
* The `summary` (not a full marker) is what extractors return and what the
|
|
12
|
+
* cache stores, because the cache is keyed by CONTENT (git blob sha): the same
|
|
13
|
+
* document at two workspace paths shares one cache entry, and baking a path
|
|
14
|
+
* into the cached value would surface the wrong path on the second read.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
/** A successful extraction: the marker-summary line + the extracted text. */
|
|
18
|
+
export interface ExtractedDoc {
|
|
19
|
+
/**
|
|
20
|
+
* One-line description for the marker header, e.g.
|
|
21
|
+
* `14 slides + notes; layout, images and formatting omitted`.
|
|
22
|
+
*/
|
|
23
|
+
summary: string;
|
|
24
|
+
/** The extracted text: paragraphs/rows as lines, with `[slide N]` / `[sheet: Name]` / `[page N]` structure markers. */
|
|
25
|
+
text: string;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Extraction outcome. `ok: false` is a TYPED failure (corrupt/unparseable
|
|
30
|
+
* file) the consumers turn into an honest message — extraction never throws
|
|
31
|
+
* for bad file content, so a broken upload cannot 500 a read.
|
|
32
|
+
*/
|
|
33
|
+
export type ExtractResult =
|
|
34
|
+
| ({ ok: true } & ExtractedDoc)
|
|
35
|
+
| { ok: false; message: string };
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* A format's PURE extract function — bytes in, `ExtractResult` out (async for
|
|
39
|
+
* pdf.js). One per supported format (extract-docx.ts and friends); each is
|
|
40
|
+
* paired with its extension by a `DocumentReader` entry in the file-reader
|
|
41
|
+
* registry, and cache-wrapped by `DocExtractService`.
|
|
42
|
+
*/
|
|
43
|
+
export type ExtractFn = (bytes: Buffer) => ExtractResult | Promise<ExtractResult>;
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Build the honest ONE-LINE header consumers prepend to extracted text.
|
|
47
|
+
*
|
|
48
|
+
* One line is a promise the rest of the read path keeps: grep counts on the
|
|
49
|
+
* marker occupying exactly one, so read_file and grep agree on line numbers.
|
|
50
|
+
* A path may legally carry a CR or LF, which would forge extra lines and shift
|
|
51
|
+
* every number after it, so those are shown as escapes rather than obeyed.
|
|
52
|
+
*/
|
|
53
|
+
export function extractionMarker(path: string, summary: string): string {
|
|
54
|
+
const oneLine = path.replace(/[\r\n]/g, (c) => (c === '\r' ? '\\r' : '\\n'));
|
|
55
|
+
return `[extracted text of ${oneLine} — ${summary}]`;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** Lowercased extension of `path` including the dot, or '' when there is none. */
|
|
59
|
+
export function fileExtension(path: string): string {
|
|
60
|
+
const name = path.slice(path.lastIndexOf('/') + 1);
|
|
61
|
+
const dot = name.lastIndexOf('.');
|
|
62
|
+
return dot > 0 ? name.slice(dot).toLowerCase() : '';
|
|
63
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type { DocExtractService } from './doc-extract.service.js';
|
|
2
|
+
import type { ExtractFn } from './doc-extract.types.js';
|
|
3
|
+
import { displayPath, oneLine, type FileReader, type ReadResult } from './file-reader.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* FileReader over one document format: a thin wrapper pairing the format's
|
|
7
|
+
* PURE extract function (extract-docx.ts and friends) with the shared
|
|
8
|
+
* content-hash extraction cache (`DocExtractService`). Reads return the
|
|
9
|
+
* extraction under its honest `[extracted text of …]` marker; a parse failure
|
|
10
|
+
* becomes a refusal message, never a 500.
|
|
11
|
+
*
|
|
12
|
+
* `grep` semantics: `greppableText` serves only the CACHED extraction (a hit
|
|
13
|
+
* costs one small JSON read), returning null for a cold document — whether to
|
|
14
|
+
* spend grep's per-walk extraction budget on a cold one is the walk's call,
|
|
15
|
+
* which then extracts through `read` (see grepWalk in workspace.tools.ts).
|
|
16
|
+
* grep also branches on `instanceof DocumentReader` for exactly that decision.
|
|
17
|
+
*/
|
|
18
|
+
export class DocumentReader implements FileReader {
|
|
19
|
+
readonly extensions: readonly string[];
|
|
20
|
+
/**
|
|
21
|
+
* `read_file` returns an EXTRACTION for these types, so text written back
|
|
22
|
+
* could not round-trip — the write tools refuse (documents are replaced by
|
|
23
|
+
* uploading a new version).
|
|
24
|
+
*/
|
|
25
|
+
readonly textEditable = false;
|
|
26
|
+
|
|
27
|
+
constructor(
|
|
28
|
+
extension: string,
|
|
29
|
+
private readonly extract: ExtractFn,
|
|
30
|
+
private readonly service: DocExtractService,
|
|
31
|
+
) {
|
|
32
|
+
this.extensions = [extension];
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
async read(bytes: Buffer, path: string): Promise<ReadResult> {
|
|
36
|
+
// An extractor is contracted to ANSWER for bad content rather than throw,
|
|
37
|
+
// but it wraps third-party parsers, and one of those throwing past its own
|
|
38
|
+
// guard is a corrupt file — not a server fault. read_file says so instead
|
|
39
|
+
// of failing the whole tool call with a 500.
|
|
40
|
+
let res: Awaited<ReturnType<typeof this.service.extract>>;
|
|
41
|
+
try {
|
|
42
|
+
res = await this.service.extract(path, bytes, this.extract);
|
|
43
|
+
} catch (err) {
|
|
44
|
+
const reason = err instanceof Error ? err.message : String(err);
|
|
45
|
+
return {
|
|
46
|
+
kind: 'refusal',
|
|
47
|
+
message: `[${displayPath(path)} could not be extracted (${oneLine(reason)}) — the file may be corrupt or mislabeled. To fix it, replace the document by uploading a new version.]`,
|
|
48
|
+
};
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
return res.ok
|
|
52
|
+
? { kind: 'text', text: `${res.marker}\n${res.text}` }
|
|
53
|
+
: {
|
|
54
|
+
kind: 'refusal',
|
|
55
|
+
message: `[${displayPath(path)} ${oneLine(res.message)} — the file may be corrupt or mislabeled. To fix it, replace the document by uploading a new version.]`,
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** The cached extraction (marker line included, so grep's line numbers match read_file's), or null when cold. */
|
|
60
|
+
async greppableText(bytes: Buffer, path: string): Promise<string | null> {
|
|
61
|
+
const hit = await this.service.getCached(path, bytes);
|
|
62
|
+
return hit ? `${hit.marker}\n${hit.text}` : null;
|
|
63
|
+
}
|
|
64
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import { DocumentReader } from './document-reader.js';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* FileReader over one email format (`.eml` / `.msg`): a `DocumentReader` in
|
|
5
|
+
* every mechanical respect — cache-wrapped extraction, greppable when cached,
|
|
6
|
+
* corrupt files answered with the typed could-not-be-parsed refusal — with
|
|
7
|
+
* only the WRITE-refusal copy specialized. The generic document refusal talks
|
|
8
|
+
* about round-tripping an office document; an email deserves the honest
|
|
9
|
+
* version of the same "no": the file is a snapshot of a message, and editing
|
|
10
|
+
* a snapshot's text is not a thing.
|
|
11
|
+
*/
|
|
12
|
+
export class EmailReader extends DocumentReader {
|
|
13
|
+
/** The write-refusal for the agent text-editing tools (see `assertNotDocumentEdit`). */
|
|
14
|
+
editRefusal(path: string): string {
|
|
15
|
+
return (
|
|
16
|
+
`"${path}" is an email file — a snapshot of a message. read_file returns EXTRACTED text for it, ` +
|
|
17
|
+
'and a snapshot cannot be text-edited; to change what is stored, replace the file by uploading ' +
|
|
18
|
+
'a new version.'
|
|
19
|
+
);
|
|
20
|
+
}
|
|
21
|
+
}
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared shaping for the email extractors (`extract-eml.ts` / `extract-msg.ts`).
|
|
3
|
+
*
|
|
4
|
+
* Both formats extract to the SAME text shape, so an agent greps a mailbox
|
|
5
|
+
* without caring which client saved the file:
|
|
6
|
+
*
|
|
7
|
+
* [from] Ada Lovelace <ada@example.com>
|
|
8
|
+
* [to] Bob <bob@example.com>, carol@example.com
|
|
9
|
+
* [subject] Quarterly numbers
|
|
10
|
+
* [date] 2026-01-05T10:00:00.000Z
|
|
11
|
+
*
|
|
12
|
+
* the body…
|
|
13
|
+
*
|
|
14
|
+
* [attachments]
|
|
15
|
+
* report.pdf (application/pdf, 48211 bytes)
|
|
16
|
+
*
|
|
17
|
+
* Header lines are omitted when the message lacks the field (never printed
|
|
18
|
+
* empty). The body prefers the plain-text part; an HTML-only body is stripped
|
|
19
|
+
* to text (block tags become newlines so paragraphs survive) and the marker
|
|
20
|
+
* summary says so. Attachments are LISTED by name only — v1 does not extract
|
|
21
|
+
* inside them, and the summary says that too.
|
|
22
|
+
*/
|
|
23
|
+
import type { ExtractedDoc } from './doc-extract.types.js';
|
|
24
|
+
import { Parser } from 'htmlparser2';
|
|
25
|
+
import { MAX_ELEMENT_DEPTH, TOO_DEEP } from './ooxml-text.js';
|
|
26
|
+
|
|
27
|
+
/** One listed attachment. Size/type are printed only when known. */
|
|
28
|
+
export interface EmailAttachment {
|
|
29
|
+
name: string;
|
|
30
|
+
mimeType?: string;
|
|
31
|
+
sizeBytes?: number;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** Where the body text came from — drives the honest summary + body notes. */
|
|
35
|
+
export type EmailBodySource = 'text' | 'html' | 'rtf-only' | 'none';
|
|
36
|
+
|
|
37
|
+
/** The format-independent email, as far as the extraction cares. */
|
|
38
|
+
export interface EmailModel {
|
|
39
|
+
from?: string;
|
|
40
|
+
to?: string;
|
|
41
|
+
cc?: string;
|
|
42
|
+
bcc?: string;
|
|
43
|
+
subject?: string;
|
|
44
|
+
/** ISO timestamp when the date parsed, the raw header value otherwise. */
|
|
45
|
+
date?: string;
|
|
46
|
+
/** Body as plain text ('' when there is none or it is RTF-only). */
|
|
47
|
+
body: string;
|
|
48
|
+
bodySource: EmailBodySource;
|
|
49
|
+
attachments: EmailAttachment[];
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** The line a body that exists only as RTF gets INSTEAD of body text. */
|
|
53
|
+
export const RTF_ONLY_BODY_LINE = '[body is RTF; no plain-text part]';
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Strip an HTML email body to plain text. Deliberately simple (the same
|
|
57
|
+
* stance as `ooxml-text.ts`): comments and `<style>`/`<script>`/`<head>`/
|
|
58
|
+
* `<title>` containers are dropped whole, `<br>` and block-level tag
|
|
59
|
+
* boundaries become newlines so paragraphs survive, every other tag is
|
|
60
|
+
* removed, and entities are decoded through the module's shared
|
|
61
|
+
* `decodeXmlEntities` (plus ` `, which HTML has and XML does not). Runs
|
|
62
|
+
* of blank lines collapse to one.
|
|
63
|
+
*
|
|
64
|
+
* Implemented as a SINGLE-PASS linear scanner, not regexes: the earlier
|
|
65
|
+
* quote-aware tag regexes re-scanned the remaining body from every `<` when a
|
|
66
|
+
* quoted attribute never closed — malformed input with many `<` characters
|
|
67
|
+
* plus one unterminated quote pinned the server quadratically. The scanner is
|
|
68
|
+
* quote-aware the same way (a `>` INSIDE a quoted attribute value, as in
|
|
69
|
+
* `<a title="a > b">`, never ends the tag early) but amortizes the failures:
|
|
70
|
+
* a scan that reaches end-of-input marks every `<` it passed OUTSIDE quotes
|
|
71
|
+
* as known-literal (their scans would be identical tails), so no position is
|
|
72
|
+
* rescanned from more than the three possible quote states. An unterminated
|
|
73
|
+
* tag is literal text, not a tag — and so is a `<…>` span that names no
|
|
74
|
+
* element at all, which is how `1 < 2 > 0` survives into the body.
|
|
75
|
+
*/
|
|
76
|
+
/** Elements whose CONTENT is not body text and is dropped with the element. */
|
|
77
|
+
const CONTAINER_TAGS = new Set(['script', 'style', 'head', 'title']);
|
|
78
|
+
|
|
79
|
+
/** Elements whose boundaries end a line, so paragraphs survive as paragraphs. */
|
|
80
|
+
const BLOCK_TAGS = new Set([
|
|
81
|
+
'p', 'div', 'section', 'article', 'header', 'footer', 'main',
|
|
82
|
+
'table', 'thead', 'tbody', 'tfoot', 'tr', 'td', 'th',
|
|
83
|
+
'li', 'ul', 'ol', 'dl', 'dt', 'dd', 'blockquote', 'pre', 'hr',
|
|
84
|
+
'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
|
|
85
|
+
]);
|
|
86
|
+
|
|
87
|
+
// Depth bound shared with the OOXML/ODF extractors (`ooxml-text.ts`): mail
|
|
88
|
+
// nests a few dozen levels even at its most table-happy; a crafted body can
|
|
89
|
+
// nest as deep as it has bytes. What was read before the bound is kept.
|
|
90
|
+
|
|
91
|
+
export function htmlToEmailText(html: string): string {
|
|
92
|
+
let s = '';
|
|
93
|
+
let depth = 0;
|
|
94
|
+
let skipping = 0; // inside a container whose content is not body text
|
|
95
|
+
const parser = new Parser(
|
|
96
|
+
{
|
|
97
|
+
onopentag(name) {
|
|
98
|
+
depth++;
|
|
99
|
+
if (CONTAINER_TAGS.has(name)) skipping++;
|
|
100
|
+
else if (name === 'br' || BLOCK_TAGS.has(name)) s += '\n';
|
|
101
|
+
if (depth > MAX_ELEMENT_DEPTH) throw TOO_DEEP;
|
|
102
|
+
},
|
|
103
|
+
ontext(text) {
|
|
104
|
+
if (skipping === 0) s += text;
|
|
105
|
+
},
|
|
106
|
+
onclosetag(name) {
|
|
107
|
+
if (CONTAINER_TAGS.has(name)) {
|
|
108
|
+
if (skipping > 0) skipping--;
|
|
109
|
+
} else if (BLOCK_TAGS.has(name)) s += '\n';
|
|
110
|
+
depth--;
|
|
111
|
+
},
|
|
112
|
+
},
|
|
113
|
+
// HTML mode, not XML: an email body is HTML, with its void elements, its
|
|
114
|
+
// implied closes, its raw-text `<script>`, and its named entities.
|
|
115
|
+
{ decodeEntities: true },
|
|
116
|
+
);
|
|
117
|
+
try {
|
|
118
|
+
parser.write(html);
|
|
119
|
+
parser.end();
|
|
120
|
+
} catch (err) {
|
|
121
|
+
if (err !== TOO_DEEP) throw err;
|
|
122
|
+
parser.reset();
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
// ` ` decodes to U+00A0, which LOOKS like a space and is not one: an
|
|
126
|
+
// agent grepping the extraction for "Para one" would miss a line that reads
|
|
127
|
+
// exactly that. Extracted text is for reading and searching, so the
|
|
128
|
+
// non-breaking space becomes an ordinary one.
|
|
129
|
+
const out: string[] = [];
|
|
130
|
+
for (const raw of s.replace(/ /g, ' ').split('\n')) {
|
|
131
|
+
const line = raw.trim();
|
|
132
|
+
if (line !== '') out.push(line);
|
|
133
|
+
else if (out.length > 0 && out[out.length - 1] !== '') out.push('');
|
|
134
|
+
}
|
|
135
|
+
while (out.length > 0 && out[out.length - 1] === '') out.pop();
|
|
136
|
+
return out.join('\n');
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/** `name (mimeType, N bytes)` with the parenthesis dropped when nothing is known.
|
|
140
|
+
* CR/LF in the metadata are shown as escapes rather than obeyed, so each
|
|
141
|
+
* attachment stays on ONE line and grep line numbers hold. */
|
|
142
|
+
function attachmentLine(a: EmailAttachment): string {
|
|
143
|
+
// The extractors default a MISSING name, but an EMPTY (or whitespace-only)
|
|
144
|
+
// one reaches here as ''; without the same fallback the section printed a
|
|
145
|
+
// blank line — or one starting with the parenthesis — for that attachment.
|
|
146
|
+
const name = a.name.trim() === '' ? 'unnamed attachment' : a.name;
|
|
147
|
+
const details = [a.mimeType, a.sizeBytes !== undefined ? `${a.sizeBytes} bytes` : undefined]
|
|
148
|
+
.filter((d): d is string => d !== undefined && d !== '')
|
|
149
|
+
.join(', ');
|
|
150
|
+
const line = details === '' ? name : `${name} (${details})`;
|
|
151
|
+
return line.replace(/[\r\n]/g, (c) => (c === '\r' ? '\\r' : '\\n'));
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/** A header value with its line breaks made visible instead of obeyed. */
|
|
155
|
+
function oneLine(value: string): string {
|
|
156
|
+
return value.replace(/[\r\n]/g, (c) => (c === '\r' ? '\\r' : '\\n'));
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/** Render the model into the marker summary + extraction text (see module doc). */
|
|
160
|
+
export function emailExtraction(model: EmailModel): ExtractedDoc {
|
|
161
|
+
const header: string[] = [];
|
|
162
|
+
// Absent OR blank fields are omitted — a header marker is never printed empty.
|
|
163
|
+
const pushHeader = (label: string, value: string | undefined): void => {
|
|
164
|
+
if (value === undefined || value.trim() === '') return;
|
|
165
|
+
// A header marker is ONE line. A From or Subject carrying CR/LF would
|
|
166
|
+
// otherwise forge further lines into the extraction — including lines that
|
|
167
|
+
// read like other markers — so the breaks are shown as escapes.
|
|
168
|
+
header.push(`[${label}] ${oneLine(value)}`);
|
|
169
|
+
};
|
|
170
|
+
pushHeader('from', model.from);
|
|
171
|
+
pushHeader('to', model.to);
|
|
172
|
+
pushHeader('cc', model.cc);
|
|
173
|
+
pushHeader('bcc', model.bcc);
|
|
174
|
+
pushHeader('subject', model.subject);
|
|
175
|
+
pushHeader('date', model.date);
|
|
176
|
+
|
|
177
|
+
const sections: string[] = [];
|
|
178
|
+
if (header.length > 0) sections.push(header.join('\n'));
|
|
179
|
+
if (model.bodySource === 'rtf-only') sections.push(RTF_ONLY_BODY_LINE);
|
|
180
|
+
else if (model.body !== '') sections.push(model.body);
|
|
181
|
+
if (model.attachments.length > 0) {
|
|
182
|
+
sections.push(`[attachments]\n${model.attachments.map(attachmentLine).join('\n')}`);
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
const n = model.attachments.length;
|
|
186
|
+
const parts = ['email message'];
|
|
187
|
+
if (n > 0) parts.push(`${n} attachment${n === 1 ? '' : 's'} listed (names only; not extracted)`);
|
|
188
|
+
if (model.bodySource === 'html') parts.push('HTML body rendered as plain text');
|
|
189
|
+
if (model.bodySource === 'rtf-only') parts.push('body is RTF; no plain-text part');
|
|
190
|
+
parts.push('formatting and full headers omitted');
|
|
191
|
+
|
|
192
|
+
return { summary: parts.join('; '), text: sections.join('\n\n') };
|
|
193
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import AdmZip from 'adm-zip';
|
|
2
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
3
|
+
import { localBlocks, localElementBlocks, localName, paragraphRunText, zipEntryOversize } from './ooxml-text.js';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Extract the BODY text of a `.docx` (Word) document.
|
|
7
|
+
*
|
|
8
|
+
* A docx is a zip whose main part is `word/document.xml`. v1 extracts the body
|
|
9
|
+
* only — headers/footers are skipped, and the marker summary says so.
|
|
10
|
+
*
|
|
11
|
+
* - Paragraphs become lines. `<w:t>` runs are concatenated with NO separator
|
|
12
|
+
* (Word splits runs mid-word on formatting boundaries).
|
|
13
|
+
* - Tables become lines with cell text tab-separated, one line per row.
|
|
14
|
+
*/
|
|
15
|
+
export function extractDocx(bytes: Buffer): ExtractResult {
|
|
16
|
+
let xml: string;
|
|
17
|
+
try {
|
|
18
|
+
const zip = new AdmZip(bytes);
|
|
19
|
+
const entry = zip.getEntry('word/document.xml');
|
|
20
|
+
if (!entry) {
|
|
21
|
+
return { ok: false, message: 'could not be parsed as a .docx (no word/document.xml inside the archive)' };
|
|
22
|
+
}
|
|
23
|
+
// Declared-uncompressed-size bound BEFORE inflation — see zipEntryOversize.
|
|
24
|
+
const oversize = zipEntryOversize(entry);
|
|
25
|
+
if (oversize) return { ok: false, message: `could not be extracted as a .docx (${oversize})` };
|
|
26
|
+
xml = entry.getData().toString('utf8');
|
|
27
|
+
} catch (err) {
|
|
28
|
+
return { ok: false, message: `could not be parsed as a .docx (${(err as Error).message})` };
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// The BODY element read by the parser rather than matched by a regex: a
|
|
32
|
+
// comment or CDATA section mentioning `<w:body>` could answer as the
|
|
33
|
+
// document body and hand back its text instead of the real one.
|
|
34
|
+
const body = localBlocks(xml, 'body')[0] ?? xml;
|
|
35
|
+
|
|
36
|
+
const lines: string[] = [];
|
|
37
|
+
let paragraphs = 0;
|
|
38
|
+
let tables = 0;
|
|
39
|
+
// Tables and paragraphs in DOCUMENT order, straight from the parser: a match
|
|
40
|
+
// is never descended into, so a table's own paragraphs stay inside it and are
|
|
41
|
+
// rendered once, as its rows. The hand-rolled splitter this replaces counted
|
|
42
|
+
// `<w:tbl>` opens and closes with a regex, which a comment or CDATA section
|
|
43
|
+
// mentioning either could throw off — splitting a paragraph in half and
|
|
44
|
+
// dropping its text.
|
|
45
|
+
for (const block of localElementBlocks(body, ['tbl', 'p'])) {
|
|
46
|
+
if (localName(block.name) === 'tbl') {
|
|
47
|
+
tables++;
|
|
48
|
+
for (const row of localBlocks(block.body ?? '', 'tr')) {
|
|
49
|
+
const cells = localBlocks(row, 'tc').map((cell) => paragraphRunText(cell, 't'));
|
|
50
|
+
lines.push(cells.join(' '));
|
|
51
|
+
}
|
|
52
|
+
} else {
|
|
53
|
+
lines.push(paragraphRunText(block.body ?? '', 't'));
|
|
54
|
+
paragraphs++;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
while (lines.length > 0 && lines[lines.length - 1].trim() === '') lines.pop();
|
|
58
|
+
|
|
59
|
+
const parts = [`${paragraphs} paragraph${paragraphs === 1 ? '' : 's'}`];
|
|
60
|
+
if (tables > 0) parts.push(`${tables} table${tables === 1 ? '' : 's'}`);
|
|
61
|
+
return {
|
|
62
|
+
ok: true,
|
|
63
|
+
summary: `${parts.join(' + ')}; body only (headers/footers skipped); layout, images and formatting omitted`,
|
|
64
|
+
text: lines.join('\n'),
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
import PostalMime, { type Address, type Email } from 'postal-mime';
|
|
2
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
3
|
+
import { emailExtraction, htmlToEmailText, type EmailAttachment, type EmailModel } from './email-text.js';
|
|
4
|
+
import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Extract a `.eml` (RFC 822 / MIME) email into the shared email text shape
|
|
8
|
+
* (see `email-text.ts`).
|
|
9
|
+
*
|
|
10
|
+
* Parsing is `postal-mime` (postalsys, MIT-0): small, dependency-free, ESM,
|
|
11
|
+
* and the same code runs in the browser — which keeps the frontend viewer's
|
|
12
|
+
* story identical to the agent's. It normalizes encoded-word headers and
|
|
13
|
+
* multipart bodies, and hands the `Date:` header over as ISO when it parses.
|
|
14
|
+
*
|
|
15
|
+
* postal-mime is LENIENT — random bytes "parse" to an empty message rather
|
|
16
|
+
* than throwing — so recognizability is checked here: a file yielding NO
|
|
17
|
+
* email headers at all (no From/To/Cc/Bcc/Subject/Date/Message-ID) and no
|
|
18
|
+
* attachments is not an email, and gets the typed could-not-be-parsed
|
|
19
|
+
* failure instead of an empty extraction.
|
|
20
|
+
*/
|
|
21
|
+
export async function extractEml(bytes: Buffer): Promise<ExtractResult> {
|
|
22
|
+
// The same raw-size bound the PDF extractor applies: no container to
|
|
23
|
+
// pre-scan, so the bound is the file's size, checked before parsing.
|
|
24
|
+
if (bytes.length > MAX_DOC_PART_BYTES) {
|
|
25
|
+
return {
|
|
26
|
+
ok: false,
|
|
27
|
+
message: `could not be extracted as a .eml (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
|
|
28
|
+
};
|
|
29
|
+
}
|
|
30
|
+
let email: Email;
|
|
31
|
+
try {
|
|
32
|
+
email = await PostalMime.parse(bytes);
|
|
33
|
+
} catch (err) {
|
|
34
|
+
return { ok: false, message: `could not be parsed as a .eml (${(err as Error).message})` };
|
|
35
|
+
}
|
|
36
|
+
if (
|
|
37
|
+
email.from === undefined &&
|
|
38
|
+
email.to === undefined &&
|
|
39
|
+
email.cc === undefined &&
|
|
40
|
+
email.bcc === undefined &&
|
|
41
|
+
email.subject === undefined &&
|
|
42
|
+
email.date === undefined &&
|
|
43
|
+
email.messageId === undefined &&
|
|
44
|
+
email.attachments.length === 0
|
|
45
|
+
) {
|
|
46
|
+
return { ok: false, message: 'could not be parsed as a .eml (no email headers found)' };
|
|
47
|
+
}
|
|
48
|
+
return { ok: true, ...emailExtraction(emlModel(email)) };
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** postal-mime's parse, shaped into the format-independent email model. */
|
|
52
|
+
function emlModel(email: Email): EmailModel {
|
|
53
|
+
const text = email.text !== undefined && email.text.trim() !== '' ? email.text : undefined;
|
|
54
|
+
const html = email.html !== undefined && email.html.trim() !== '' ? email.html : undefined;
|
|
55
|
+
const body = text ?? (html !== undefined ? htmlToEmailText(html) : '');
|
|
56
|
+
return {
|
|
57
|
+
from: email.from && addressListText([email.from]),
|
|
58
|
+
to: email.to && addressListText(email.to),
|
|
59
|
+
cc: email.cc && addressListText(email.cc),
|
|
60
|
+
bcc: email.bcc && addressListText(email.bcc),
|
|
61
|
+
subject: email.subject,
|
|
62
|
+
date: email.date !== undefined ? isoDate(email.date) : undefined,
|
|
63
|
+
body: body.replace(/\s+$/, ''),
|
|
64
|
+
bodySource: text !== undefined ? 'text' : html !== undefined ? 'html' : 'none',
|
|
65
|
+
attachments: email.attachments.map(
|
|
66
|
+
(a): EmailAttachment => ({
|
|
67
|
+
name: a.filename ?? 'unnamed attachment',
|
|
68
|
+
mimeType: a.mimeType,
|
|
69
|
+
sizeBytes: typeof a.content === 'string' ? Buffer.byteLength(a.content) : a.content.byteLength,
|
|
70
|
+
}),
|
|
71
|
+
),
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** `Name <addr>, addr2, Group: member, member` — groups flattened inline. */
|
|
76
|
+
function addressListText(list: Address[]): string | undefined {
|
|
77
|
+
const s = list
|
|
78
|
+
.map(function one(a: Address): string {
|
|
79
|
+
if (a.group !== undefined) return `${a.name}: ${a.group.map(one).join(', ')}`;
|
|
80
|
+
if (a.address === undefined || a.address === '') return a.name;
|
|
81
|
+
return a.name !== '' && a.name !== a.address ? `${a.name} <${a.address}>` : a.address;
|
|
82
|
+
})
|
|
83
|
+
.filter((t) => t !== '')
|
|
84
|
+
.join(', ');
|
|
85
|
+
return s === '' ? undefined : s;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** postal-mime already normalizes parseable dates to ISO; keep the raw value when it could not. */
|
|
89
|
+
function isoDate(value: string): string {
|
|
90
|
+
const d = new Date(value);
|
|
91
|
+
return Number.isNaN(d.getTime()) ? value : d.toISOString();
|
|
92
|
+
}
|