@bevel-software/platform-core-backend 0.11.2 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. package/THIRD-PARTY-NOTICES.md +1165 -427
  2. package/dist/core/create-core-server.js +1 -1
  3. package/dist/core/create-core-server.js.map +1 -1
  4. package/dist/core/create-core-services.d.ts +2 -0
  5. package/dist/core/create-core-services.d.ts.map +1 -1
  6. package/dist/core/create-core-services.js +5 -0
  7. package/dist/core/create-core-services.js.map +1 -1
  8. package/dist/core-config.d.ts +7 -0
  9. package/dist/core-config.d.ts.map +1 -1
  10. package/dist/core-config.js +9 -0
  11. package/dist/core-config.js.map +1 -1
  12. package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
  13. package/dist/modules/code-mode/code-mode.tool.js +7 -1
  14. package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
  15. package/dist/modules/kb-fs/clone-config.d.ts +40 -2
  16. package/dist/modules/kb-fs/clone-config.d.ts.map +1 -1
  17. package/dist/modules/kb-fs/clone-config.js +94 -2
  18. package/dist/modules/kb-fs/clone-config.js.map +1 -1
  19. package/dist/modules/workflow/workflow.service.d.ts +38 -0
  20. package/dist/modules/workflow/workflow.service.d.ts.map +1 -1
  21. package/dist/modules/workflow/workflow.service.js +112 -6
  22. package/dist/modules/workflow/workflow.service.js.map +1 -1
  23. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
  24. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
  25. package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
  26. package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
  27. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
  28. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
  29. package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
  30. package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
  31. package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
  32. package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
  33. package/dist/modules/workspace/file-readers/document-reader.js +59 -0
  34. package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
  35. package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
  36. package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
  37. package/dist/modules/workspace/file-readers/email-reader.js +19 -0
  38. package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
  39. package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
  40. package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
  41. package/dist/modules/workspace/file-readers/email-text.js +151 -0
  42. package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
  43. package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
  44. package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
  45. package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
  46. package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
  47. package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
  48. package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
  49. package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
  50. package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
  51. package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
  52. package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
  53. package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
  54. package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
  55. package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
  56. package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
  57. package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
  58. package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
  59. package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
  60. package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
  61. package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
  62. package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
  63. package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
  64. package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
  65. package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
  66. package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
  67. package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
  68. package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
  69. package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
  70. package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
  71. package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
  72. package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
  73. package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
  74. package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
  75. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
  76. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
  77. package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
  78. package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
  79. package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
  80. package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
  81. package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
  82. package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
  83. package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
  84. package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
  85. package/dist/modules/workspace/file-readers/file-reader.js +55 -0
  86. package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
  87. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
  88. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
  89. package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
  90. package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
  91. package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
  92. package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
  93. package/dist/modules/workspace/file-readers/image-read.js +108 -0
  94. package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
  95. package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
  96. package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
  97. package/dist/modules/workspace/file-readers/image-reader.js +30 -0
  98. package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
  99. package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
  100. package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
  101. package/dist/modules/workspace/file-readers/odf-text.js +116 -0
  102. package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
  103. package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
  104. package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
  105. package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
  106. package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
  107. package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
  108. package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
  109. package/dist/modules/workspace/file-readers/text-reader.js +117 -0
  110. package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
  111. package/dist/modules/workspace/startup/kb-git.d.ts.map +1 -1
  112. package/dist/modules/workspace/startup/kb-git.js +21 -4
  113. package/dist/modules/workspace/startup/kb-git.js.map +1 -1
  114. package/dist/modules/workspace/workspace.service.d.ts +52 -8
  115. package/dist/modules/workspace/workspace.service.d.ts.map +1 -1
  116. package/dist/modules/workspace/workspace.service.js +121 -23
  117. package/dist/modules/workspace/workspace.service.js.map +1 -1
  118. package/dist/modules/workspace/workspace.tools.d.ts +2 -1
  119. package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
  120. package/dist/modules/workspace/workspace.tools.js +158 -15
  121. package/dist/modules/workspace/workspace.tools.js.map +1 -1
  122. package/package.json +11 -6
  123. package/src/core/create-core-server.ts +1 -1
  124. package/src/core/create-core-services.ts +6 -0
  125. package/src/core-config.ts +9 -0
  126. package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
  127. package/src/modules/code-mode/code-mode.tool.ts +7 -1
  128. package/src/modules/kb-fs/__tests__/clone-config.test.ts +63 -2
  129. package/src/modules/kb-fs/clone-config.ts +97 -2
  130. package/src/modules/secrets-vault/secrets-vault.routes.ts +582 -582
  131. package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
  132. package/src/modules/workflow/__tests__/workflow.service.commitFileWhileLocked.test.ts +11 -5
  133. package/src/modules/workflow/__tests__/workflow.service.releaseLock.test.ts +172 -7
  134. package/src/modules/workflow/workflow.service.ts +118 -6
  135. package/src/modules/workspace/__tests__/workspace.service.test.ts +1 -1
  136. package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
  137. package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
  138. package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
  139. package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
  140. package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
  141. package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
  142. package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
  143. package/src/modules/workspace/file-readers/document-reader.ts +64 -0
  144. package/src/modules/workspace/file-readers/email-reader.ts +21 -0
  145. package/src/modules/workspace/file-readers/email-text.ts +193 -0
  146. package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
  147. package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
  148. package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
  149. package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
  150. package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
  151. package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
  152. package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
  153. package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
  154. package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
  155. package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
  156. package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
  157. package/src/modules/workspace/file-readers/file-reader.ts +104 -0
  158. package/src/modules/workspace/file-readers/image-read.ts +122 -0
  159. package/src/modules/workspace/file-readers/image-reader.ts +39 -0
  160. package/src/modules/workspace/file-readers/odf-text.ts +123 -0
  161. package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
  162. package/src/modules/workspace/file-readers/text-reader.ts +131 -0
  163. package/src/modules/workspace/startup/__tests__/kb-startup-runner.test.ts +141 -0
  164. package/src/modules/workspace/startup/kb-git.ts +20 -7
  165. package/src/modules/workspace/workspace.service.ts +132 -25
  166. package/src/modules/workspace/workspace.tools.ts +174 -12
@@ -0,0 +1,485 @@
1
+ import * as XLSX from 'xlsx';
2
+ import { describe, expect, it } from 'vitest';
3
+ import { RTF_ONLY_BODY_LINE } from '../email-text.js';
4
+ import { extractEml } from '../extract-eml.js';
5
+ import { extractMsg } from '../extract-msg.js';
6
+ import { MAX_DOC_PART_BYTES } from '../ooxml-text.js';
7
+
8
+ /**
9
+ * REAL fixtures, no mocks — the same stance as doc-extract.test.ts:
10
+ *
11
+ * - `.eml` fixtures are hand-written MIME, byte for byte.
12
+ * - `.msg` fixtures are REAL CFB containers built in-test with SheetJS's CFB
13
+ * writer (`XLSX.CFB`), carrying the actual MAPI streams
14
+ * (`__substg1.0_<tag><type>`, `__recip_…`, `__attach_…`,
15
+ * `__properties_version1.0`) an Outlook save produces — msgreader parses
16
+ * them exactly as it parses Outlook's output. (A smoke-check against a
17
+ * file saved by a real Outlook remains worthwhile; the container layout is
18
+ * identical but Outlook stamps many more properties.)
19
+ */
20
+
21
+ // ── .eml fixture builders ──────────────────────────────────────────────────
22
+
23
+ const CRLF = '\r\n';
24
+
25
+ function emlBytes(lines: string[]): Buffer {
26
+ return Buffer.from(lines.join(CRLF), 'utf8');
27
+ }
28
+
29
+ const MIXED_EML = emlBytes([
30
+ 'From: Ada Lovelace <ada@example.com>',
31
+ 'To: Bob <bob@example.com>, carol@example.com',
32
+ 'Cc: Dan <dan@example.com>',
33
+ 'Subject: Quarterly numbers',
34
+ 'Date: Mon, 5 Jan 2026 10:00:00 +0000',
35
+ 'Message-ID: <m1@example.com>',
36
+ 'MIME-Version: 1.0',
37
+ 'Content-Type: multipart/mixed; boundary="b1"',
38
+ '',
39
+ '--b1',
40
+ 'Content-Type: multipart/alternative; boundary="b2"',
41
+ '',
42
+ '--b2',
43
+ 'Content-Type: text/plain; charset=utf-8',
44
+ '',
45
+ 'Please see attached.',
46
+ 'Second line.',
47
+ '--b2',
48
+ 'Content-Type: text/html; charset=utf-8',
49
+ '',
50
+ '<p>Please see <b>attached</b>.</p><p>Second line.</p>',
51
+ '--b2--',
52
+ '--b1',
53
+ 'Content-Type: application/pdf; name="report.pdf"',
54
+ 'Content-Disposition: attachment; filename="report.pdf"',
55
+ 'Content-Transfer-Encoding: base64',
56
+ '',
57
+ 'JVBERi0xLjQ=', // "%PDF-1.4" — 8 bytes decoded
58
+ '--b1',
59
+ 'Content-Type: text/csv; name="data.csv"',
60
+ 'Content-Disposition: attachment; filename="data.csv"',
61
+ '',
62
+ 'a,b',
63
+ '--b1--',
64
+ '',
65
+ ]);
66
+
67
+ // ── .msg fixture builders (real CFB bytes via SheetJS) ─────────────────────
68
+
69
+ const utf16 = (s: string): Buffer => Buffer.from(s, 'utf16le');
70
+
71
+ /**
72
+ * A MAPI `__properties_version1.0` stream: `headerLen` zero bytes, then
73
+ * 16-byte records — tag (u32 LE: id<<16 | type), flags, 8 value bytes.
74
+ */
75
+ function propertiesStream(headerLen: number, records: Array<[number, Buffer]>): Buffer {
76
+ const rows = records.map(([tag, value]) => {
77
+ const b = Buffer.alloc(16);
78
+ b.writeUInt32LE(tag >>> 0, 0);
79
+ value.copy(b, 8);
80
+ return b;
81
+ });
82
+ return Buffer.concat([Buffer.alloc(headerLen), ...rows]);
83
+ }
84
+
85
+ const longValue = (n: number): Buffer => {
86
+ const b = Buffer.alloc(8);
87
+ b.writeUInt32LE(n, 0);
88
+ return b;
89
+ };
90
+
91
+ /** PT_SYSTIME value: FILETIME (100ns ticks since 1601) for the given ISO instant. */
92
+ function filetimeValue(iso: string): Buffer {
93
+ const b = Buffer.alloc(8);
94
+ b.writeBigUInt64LE((BigInt(Date.parse(iso)) + 11644473600000n) * 10000n);
95
+ return b;
96
+ }
97
+
98
+ interface MsgSpec {
99
+ strings?: Record<string, string>; // '0037' → subject, unicode substg streams
100
+ recipients?: Array<{ name?: string; email?: string; type?: number }>; // 1 to / 2 cc / 3 bcc
101
+ attachments?: Array<{ name?: string; mime?: string; content?: Buffer }>;
102
+ rtf?: boolean;
103
+ submitTimeIso?: string;
104
+ }
105
+
106
+ /** REAL `.msg` bytes: a CFB container with the MAPI streams Outlook writes. */
107
+ function msgBytes(spec: MsgSpec): Buffer {
108
+ const cfb = XLSX.CFB.utils.cfb_new();
109
+ const add = (path: string, bytes: Buffer): void => {
110
+ XLSX.CFB.utils.cfb_add(cfb, path, bytes);
111
+ };
112
+ for (const [id, value] of Object.entries(spec.strings ?? {})) add(`/__substg1.0_${id}001F`, utf16(value));
113
+ (spec.recipients ?? []).forEach((r, i) => {
114
+ const dir = `/__recip_version1.0_#${i.toString(16).toUpperCase().padStart(8, '0')}`;
115
+ if (r.name !== undefined) add(`${dir}/__substg1.0_3001001F`, utf16(r.name));
116
+ if (r.email !== undefined) add(`${dir}/__substg1.0_3003001F`, utf16(r.email));
117
+ if (r.type !== undefined) add(`${dir}/__properties_version1.0`, propertiesStream(8, [[0x0c150003, longValue(r.type)]]));
118
+ });
119
+ (spec.attachments ?? []).forEach((a, i) => {
120
+ const dir = `/__attach_version1.0_#${i.toString(16).toUpperCase().padStart(8, '0')}`;
121
+ if (a.name !== undefined) add(`${dir}/__substg1.0_3707001F`, utf16(a.name));
122
+ if (a.mime !== undefined) add(`${dir}/__substg1.0_370E001F`, utf16(a.mime));
123
+ if (a.content !== undefined) add(`${dir}/__substg1.0_37010102`, a.content);
124
+ });
125
+ if (spec.rtf) add('/__substg1.0_10090102', Buffer.from([1, 2, 3, 4]));
126
+ if (spec.submitTimeIso !== undefined) {
127
+ add('/__properties_version1.0', propertiesStream(32, [[0x00390040, filetimeValue(spec.submitTimeIso)]]));
128
+ }
129
+ return XLSX.CFB.write(cfb, { type: 'buffer' }) as Buffer;
130
+ }
131
+
132
+ // ── .eml extraction ────────────────────────────────────────────────────────
133
+
134
+ describe('extractEml', () => {
135
+ it('extracts the header block, prefers the plain-text part, and lists attachments with type and size', async () => {
136
+ const res = await extractEml(MIXED_EML);
137
+ expect(res.ok).toBe(true);
138
+ if (!res.ok) return;
139
+ expect(res.text).toBe(
140
+ [
141
+ '[from] Ada Lovelace <ada@example.com>',
142
+ '[to] Bob <bob@example.com>, carol@example.com',
143
+ '[cc] Dan <dan@example.com>',
144
+ '[subject] Quarterly numbers',
145
+ '[date] 2026-01-05T10:00:00.000Z',
146
+ '',
147
+ // The text/plain alternative, NOT the stripped HTML.
148
+ 'Please see attached.',
149
+ 'Second line.',
150
+ '',
151
+ '[attachments]',
152
+ 'report.pdf (application/pdf, 8 bytes)',
153
+ 'data.csv (text/csv, 4 bytes)', // 'a,b' + the newline before the closing boundary
154
+ ].join('\n'),
155
+ );
156
+ expect(res.summary).toBe(
157
+ 'email message; 2 attachments listed (names only; not extracted); formatting and full headers omitted',
158
+ );
159
+ });
160
+
161
+ it('a text-only email has no [attachments] section and omits absent header lines (no cc)', async () => {
162
+ const res = await extractEml(
163
+ emlBytes([
164
+ 'From: ada@example.com',
165
+ 'To: bob@example.com',
166
+ 'Subject: ping',
167
+ 'Date: Mon, 5 Jan 2026 10:00:00 +0000',
168
+ 'Content-Type: text/plain; charset=utf-8',
169
+ '',
170
+ 'Just checking in.',
171
+ ]),
172
+ );
173
+ expect(res.ok).toBe(true);
174
+ if (!res.ok) return;
175
+ expect(res.text).toBe(
176
+ [
177
+ '[from] ada@example.com',
178
+ '[to] bob@example.com',
179
+ '[subject] ping',
180
+ '[date] 2026-01-05T10:00:00.000Z',
181
+ '',
182
+ 'Just checking in.',
183
+ ].join('\n'),
184
+ );
185
+ expect(res.text).not.toContain('[cc]');
186
+ expect(res.text).not.toContain('[attachments]');
187
+ expect(res.summary).toBe('email message; formatting and full headers omitted');
188
+ });
189
+
190
+ it('an HTML-only email is stripped to text — block tags become line breaks, entities decode — and the summary says so', async () => {
191
+ const res = await extractEml(
192
+ emlBytes([
193
+ 'From: ada@example.com',
194
+ 'Subject: styled',
195
+ 'Content-Type: text/html; charset=utf-8',
196
+ '',
197
+ '<html><head><style>p{color:red}</style></head><body>' +
198
+ '<h1>Title</h1><p>Para&nbsp;one &amp; two</p><div>second<br>third</div></body></html>',
199
+ ]),
200
+ );
201
+ expect(res.ok).toBe(true);
202
+ if (!res.ok) return;
203
+ // Blocks separate as paragraphs (blank line); <br> is a plain line break.
204
+ expect(res.text).toContain('Title\n\nPara one & two\n\nsecond\nthird');
205
+ expect(res.text).not.toContain('color:red');
206
+ expect(res.text).not.toContain('<p>');
207
+ expect(res.summary).toContain('HTML body rendered as plain text');
208
+ });
209
+
210
+ it('omits every header line the message lacks — a body-only draft with just a Message-ID still extracts', async () => {
211
+ const res = await extractEml(emlBytes(['Message-ID: <draft-1@local>', '', 'only a body']));
212
+ expect(res.ok).toBe(true);
213
+ if (!res.ok) return;
214
+ expect(res.text).toBe('only a body');
215
+ });
216
+
217
+ it('lists an attachment whose filename is whitespace-only as "unnamed attachment"', async () => {
218
+ // postal-mime maps a MISSING filename to null (defaulted at the model),
219
+ // but `filename=" "` comes through verbatim — unfixed, the attachment
220
+ // line printed the whitespace and read as blank.
221
+ const res = await extractEml(
222
+ emlBytes([
223
+ 'From: ada@example.com',
224
+ 'Subject: nameless',
225
+ 'MIME-Version: 1.0',
226
+ 'Content-Type: multipart/mixed; boundary="b1"',
227
+ '',
228
+ '--b1',
229
+ 'Content-Type: text/plain; charset=utf-8',
230
+ '',
231
+ 'see attached',
232
+ '--b1',
233
+ 'Content-Type: text/csv',
234
+ 'Content-Disposition: attachment; filename=" "',
235
+ '',
236
+ 'a,b',
237
+ '--b1--',
238
+ '',
239
+ ]),
240
+ );
241
+ expect(res.ok).toBe(true);
242
+ if (!res.ok) return;
243
+ expect(res.text).toContain('[attachments]\nunnamed attachment (text/csv');
244
+ });
245
+
246
+ it('answers bytes with no email headers at all with the typed could-not-be-parsed failure', async () => {
247
+ const res = await extractEml(Buffer.from([0xd0, 0xcf, 0x11, 0xe0, 0x01, 0x02, 0x00, 0xff, 0xfe]));
248
+ expect(res).toEqual({ ok: false, message: 'could not be parsed as a .eml (no email headers found)' });
249
+ });
250
+
251
+ it('a valid email whose ONLY header is Bcc: is recognized, not rejected as header-less', async () => {
252
+ const res = await extractEml(emlBytes(['Bcc: Eve <eve@example.com>', '', 'quiet copy']));
253
+ expect(res.ok).toBe(true);
254
+ if (!res.ok) return;
255
+ expect(res.text).toBe('[bcc] Eve <eve@example.com>\n\nquiet copy');
256
+ });
257
+
258
+ it('refuses a file over the 50 MB extraction bound before parsing anything', async () => {
259
+ const res = await extractEml(Buffer.alloc(MAX_DOC_PART_BYTES + 1));
260
+ expect(res.ok).toBe(false);
261
+ if (res.ok) return;
262
+ expect(res.message).toContain('could not be extracted as a .eml');
263
+ expect(res.message).toContain('50 MB');
264
+ });
265
+ });
266
+
267
+ // ── .msg extraction ────────────────────────────────────────────────────────
268
+
269
+ describe('extractMsg', () => {
270
+ it('extracts headers (sender, typed recipients, ISO date), body and attachments from real CFB bytes', () => {
271
+ const res = extractMsg(
272
+ msgBytes({
273
+ strings: {
274
+ '0037': 'Quarterly numbers',
275
+ '0C1A': 'Ada Lovelace',
276
+ '0C1F': 'ada@example.com',
277
+ '1000': 'Please see attached.',
278
+ },
279
+ recipients: [
280
+ { name: 'Bob', email: 'bob@example.com', type: 1 },
281
+ { name: 'Dan', email: 'dan@example.com', type: 2 },
282
+ ],
283
+ attachments: [{ name: 'report.pdf', mime: 'application/pdf', content: Buffer.from('PDFDATA') }],
284
+ submitTimeIso: '2026-01-05T10:00:00Z',
285
+ }),
286
+ );
287
+ expect(res.ok).toBe(true);
288
+ if (!res.ok) return;
289
+ expect(res.text).toBe(
290
+ [
291
+ '[from] Ada Lovelace <ada@example.com>',
292
+ '[to] Bob <bob@example.com>',
293
+ '[cc] Dan <dan@example.com>',
294
+ '[subject] Quarterly numbers',
295
+ '[date] 2026-01-05T10:00:00.000Z',
296
+ '',
297
+ 'Please see attached.',
298
+ '',
299
+ '[attachments]',
300
+ 'report.pdf (application/pdf, 7 bytes)',
301
+ ].join('\n'),
302
+ );
303
+ expect(res.summary).toBe(
304
+ 'email message; 1 attachment listed (names only; not extracted); formatting and full headers omitted',
305
+ );
306
+ });
307
+
308
+ it('strips an HTML-only body to text, like .eml', () => {
309
+ const res = extractMsg(
310
+ msgBytes({ strings: { '0037': 'styled', '1013': '<p>Hello <b>there</b>&nbsp;!</p><div>bye</div>' } }),
311
+ );
312
+ expect(res.ok).toBe(true);
313
+ if (!res.ok) return;
314
+ expect(res.text).toContain('Hello there !\n\nbye');
315
+ expect(res.summary).toContain('HTML body rendered as plain text');
316
+ });
317
+
318
+ it('degrades an RTF-only body honestly: the text says so instead of pretending to decode RTF', () => {
319
+ const res = extractMsg(msgBytes({ strings: { '0037': 'compressed' }, rtf: true }));
320
+ expect(res.ok).toBe(true);
321
+ if (!res.ok) return;
322
+ expect(res.text).toBe(['[subject] compressed', '', RTF_ONLY_BODY_LINE].join('\n'));
323
+ expect(res.summary).toContain('body is RTF; no plain-text part');
324
+ });
325
+
326
+ it('keeps a recipient whose PidTagRecipientType carries flag bits — msgreader leaks the raw number', () => {
327
+ // msgreader maps only the bare MAPI values 1/2/3 to 'to'/'cc'/'bcc'; a
328
+ // resubmit-flagged value like MAPI_TO | MAPI_P1 (0x10000001) comes through
329
+ // as a NUMBER, and a string-only comparison silently dropped the line.
330
+ const res = extractMsg(
331
+ msgBytes({
332
+ strings: { '0037': 'resubmitted' },
333
+ recipients: [
334
+ { name: 'Bob', email: 'bob@example.com', type: 0x10000001 },
335
+ { name: 'Dan', email: 'dan@example.com', type: 0x10000002 },
336
+ { name: 'Eve', email: 'eve@example.com', type: 0x10000003 },
337
+ ],
338
+ }),
339
+ );
340
+ expect(res.ok).toBe(true);
341
+ if (!res.ok) return;
342
+ expect(res.text).toContain('[to] Bob <bob@example.com>');
343
+ expect(res.text).toContain('[cc] Dan <dan@example.com>');
344
+ expect(res.text).toContain('[bcc] Eve <eve@example.com>');
345
+ });
346
+
347
+ it('lists an attachment whose MAPI display name stream is EMPTY as "unnamed attachment"', () => {
348
+ // msgreader hands an empty `__substg1.0_3707001F` stream through as '',
349
+ // which the `?? 'unnamed attachment'` default at the model never catches.
350
+ const res = extractMsg(
351
+ msgBytes({
352
+ strings: { '0037': 'nameless' },
353
+ attachments: [{ name: '', mime: 'application/pdf', content: Buffer.from('DATA') }],
354
+ }),
355
+ );
356
+ expect(res.ok).toBe(true);
357
+ if (!res.ok) return;
358
+ expect(res.text).toContain('[attachments]\nunnamed attachment (application/pdf, 4 bytes)');
359
+ });
360
+
361
+ it('answers garbage bytes with the typed could-not-be-parsed failure (msgreader reports, never throws)', () => {
362
+ const res = extractMsg(Buffer.from('total garbage, not a CFB container'));
363
+ expect(res.ok).toBe(false);
364
+ if (res.ok) return;
365
+ expect(res.message).toContain('could not be parsed as a .msg');
366
+ });
367
+
368
+ it('refuses a file over the 50 MB extraction bound before parsing anything', () => {
369
+ const res = extractMsg(Buffer.alloc(MAX_DOC_PART_BYTES + 1));
370
+ expect(res.ok).toBe(false);
371
+ if (res.ok) return;
372
+ expect(res.message).toContain('could not be extracted as a .msg');
373
+ expect(res.message).toContain('50 MB');
374
+ });
375
+ });
376
+
377
+ // ── the shared HTML strip ──────────────────────────────────────────────────
378
+
379
+ describe('an HTML body, stripped to text', () => {
380
+ /** The body of an HTML-only email, as the extraction renders it. */
381
+ const bodyOf = async (html: string): Promise<string> => {
382
+ const res = await extractEml(
383
+ emlBytes([
384
+ 'From: ada@example.com',
385
+ 'Content-Type: text/html; charset=utf-8',
386
+ '',
387
+ html,
388
+ ]),
389
+ );
390
+ if (!res.ok) throw new Error(res.message);
391
+ const marker = '[from] ada@example.com';
392
+ const at = res.text.indexOf(marker);
393
+ return res.text.slice(at + marker.length).replace(/^\n+/, '');
394
+ };
395
+
396
+ it('drops style/script/head whole and keeps paragraph structure', async () => {
397
+ expect(await bodyOf('<html><head><style>b{color:red}</style><title>T</title></head><body><p>one</p><p>two</p></body></html>'),
398
+ ).toBe('one\n\ntwo');
399
+ expect(await bodyOf('<SCRIPT TYPE="text/javascript">evil()</SCRIPT>after')).toBe('after');
400
+ expect(await bodyOf('<Style Media="print">b{}</STYLE>rest')).toBe('rest');
401
+ });
402
+
403
+ it('never truncates a tag at a > inside a quoted attribute value', async () => {
404
+ expect(await bodyOf('<a title="a > b">x</a>')).toBe('x');
405
+ expect(await bodyOf("<p align='x>y'>para</p>")).toBe('para');
406
+ expect(await bodyOf('<div class="a>b">block</div>')).toBe('block');
407
+ expect(await bodyOf('<style media="x>y">b{}</style>rest')).toBe('rest');
408
+ });
409
+
410
+ it('turns br and block-tag boundaries into line breaks', async () => {
411
+ expect(await bodyOf('a<br>b<br/>c<BR />d')).toBe('a\nb\nc\nd');
412
+ expect(await bodyOf('<h1>T</h1><div>d</div><li>i</li>')).toBe('T\n\nd\n\ni');
413
+ });
414
+
415
+ it('decodes entities, and a &nbsp; lands as an ordinary space', async () => {
416
+ // U+00A0 looks like a space and is not one — an agent grepping the
417
+ // extraction for "one two" must not miss a line that reads exactly that.
418
+ expect(await bodyOf('<p>one&nbsp;two &amp; three &mdash; four</p>')).toBe('one two & three — four');
419
+ });
420
+
421
+ it('keeps a comparison in the body: `2 < 3` is text, not a tag', async () => {
422
+ // `<` followed by whitespace cannot open a tag, so the comparison survives.
423
+ expect(await bodyOf('<p>total</p> 2 < 3')).toBe('total\n2 < 3');
424
+ });
425
+
426
+ it('CHANGED: reads malformed markup the way a mail client would', async () => {
427
+ // These four all used to be answered by a hand-rolled strip with rules of
428
+ // its own. They are now htmlparser2's answers, which are a browser's — the
429
+ // useful standard here, because it is what the recipient of the mail saw.
430
+ //
431
+ // An unterminated tag swallows the rest of the body (the reader saw
432
+ // nothing either), where the strip used to emit it as literal text…
433
+ expect(await bodyOf('<a href="never closed>tail')).toBe('');
434
+ // …an unclosed <script> hides its content instead of showing the source…
435
+ expect(await bodyOf('<script>alert(1)')).toBe('');
436
+ // …a `/` in the tag name is ignored, so `<script/x>` IS a script…
437
+ expect(await bodyOf('<script/x>secret</script>tail')).toBe('tail');
438
+ // …and `</ 2 >` is a bogus end tag, dropped rather than shown.
439
+ expect(await bodyOf('x </ 2 > y')).toBe('x y');
440
+ });
441
+
442
+ it('completes on an adversarial body: 20k < characters and a quote that never closes', async () => {
443
+ // The shape the old quote-aware tag regexes died on — each `<` restarted a
444
+ // lazy expansion that could never pass the unclosed quote, so the strip was
445
+ // quadratic in the body length. The bound is deliberately loose: it exists
446
+ // to catch a return of the blow-up, not to police CI's scheduler.
447
+ const bomb = '<'.repeat(20_000) + '<b title="never closed ' + 'x'.repeat(2_000) + '>';
448
+ const t0 = performance.now();
449
+ const out = await bodyOf(bomb);
450
+ const ms = performance.now() - t0;
451
+ // Every bare `<` is text; the trailing tag is a tag and goes.
452
+ expect(out).toBe('<'.repeat(20_000));
453
+ expect(ms).toBeLessThan(5_000);
454
+ });
455
+
456
+ it('completes on a wall of comparison spans before one far-away `>`', async () => {
457
+ const bomb = '< '.repeat(50_000) + '<p>end</p>';
458
+ const t0 = performance.now();
459
+ const out = await bodyOf(bomb);
460
+ const ms = performance.now() - t0;
461
+ expect(out.endsWith('end')).toBe(true);
462
+ expect(ms).toBeLessThan(5_000);
463
+ });
464
+
465
+ it('completes on a body nested past any sane depth, keeping what it read', async () => {
466
+ // The parser holds a stack entry per open element, so MAX_ELEMENT_DEPTH
467
+ // stops a body that nests as deep as it has bytes.
468
+ const t0 = performance.now();
469
+ const out = await bodyOf('<p>top</p>' + '<div>'.repeat(200_000));
470
+ const ms = performance.now() - t0;
471
+ expect(out.startsWith('top')).toBe(true);
472
+ expect(ms).toBeLessThan(5_000);
473
+ });
474
+
475
+ it('an İ before a container does not shift what gets dropped', async () => {
476
+ // The strip this replaced kept an index-parallel lowercase copy of the
477
+ // body, and `String.toLowerCase` is not length-preserving — `İ` (U+0130)
478
+ // becomes `i` plus a combining dot — so after one of these every later
479
+ // offset was wrong and a `<script>` stopped being recognized. A parser has
480
+ // no such copy to get out of step.
481
+ expect(await bodyOf('İ<script>alert(1)</script>after')).toBe('İafter');
482
+ expect(await bodyOf('İ<p>one</p><p>two</p>')).toBe('İ\none\n\ntwo');
483
+ expect(await bodyOf('<p>aİb</p><style>c{}</style>tail')).toBe('aİb\ntail');
484
+ });
485
+ });
@@ -0,0 +1,97 @@
1
+ import { tmpdir } from 'node:os';
2
+ import { join } from 'node:path';
3
+ import { describe, expect, it } from 'vitest';
4
+ import { DocExtractService } from '../doc-extract.service.js';
5
+ import { DocumentReader } from '../document-reader.js';
6
+ import { EmailReader } from '../email-reader.js';
7
+ import { FileReaderRegistry, type FileReader, type ReadResult } from '../file-reader.js';
8
+ import { createFileReaderRegistry } from '../file-reader.registry.js';
9
+ import { ImageReader } from '../image-reader.js';
10
+ import { LegacyOfficeReader, TextReader } from '../text-reader.js';
11
+
12
+ /**
13
+ * Routing tests for THE file-reader registry: one lookup (`readerFor`) decides
14
+ * how read_file reads, what grep searches, and which files the write tools
15
+ * refuse. Extraction/read behaviour itself is covered by doc-extract.test.ts,
16
+ * image-read.test.ts and workspace.tools.test.ts — this suite pins who OWNS
17
+ * which extension.
18
+ */
19
+
20
+ const registry = createFileReaderRegistry(
21
+ new DocExtractService(join(tmpdir(), 'bevel-test-file-reader-registry')),
22
+ );
23
+
24
+ describe('file-reader registry routing', () => {
25
+ it('routes the seven document extensions to a DocumentReader that is not text-editable', () => {
26
+ for (const p of ['a.docx', 'b.pptx', 'x/y.xlsx', 'r.pdf', 'n.odt', 'd.odp', 'b/s.ods']) {
27
+ const reader = registry.readerFor(p);
28
+ expect(reader, p).toBeInstanceOf(DocumentReader);
29
+ expect(reader.textEditable, p).toBe(false);
30
+ }
31
+ });
32
+
33
+ it('routes the two email extensions to an EmailReader — a DocumentReader with the email write-refusal', () => {
34
+ for (const p of ['Inbox/offer.eml', 'a/b/thread.msg', 'Deals/OFFER.EML', 'old.MSG']) {
35
+ const reader = registry.readerFor(p);
36
+ expect(reader, p).toBeInstanceOf(EmailReader);
37
+ // An EmailReader IS a DocumentReader — grep's cold-extraction budget
38
+ // branch (`instanceof DocumentReader`) must keep covering emails.
39
+ expect(reader, p).toBeInstanceOf(DocumentReader);
40
+ expect(reader.textEditable, p).toBe(false);
41
+ expect(reader.editRefusal?.(p), p).toContain('email file');
42
+ expect(reader.editRefusal?.(p), p).toContain('snapshot');
43
+ expect(reader.editRefusal?.(p), p).toContain('uploading a new version');
44
+ }
45
+ });
46
+
47
+ it('routes case-insensitively (the extension is lowercased before lookup)', () => {
48
+ expect(registry.readerFor('Plugins/GTM/Deck.PPTX')).toBeInstanceOf(DocumentReader);
49
+ expect(registry.readerFor('d.ODP')).toBeInstanceOf(DocumentReader);
50
+ expect(registry.readerFor('shot.PNG')).toBeInstanceOf(ImageReader);
51
+ expect(registry.readerFor('old.DOC')).toBeInstanceOf(LegacyOfficeReader);
52
+ });
53
+
54
+ it('routes images to the ImageReader (never greppable) and legacy office formats to the LegacyOfficeReader', () => {
55
+ for (const p of ['a.png', 'b.jpg', 'c.jpeg', 'd.gif', 'e.webp']) {
56
+ const reader = registry.readerFor(p);
57
+ expect(reader, p).toBeInstanceOf(ImageReader);
58
+ expect(reader.greppableText, p).toBeUndefined();
59
+ }
60
+ for (const p of ['a.doc', 'b.ppt', 'c.xls']) {
61
+ const reader = registry.readerFor(p);
62
+ expect(reader, p).toBeInstanceOf(LegacyOfficeReader);
63
+ // Not text-editable: read_file can't extract a legacy binary, so an
64
+ // edit_file would destroy it — and the refusal names the way out.
65
+ expect(reader.textEditable, p).toBe(false);
66
+ expect(reader.editRefusal?.(p), p).toContain('Convert the document');
67
+ expect(reader.editRefusal?.(p), p).toContain('uploading a new version');
68
+ }
69
+ });
70
+
71
+ it('falls back to the plain TextReader for unknown extensions, no extension and dot-files', () => {
72
+ // `.svg` is markup and `.zip`/`.mp3` are binary-notice cases — all the
73
+ // text reader's business; `.docx` as a bare dot-file has no extension.
74
+ for (const p of ['notes.md', 'icon.svg', 'bundle.zip', 'song.mp3', 'noext', 'dir/.gitignore', '.docx']) {
75
+ const reader = registry.readerFor(p);
76
+ expect(reader, p).toBeInstanceOf(TextReader);
77
+ expect(reader, p).not.toBeInstanceOf(LegacyOfficeReader);
78
+ expect(reader.textEditable, p).toBe(true);
79
+ }
80
+ });
81
+
82
+ it('adding a format is ONE registry entry — the new reader owns its extension, everything else keeps working', async () => {
83
+ const fake: FileReader = {
84
+ extensions: ['.foo'],
85
+ textEditable: false,
86
+ read: async (): Promise<ReadResult> => ({ kind: 'text', text: 'from the fake reader' }),
87
+ };
88
+ const custom = new FileReaderRegistry([fake, new ImageReader()], new TextReader());
89
+ expect(custom.readerFor('x.foo')).toBe(fake);
90
+ expect(await custom.readerFor('x.foo').read(Buffer.from(''), 'x.foo')).toEqual({
91
+ kind: 'text',
92
+ text: 'from the fake reader',
93
+ });
94
+ expect(custom.readerFor('x.png')).toBeInstanceOf(ImageReader);
95
+ expect(custom.readerFor('x.md')).toBeInstanceOf(TextReader);
96
+ });
97
+ });
@@ -0,0 +1,100 @@
1
+ import { describe, expect, it } from 'vitest';
2
+ import {
3
+ IMAGE_MAX_RAW_BYTES,
4
+ imageDimensions,
5
+ imageMimeType,
6
+ imageNote,
7
+ isImageFile,
8
+ oversizedImageNotice,
9
+ } from '../image-read.js';
10
+
11
+ /** A real, complete 1×1 transparent PNG. */
12
+ const PNG_1X1 = Buffer.from(
13
+ 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg==',
14
+ 'base64',
15
+ );
16
+ /** A real, complete 1×1 GIF89a. */
17
+ const GIF_1X1 = Buffer.from('R0lGODlhAQABAIAAAP///wAAACH5BAEAAAAALAAAAAABAAEAAAICRAEAOw==', 'base64');
18
+
19
+ /** A minimal JPEG prefix: SOI, an APP0 stub, then an SOF0 declaring 4×3. */
20
+ function jpegWithSof(width: number, height: number): Buffer {
21
+ const app0 = Buffer.from([0xff, 0xe0, 0x00, 0x04, 0x00, 0x00]); // 4-byte segment
22
+ const sof0 = Buffer.alloc(2 + 2 + 5 + 3);
23
+ sof0.set([0xff, 0xc0]); // SOF0 marker
24
+ sof0.writeUInt16BE(sof0.length - 2, 2); // segment length
25
+ sof0[4] = 8; // precision
26
+ sof0.writeUInt16BE(height, 5);
27
+ sof0.writeUInt16BE(width, 7);
28
+ return Buffer.concat([Buffer.from([0xff, 0xd8]), app0, sof0]);
29
+ }
30
+
31
+ describe('isImageFile / imageMimeType', () => {
32
+ it('covers exactly the five raster extensions, case-insensitively', () => {
33
+ expect(isImageFile('a/b/logo.png')).toBe(true);
34
+ expect(isImageFile('photo.JPG')).toBe(true);
35
+ expect(isImageFile('photo.jpeg')).toBe(true);
36
+ expect(isImageFile('anim.gif')).toBe(true);
37
+ expect(isImageFile('pic.webp')).toBe(true);
38
+ // SVG is TEXT — must flow through the normal text path.
39
+ expect(isImageFile('icon.svg')).toBe(false);
40
+ expect(isImageFile('favicon.ico')).toBe(false);
41
+ expect(isImageFile('image.bmp')).toBe(false);
42
+ expect(isImageFile('doc.pdf')).toBe(false);
43
+ });
44
+
45
+ it('maps jpg and jpeg to image/jpeg', () => {
46
+ expect(imageMimeType('a.jpg')).toBe('image/jpeg');
47
+ expect(imageMimeType('a.jpeg')).toBe('image/jpeg');
48
+ expect(imageMimeType('a.png')).toBe('image/png');
49
+ });
50
+ });
51
+
52
+ describe('imageDimensions', () => {
53
+ it('reads PNG IHDR', () => {
54
+ expect(imageDimensions(PNG_1X1)).toEqual({ width: 1, height: 1 });
55
+ });
56
+
57
+ it('reads the GIF logical screen descriptor', () => {
58
+ expect(imageDimensions(GIF_1X1)).toEqual({ width: 1, height: 1 });
59
+ });
60
+
61
+ it('requires the FULL GIF87a/GIF89a signature — a bare "GIF" prefix parses no dimensions', () => {
62
+ const fake = Buffer.from('GIFFY!\x05\x00\x07\x00 not a gif at all', 'latin1');
63
+ expect(imageDimensions(fake)).toBeUndefined();
64
+ });
65
+
66
+ it('walks JPEG segments to the SOF0 frame header', () => {
67
+ expect(imageDimensions(jpegWithSof(640, 480))).toEqual({ width: 640, height: 480 });
68
+ });
69
+
70
+ it('returns undefined (never throws) for truncated or unknown bytes', () => {
71
+ expect(imageDimensions(Buffer.from([0x89, 0x50]))).toBeUndefined();
72
+ expect(imageDimensions(Buffer.from([0xff, 0xd8, 0xff, 0xd9]))).toBeUndefined();
73
+ expect(imageDimensions(Buffer.from('RIFFxxxxWEBP'))).toBeUndefined(); // webp: honestly dimensionless
74
+ expect(imageDimensions(Buffer.alloc(0))).toBeUndefined();
75
+ });
76
+ });
77
+
78
+ describe('cap + notes', () => {
79
+ it('the cap is 3.5 MiB raw, whose base64 stays under the ~5 MB client reject line', () => {
80
+ expect(IMAGE_MAX_RAW_BYTES).toBe(3_670_016);
81
+ const encodedChars = Math.ceil(IMAGE_MAX_RAW_BYTES / 3) * 4;
82
+ expect(encodedChars).toBeLessThan(5 * 1024 * 1024);
83
+ });
84
+
85
+ it('imageNote includes dimensions when known and omits them cleanly when not', () => {
86
+ expect(imageNote('a/logo.png', 'image/png', 67, { width: 1, height: 1 })).toBe(
87
+ '[image: a/logo.png — image/png, 67 bytes, 1×1 px]',
88
+ );
89
+ expect(imageNote('p.webp', 'image/webp', 10, undefined)).toBe('[image: p.webp — image/webp, 10 bytes]');
90
+ });
91
+
92
+ it('oversizedImageNotice names the file, the cap arithmetic and the way out', () => {
93
+ const notice = oversizedImageNotice('big.png', 'image/png', 9_999_999);
94
+ expect(notice).toContain('big.png');
95
+ expect(notice).toContain('9999999 bytes');
96
+ expect(notice).toContain('3670016 bytes');
97
+ expect(notice).toContain('Downscale');
98
+ expect(notice).toContain('upload a smaller export');
99
+ });
100
+ });