@n0zer0d4y/vulcan-file-ops 1.2.14 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,478 @@
1
+ /**
2
+ * Image sanitization for HTML that is converted to PDF/DOCX (VFO-13).
3
+ *
4
+ * Only inline `data:image/...;base64,` images are allowed into generated
5
+ * documents. Every other image reference (relative/absolute paths, `file:`,
6
+ * `http(s):`, malformed data URLs, `srcset`, ...) is replaced with the image's
7
+ * `alt` text, because:
8
+ * - pdfmake's browser build rejects such images inside its own promise chain,
9
+ * which previously hung the write and crashed the server (unhandled
10
+ * rejection), and
11
+ * - html-to-docx downloads `http(s):` image URLs (server-side request).
12
+ *
13
+ * The sanitizer is a small tokenizer that follows the WHATWG HTML tag and
14
+ * attribute tokenization rules. It deliberately over-approximates: every
15
+ * `<img`, `<image` and `<source` start tag anywhere in the input (including
16
+ * inside comments, attribute values or raw-text elements) is rewritten, so a
17
+ * downstream parser can never see an image tag that was not sanitized. Image
18
+ * tags are re-serialized canonically (first attribute wins, values re-quoted)
19
+ * so the downstream parser sees exactly the attributes that were validated.
20
+ */
21
+ import zlib from "zlib";
22
+ import { MAX_DECODED_IMAGE_PIXELS, MAX_EMBEDDED_IMAGE_BYTES, } from "./limits.js";
23
+ export const ALL_EMBEDDABLE_IMAGE_TYPES = [
24
+ "png",
25
+ "jpeg",
26
+ "gif",
27
+ "webp",
28
+ "bmp",
29
+ ];
30
+ const DATA_URL_PREFIX_RE = /^data:image\/(png|jpe?g|gif|webp|bmp);base64,/i;
31
+ const BASE64_BODY_RE = /^[A-Za-z0-9+/]*={0,2}$/;
32
+ const HTML_WHITESPACE_RE = /[\t\n\f\r ]+/g;
33
+ function hasSignature(bytes, type) {
34
+ switch (type) {
35
+ case "png":
36
+ return (bytes.length >= 8 &&
37
+ bytes
38
+ .subarray(0, 8)
39
+ .equals(Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])));
40
+ case "jpeg":
41
+ return (bytes.length >= 3 &&
42
+ bytes[0] === 0xff &&
43
+ bytes[1] === 0xd8 &&
44
+ bytes[2] === 0xff);
45
+ case "gif": {
46
+ const sig = bytes.subarray(0, 6).toString("latin1");
47
+ return sig === "GIF87a" || sig === "GIF89a";
48
+ }
49
+ case "webp":
50
+ return (bytes.length >= 12 &&
51
+ bytes.subarray(0, 4).toString("latin1") === "RIFF" &&
52
+ bytes.subarray(8, 12).toString("latin1") === "WEBP");
53
+ case "bmp":
54
+ return bytes.length >= 2 && bytes[0] === 0x42 && bytes[1] === 0x4d;
55
+ }
56
+ }
57
+ /**
58
+ * Validate an image source and return its canonical form
59
+ * (`data:image/<type>;base64,<base64 without whitespace>`), or `null` if the
60
+ * source must not be embedded.
61
+ */
62
+ export function normalizeEmbeddableImageSrc(src, options = {}) {
63
+ if (typeof src !== "string")
64
+ return null;
65
+ const memo = options.memo;
66
+ if (memo?.has(src))
67
+ return memo.get(src) ?? null;
68
+ const result = normalizeImageSrcUncached(src, options);
69
+ memo?.set(src, result);
70
+ if (result !== null)
71
+ memo?.set(result, result);
72
+ return result;
73
+ }
74
+ function normalizeImageSrcUncached(src, options) {
75
+ const allowedTypes = options.allowedTypes ?? ALL_EMBEDDABLE_IMAGE_TYPES;
76
+ const maxBytes = options.maxImageBytes ?? MAX_EMBEDDED_IMAGE_BYTES;
77
+ // Cheap upper bound before any further work on very large strings:
78
+ // base64 is 4 chars per 3 bytes; allow generous slack for whitespace.
79
+ if (src.length > Math.ceil(maxBytes / 3) * 4 * 2 + 1024)
80
+ return null;
81
+ const trimmed = src.replace(/^[\t\n\f\r ]+|[\t\n\f\r ]+$/g, "");
82
+ const prefix = DATA_URL_PREFIX_RE.exec(trimmed);
83
+ if (!prefix)
84
+ return null;
85
+ const declared = prefix[1].toLowerCase();
86
+ const type = declared === "jpg" ? "jpeg" : declared;
87
+ if (!allowedTypes.includes(type))
88
+ return null;
89
+ const body = trimmed.slice(prefix[0].length).replace(HTML_WHITESPACE_RE, "");
90
+ if (body.length === 0 || body.length % 4 !== 0)
91
+ return null;
92
+ if (!BASE64_BODY_RE.test(body))
93
+ return null;
94
+ const padding = body.endsWith("==") ? 2 : body.endsWith("=") ? 1 : 0;
95
+ const decodedBytes = (body.length / 4) * 3 - padding;
96
+ if (decodedBytes > maxBytes)
97
+ return null;
98
+ // The declared type must match the actual content.
99
+ const head = Buffer.from(body.slice(0, 16), "base64");
100
+ if (!hasSignature(head, type))
101
+ return null;
102
+ if (type === "png" && options.verifyPngForPdfkit) {
103
+ const maxPixels = options.maxDecodedPixels ?? MAX_DECODED_IMAGE_PIXELS;
104
+ if (!isPngSafeForPdfkit(Buffer.from(body, "base64"), maxPixels)) {
105
+ return null;
106
+ }
107
+ }
108
+ return `data:image/${type};base64,${body}`;
109
+ }
110
+ // ---------------------------------------------------------------------------
111
+ // PNG validation for pdfkit
112
+ // ---------------------------------------------------------------------------
113
+ const PNG_ALLOWED_BIT_DEPTHS = {
114
+ 0: [1, 2, 4, 8, 16], // greyscale
115
+ 2: [8, 16], // truecolor
116
+ 3: [1, 2, 4, 8], // indexed
117
+ 4: [8, 16], // greyscale + alpha
118
+ 6: [8, 16], // truecolor + alpha
119
+ };
120
+ /**
121
+ * pdfkit (bundled in pdfmake) embeds most PNGs as-is, but PNGs with an alpha
122
+ * channel, indexed transparency (tRNS) or interlacing are decoded with png-js
123
+ * inside an asynchronous zlib callback. Any error there (corrupt deflate
124
+ * stream, invalid scanline filter) is thrown from that callback as an
125
+ * uncaught exception, which terminates the server. Such PNGs are therefore
126
+ * decoded here first, synchronously and with Node's zlib, replicating png-js'
127
+ * scanline walk, so only images that png-js can decode without throwing are
128
+ * embedded. Also bounds the memory png-js would allocate.
129
+ */
130
+ export function isPngSafeForPdfkit(png, maxPixels) {
131
+ try {
132
+ let pos = 8;
133
+ let ihdr = null;
134
+ let hasPalette = false;
135
+ let indexedTransparency = false;
136
+ const idat = [];
137
+ let sawEnd = false;
138
+ while (!sawEnd) {
139
+ if (pos + 8 > png.length)
140
+ return false;
141
+ const length = png.readUInt32BE(pos);
142
+ const type = png.toString("latin1", pos + 4, pos + 8);
143
+ pos += 8;
144
+ if (length > 0x7fffffff || pos + length + 4 > png.length)
145
+ return false;
146
+ const data = png.subarray(pos, pos + length);
147
+ if (ihdr === null && type !== "IHDR")
148
+ return false;
149
+ switch (type) {
150
+ case "IHDR":
151
+ if (ihdr !== null || length !== 13)
152
+ return false;
153
+ ihdr = {
154
+ width: data.readUInt32BE(0),
155
+ height: data.readUInt32BE(4),
156
+ bits: data[8],
157
+ colorType: data[9],
158
+ interlace: data[12],
159
+ };
160
+ if (data[10] !== 0 || data[11] !== 0)
161
+ return false;
162
+ break;
163
+ case "PLTE":
164
+ hasPalette = length > 0 && length % 3 === 0;
165
+ break;
166
+ case "IDAT":
167
+ idat.push(data);
168
+ break;
169
+ case "tRNS":
170
+ if (ihdr?.colorType === 3)
171
+ indexedTransparency = true;
172
+ break;
173
+ case "IEND":
174
+ sawEnd = true;
175
+ break;
176
+ }
177
+ pos += length + 4; // data + CRC
178
+ }
179
+ if (!ihdr || idat.length === 0)
180
+ return false;
181
+ const { width, height, bits, colorType, interlace } = ihdr;
182
+ if (width < 1 || height < 1 || width > 0x7fffffff || height > 0x7fffffff) {
183
+ return false;
184
+ }
185
+ const allowedBits = PNG_ALLOWED_BIT_DEPTHS[colorType];
186
+ if (!allowedBits || !allowedBits.includes(bits))
187
+ return false;
188
+ if (interlace !== 0 && interlace !== 1)
189
+ return false;
190
+ if (colorType === 3 && !hasPalette)
191
+ return false;
192
+ const hasAlpha = colorType === 4 || colorType === 6;
193
+ const needsDecode = hasAlpha || indexedTransparency || interlace === 1;
194
+ if (!needsDecode)
195
+ return true; // embedded verbatim, never decoded
196
+ if (width * height > maxPixels)
197
+ return false;
198
+ // Same arithmetic as png-js (including its fractional pixelBytes).
199
+ const colors = colorType === 2 || colorType === 6 ? 3 : 1;
200
+ const pixelBytes = (bits * (colors + (hasAlpha ? 1 : 0))) / 8;
201
+ const passes = interlace === 1
202
+ ? [
203
+ [0, 0, 8, 8],
204
+ [4, 0, 8, 8],
205
+ [0, 4, 4, 8],
206
+ [2, 0, 4, 4],
207
+ [0, 2, 2, 4],
208
+ [1, 0, 2, 2],
209
+ [0, 1, 1, 2],
210
+ ]
211
+ : [[0, 0, 1, 1]];
212
+ let expected = 0;
213
+ for (const [x0, y0, dx, dy] of passes) {
214
+ const w = Math.ceil((width - x0) / dx);
215
+ const h = Math.ceil((height - y0) / dy);
216
+ if (h > 0)
217
+ expected += h * (1 + Math.max(0, Math.ceil(pixelBytes * w)));
218
+ }
219
+ const compressed = Buffer.concat(idat);
220
+ const { buffer: inflated, engine } = zlib.inflateSync(compressed, {
221
+ info: true,
222
+ maxOutputLength: expected + 1024,
223
+ });
224
+ // Reject trailing data after the zlib stream (strict parity with pako).
225
+ if (engine.bytesWritten !== compressed.length)
226
+ return false;
227
+ // Walk the scanlines exactly like png-js and check every filter byte.
228
+ let p = 0;
229
+ const length = inflated.length;
230
+ for (const [x0, y0, dx, dy] of passes) {
231
+ const w = Math.ceil((width - x0) / dx);
232
+ const h = Math.ceil((height - y0) / dy);
233
+ const rowBytes = Math.max(0, Math.ceil(pixelBytes * w));
234
+ for (let row = 0; row < h && p < length; row++) {
235
+ if (inflated[p++] > 4)
236
+ return false;
237
+ p += rowBytes;
238
+ }
239
+ }
240
+ return true;
241
+ }
242
+ catch {
243
+ return false;
244
+ }
245
+ }
246
+ const isTagWhitespace = (ch) => ch === "\t" || ch === "\n" || ch === "\f" || ch === "\r" || ch === " ";
247
+ /**
248
+ * Parse the attributes of a start tag whose name ends at `pos`.
249
+ * Follows the HTML tokenizer states "before attribute name" through
250
+ * "after attribute value (quoted)".
251
+ */
252
+ function parseTagAttributes(html, pos) {
253
+ const attributes = new Map();
254
+ const len = html.length;
255
+ let i = pos;
256
+ while (i < len) {
257
+ // Before attribute name: skip whitespace and stray solidus.
258
+ const ch = html[i];
259
+ if (isTagWhitespace(ch) || ch === "/") {
260
+ i++;
261
+ continue;
262
+ }
263
+ if (ch === ">") {
264
+ return { attributes, end: i + 1 };
265
+ }
266
+ // Attribute name (a leading "=" is part of the name per spec).
267
+ let nameStart = i;
268
+ i++;
269
+ while (i < len) {
270
+ const c = html[i];
271
+ if (isTagWhitespace(c) || c === "/" || c === ">" || c === "=")
272
+ break;
273
+ i++;
274
+ }
275
+ const name = html.slice(nameStart, i).toLowerCase();
276
+ // After attribute name.
277
+ while (i < len && isTagWhitespace(html[i]))
278
+ i++;
279
+ let value = "";
280
+ if (i < len && html[i] === "=") {
281
+ i++;
282
+ while (i < len && isTagWhitespace(html[i]))
283
+ i++;
284
+ if (i >= len)
285
+ break;
286
+ const q = html[i];
287
+ if (q === '"' || q === "'") {
288
+ const close = html.indexOf(q, i + 1);
289
+ if (close === -1) {
290
+ i = len; // EOF inside quoted value
291
+ break;
292
+ }
293
+ value = html.slice(i + 1, close);
294
+ i = close + 1;
295
+ }
296
+ else if (q !== ">") {
297
+ const valueStart = i;
298
+ while (i < len && !isTagWhitespace(html[i]) && html[i] !== ">")
299
+ i++;
300
+ value = html.slice(valueStart, i);
301
+ }
302
+ }
303
+ if (!attributes.has(name))
304
+ attributes.set(name, value);
305
+ }
306
+ // EOF inside the tag: per spec the tag is dropped.
307
+ return { attributes, end: -1 };
308
+ }
309
+ const NAMED_REFS = {
310
+ amp: "&",
311
+ lt: "<",
312
+ gt: ">",
313
+ quot: '"',
314
+ apos: "'",
315
+ nbsp: " ",
316
+ };
317
+ /** Decode the character references that matter for alt text. */
318
+ function decodeBasicEntities(text) {
319
+ return text.replace(/&(#[xX][0-9a-fA-F]{1,6}|#[0-9]{1,7}|[a-zA-Z]+);/g, (match, ref) => {
320
+ if (ref[0] === "#") {
321
+ const code = ref[1] === "x" || ref[1] === "X"
322
+ ? parseInt(ref.slice(2), 16)
323
+ : parseInt(ref.slice(1), 10);
324
+ if (!Number.isFinite(code) ||
325
+ code <= 0 ||
326
+ code > 0x10ffff ||
327
+ (code >= 0xd800 && code <= 0xdfff)) {
328
+ return "�";
329
+ }
330
+ return String.fromCodePoint(code);
331
+ }
332
+ const named = NAMED_REFS[ref.toLowerCase()];
333
+ return named ?? match;
334
+ });
335
+ }
336
+ export function escapeHtmlText(text) {
337
+ return text
338
+ .replace(/&/g, "&amp;")
339
+ .replace(/</g, "&lt;")
340
+ .replace(/>/g, "&gt;")
341
+ .replace(/"/g, "&quot;")
342
+ .replace(/'/g, "&#39;");
343
+ }
344
+ const SAFE_ATTR_NAME_RE = /^[a-zA-Z_:][-a-zA-Z0-9_:.]*$/;
345
+ /** Attributes that can make a converter fetch or reference an image. */
346
+ const IMAGE_SOURCE_ATTRS = new Set(["src", "href", "xlink:href"]);
347
+ const ALWAYS_DROPPED_ATTRS = new Set([
348
+ "srcset",
349
+ "data-src",
350
+ "data-srcset",
351
+ "lowsrc",
352
+ "dynsrc",
353
+ "longdesc",
354
+ "usemap",
355
+ "imagesizes",
356
+ "imagesrcset",
357
+ ]);
358
+ function serializeTag(tagName, attributes) {
359
+ let out = `<${tagName}`;
360
+ for (const [name, value] of attributes) {
361
+ if (!SAFE_ATTR_NAME_RE.test(name))
362
+ continue;
363
+ // Keep character references as written (they decode identically when
364
+ // re-parsed inside a double-quoted value); only the quote must be escaped.
365
+ out += ` ${name}="${value.replace(/"/g, "&quot;")}"`;
366
+ }
367
+ return out + ">";
368
+ }
369
+ function sanitizeImageTag(tagName, attributes, options) {
370
+ const kept = new Map();
371
+ let hasSource = false;
372
+ for (const [name, value] of attributes) {
373
+ if (ALWAYS_DROPPED_ATTRS.has(name))
374
+ continue;
375
+ if (IMAGE_SOURCE_ATTRS.has(name)) {
376
+ const normalized = normalizeEmbeddableImageSrc(value, options);
377
+ if (normalized === null)
378
+ continue;
379
+ kept.set(name, normalized);
380
+ hasSource = true;
381
+ continue;
382
+ }
383
+ kept.set(name, value);
384
+ }
385
+ // A disallowed source is not "partially" kept: if any source attribute was
386
+ // rejected, the whole image is replaced (avoids parser disagreements over
387
+ // which of src/href wins).
388
+ const rejectedSource = [...attributes.keys()].some((name) => IMAGE_SOURCE_ATTRS.has(name) && !kept.has(name));
389
+ if (hasSource && !rejectedSource) {
390
+ return serializeTag(tagName, kept);
391
+ }
392
+ const alt = attributes.get("alt");
393
+ return alt ? escapeHtmlText(decodeBasicEntities(alt)) : "";
394
+ }
395
+ function sanitizeSourceTag(attributes) {
396
+ const kept = new Map();
397
+ for (const [name, value] of attributes) {
398
+ if (name === "src" || name === "srcset" || ALWAYS_DROPPED_ATTRS.has(name)) {
399
+ continue;
400
+ }
401
+ kept.set(name, value);
402
+ }
403
+ return serializeTag("source", kept);
404
+ }
405
+ const IMAGE_TAG_START_RE = /<(img|image|source)(?=[\t\n\f\r />]|$)/giy;
406
+ /**
407
+ * Rewrite every image-bearing tag in `html` so that only validated inline
408
+ * `data:` images remain. Non-embeddable images are replaced with their
409
+ * escaped `alt` text (or removed when there is none).
410
+ */
411
+ export function sanitizeHtmlImages(html, options = {}) {
412
+ let out = "";
413
+ let last = 0;
414
+ let searchFrom = 0;
415
+ while (searchFrom < html.length) {
416
+ const lt = html.indexOf("<", searchFrom);
417
+ if (lt === -1)
418
+ break;
419
+ IMAGE_TAG_START_RE.lastIndex = lt;
420
+ const match = IMAGE_TAG_START_RE.exec(html);
421
+ if (!match) {
422
+ searchFrom = lt + 1;
423
+ continue;
424
+ }
425
+ const tagName = match[1].toLowerCase();
426
+ const parsed = parseTagAttributes(html, lt + match[0].length);
427
+ out += html.slice(last, lt);
428
+ if (parsed.end === -1) {
429
+ // Unterminated tag at end of input: drop it.
430
+ last = html.length;
431
+ break;
432
+ }
433
+ out +=
434
+ tagName === "source"
435
+ ? sanitizeSourceTag(parsed.attributes)
436
+ : sanitizeImageTag(tagName, parsed.attributes, options);
437
+ last = parsed.end;
438
+ searchFrom = parsed.end;
439
+ }
440
+ return out + html.slice(last);
441
+ }
442
+ /**
443
+ * Defense in depth for the PDF path: walk a pdfmake content tree and replace
444
+ * any `image` node whose source is not an embeddable data URL (e.g. one that
445
+ * entered through html-to-pdfmake's `data-pdfmake` attribute) with empty text.
446
+ * Mutates and returns the tree.
447
+ */
448
+ export function scrubPdfmakeImages(content, options = {}) {
449
+ const seen = new WeakSet();
450
+ const visit = (node) => {
451
+ if (Array.isArray(node)) {
452
+ if (seen.has(node))
453
+ return node;
454
+ seen.add(node);
455
+ for (let i = 0; i < node.length; i++)
456
+ node[i] = visit(node[i]);
457
+ return node;
458
+ }
459
+ if (node && typeof node === "object") {
460
+ if (seen.has(node))
461
+ return node;
462
+ seen.add(node);
463
+ const record = node;
464
+ if (Object.prototype.hasOwnProperty.call(record, "image")) {
465
+ const normalized = normalizeEmbeddableImageSrc(record.image, options);
466
+ if (normalized === null)
467
+ return { text: "" };
468
+ record.image = normalized;
469
+ }
470
+ for (const key of Object.keys(record)) {
471
+ record[key] = visit(record[key]);
472
+ }
473
+ }
474
+ return node;
475
+ };
476
+ return visit(content);
477
+ }
478
+ //# sourceMappingURL=html-image-sanitizer.js.map
@@ -9,6 +9,8 @@
9
9
  * - html-to-docx: HTML → DOCX conversion
10
10
  * - jsdom: DOM emulation for Node.js
11
11
  */
12
+ import { sanitizeHtmlImages, scrubPdfmakeImages, } from "./html-image-sanitizer.js";
13
+ import { PDF_GENERATION_TIMEOUT_MS } from "./limits.js";
12
14
  // Lazy-loaded libraries (imported only when needed)
13
15
  let pdfMake = null;
14
16
  let pdfFonts = null;
@@ -61,10 +63,13 @@ function sanitizeHTMLForDOCX(html) {
61
63
  // Common typographic characters
62
64
  .replace(/—/g, "&mdash;") // Em dash
63
65
  .replace(/–/g, "&ndash;") // En dash
64
- .replace(/"/g, "&ldquo;") // Left double quote
65
- .replace(/"/g, "&rdquo;") // Right double quote
66
- .replace(/'/g, "&lsquo;") // Left single quote
67
- .replace(/'/g, "&rsquo;") // Right single quote
66
+ // Curly quotes only (written as escapes so they cannot be "normalized"
67
+ // into ASCII quotes, which previously turned every quoted attribute
68
+ // value - e.g. src="data:..." or style="..." - into garbage).
69
+ .replace(/“/g, "&ldquo;") // Left double quote
70
+ .replace(/”/g, "&rdquo;") // Right double quote
71
+ .replace(/‘/g, "&lsquo;") // Left single quote
72
+ .replace(/’/g, "&rsquo;") // Right single quote
68
73
  .replace(/…/g, "&hellip;") // Ellipsis
69
74
  // Degree and other symbols
70
75
  .replace(/°/g, "&deg;") // Degree
@@ -124,8 +129,14 @@ export async function htmlToPDF(htmlContent, options = {}) {
124
129
  }
125
130
  // Create DOM window for html-to-pdfmake
126
131
  const { window } = new jsdom("");
132
+ // Only validated inline data: images may reach pdfmake (VFO-13).
133
+ const imageOptions = {
134
+ ...PDF_IMAGE_OPTIONS,
135
+ memo: new Map(),
136
+ };
137
+ const safeHTML = sanitizeHtmlImages(htmlContent, imageOptions);
127
138
  // Convert HTML to PDFMake format with styling
128
- const converted = htmlToPdfmake(htmlContent, {
139
+ const converted = htmlToPdfmake(safeHTML, {
129
140
  window,
130
141
  defaultStyles: {
131
142
  // Headings with colors
@@ -227,16 +238,66 @@ export async function htmlToPDF(htmlContent, options = {}) {
227
238
  pageSize: "A4",
228
239
  pageMargins: [40, 60, 40, 60],
229
240
  };
230
- // Generate PDF and return as Buffer
241
+ // Defense in depth: html-to-pdfmake can also emit image nodes from
242
+ // `data-pdfmake` attributes, so validate the final tree as well.
243
+ scrubPdfmakeImages(converted, imageOptions);
244
+ return renderPdfBuffer(docDefinition, options.timeoutMs ?? PDF_GENERATION_TIMEOUT_MS);
245
+ }
246
+ /**
247
+ * pdfkit (used by pdfmake) can only embed PNG and JPEG images; anything else
248
+ * would make the whole PDF fail, so other formats degrade to alt text.
249
+ */
250
+ const PDF_IMAGE_OPTIONS = {
251
+ allowedTypes: ["png", "jpeg"],
252
+ verifyPngForPdfkit: true,
253
+ };
254
+ const DOCX_IMAGE_OPTIONS = {};
255
+ function toError(error) {
256
+ if (error instanceof Error)
257
+ return error;
258
+ return new Error(typeof error === "string" ? error : String(error));
259
+ }
260
+ /**
261
+ * Render a pdfmake document definition to a Buffer.
262
+ *
263
+ * pdfmake 0.2.x's `getBuffer(cb)` builds the document inside an internal
264
+ * promise chain: if building fails (e.g. an invalid image), the callback is
265
+ * never called and the rejection is unhandled, which hangs the write and
266
+ * terminates Node (VFO-13). pdfmake exposes no promise API to catch that, so
267
+ * instead we use `getStream()` without a callback: it builds the PDFKit
268
+ * document synchronously (errors are thrown to us here) and skips pdfmake's
269
+ * URL resolver entirely (no network fetches of fonts/images). The stream is
270
+ * then drained with error handling and a hard timeout, so this promise always
271
+ * settles.
272
+ */
273
+ function renderPdfBuffer(docDefinition, timeoutMs) {
231
274
  return new Promise((resolve, reject) => {
275
+ let settled = false;
276
+ const timer = setTimeout(() => {
277
+ settle(new Error(`PDF generation timed out after ${timeoutMs} ms`));
278
+ }, timeoutMs);
279
+ timer.unref?.();
280
+ function settle(error, buffer) {
281
+ if (settled)
282
+ return;
283
+ settled = true;
284
+ clearTimeout(timer);
285
+ if (error)
286
+ reject(error);
287
+ else
288
+ resolve(buffer);
289
+ }
232
290
  try {
233
291
  const pdfDoc = pdfMake.createPdf(docDefinition);
234
- pdfDoc.getBuffer((buffer) => {
235
- resolve(buffer);
236
- });
292
+ const stream = pdfDoc.getStream();
293
+ const chunks = [];
294
+ stream.on("data", (chunk) => chunks.push(chunk));
295
+ stream.on("end", () => settle(null, Buffer.concat(chunks)));
296
+ stream.on("error", (error) => settle(toError(error)));
297
+ stream.end();
237
298
  }
238
299
  catch (error) {
239
- reject(error);
300
+ settle(toError(error));
240
301
  }
241
302
  });
242
303
  }
@@ -262,8 +323,11 @@ export async function htmlToDOCX(htmlContent, options = {}) {
262
323
  const module = await import("@turbodocx/html-to-docx");
263
324
  HTMLtoDOCX = module.default || module;
264
325
  }
326
+ // Only validated inline data: images may reach html-to-docx, which would
327
+ // otherwise download http(s) image URLs (VFO-13).
328
+ const imageSafeHTML = sanitizeHtmlImages(htmlContent, DOCX_IMAGE_OPTIONS);
265
329
  // Sanitize HTML to handle problematic Unicode characters
266
- const sanitizedHTML = sanitizeHTMLForDOCX(htmlContent);
330
+ const sanitizedHTML = sanitizeHTMLForDOCX(imageSafeHTML);
267
331
  // DOCX generation options
268
332
  const docxOptions = {
269
333
  title: options.title || "Document",