@n0zer0d4y/vulcan-file-ops 1.2.13 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +99 -27
- package/README.md +153 -118
- package/dist/cli.js +16 -0
- package/dist/server/index.js +150 -39
- package/dist/tools/filesystem-tools.js +222 -39
- package/dist/tools/read-tools.js +89 -16
- package/dist/tools/shell-tool.js +48 -62
- package/dist/tools/write-tools.js +19 -17
- package/dist/types/index.js +6 -3
- package/dist/utils/command-path-extraction.js +169 -174
- package/dist/utils/command-validation.js +62 -10
- package/dist/utils/document-parser.js +43 -5
- package/dist/utils/html-image-sanitizer.js +478 -0
- package/dist/utils/html-to-document.js +75 -11
- package/dist/utils/lib.js +357 -80
- package/dist/utils/limits.js +47 -0
- package/dist/utils/regex-worker.js +262 -0
- package/dist/utils/shell-parser.js +225 -0
- package/dist/utils/zip-guard.js +266 -0
- package/package.json +11 -8
|
@@ -0,0 +1,478 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Image sanitization for HTML that is converted to PDF/DOCX (VFO-13).
|
|
3
|
+
*
|
|
4
|
+
* Only inline `data:image/...;base64,` images are allowed into generated
|
|
5
|
+
* documents. Every other image reference (relative/absolute paths, `file:`,
|
|
6
|
+
* `http(s):`, malformed data URLs, `srcset`, ...) is replaced with the image's
|
|
7
|
+
* `alt` text, because:
|
|
8
|
+
* - pdfmake's browser build rejects such images inside its own promise chain,
|
|
9
|
+
* which previously hung the write and crashed the server (unhandled
|
|
10
|
+
* rejection), and
|
|
11
|
+
* - html-to-docx downloads `http(s):` image URLs (server-side request).
|
|
12
|
+
*
|
|
13
|
+
* The sanitizer is a small tokenizer that follows the WHATWG HTML tag and
|
|
14
|
+
* attribute tokenization rules. It deliberately over-approximates: every
|
|
15
|
+
* `<img`, `<image` and `<source` start tag anywhere in the input (including
|
|
16
|
+
* inside comments, attribute values or raw-text elements) is rewritten, so a
|
|
17
|
+
* downstream parser can never see an image tag that was not sanitized. Image
|
|
18
|
+
* tags are re-serialized canonically (first attribute wins, values re-quoted)
|
|
19
|
+
* so the downstream parser sees exactly the attributes that were validated.
|
|
20
|
+
*/
|
|
21
|
+
import zlib from "zlib";
|
|
22
|
+
import { MAX_DECODED_IMAGE_PIXELS, MAX_EMBEDDED_IMAGE_BYTES, } from "./limits.js";
|
|
23
|
+
export const ALL_EMBEDDABLE_IMAGE_TYPES = [
|
|
24
|
+
"png",
|
|
25
|
+
"jpeg",
|
|
26
|
+
"gif",
|
|
27
|
+
"webp",
|
|
28
|
+
"bmp",
|
|
29
|
+
];
|
|
30
|
+
const DATA_URL_PREFIX_RE = /^data:image\/(png|jpe?g|gif|webp|bmp);base64,/i;
|
|
31
|
+
const BASE64_BODY_RE = /^[A-Za-z0-9+/]*={0,2}$/;
|
|
32
|
+
const HTML_WHITESPACE_RE = /[\t\n\f\r ]+/g;
|
|
33
|
+
function hasSignature(bytes, type) {
|
|
34
|
+
switch (type) {
|
|
35
|
+
case "png":
|
|
36
|
+
return (bytes.length >= 8 &&
|
|
37
|
+
bytes
|
|
38
|
+
.subarray(0, 8)
|
|
39
|
+
.equals(Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])));
|
|
40
|
+
case "jpeg":
|
|
41
|
+
return (bytes.length >= 3 &&
|
|
42
|
+
bytes[0] === 0xff &&
|
|
43
|
+
bytes[1] === 0xd8 &&
|
|
44
|
+
bytes[2] === 0xff);
|
|
45
|
+
case "gif": {
|
|
46
|
+
const sig = bytes.subarray(0, 6).toString("latin1");
|
|
47
|
+
return sig === "GIF87a" || sig === "GIF89a";
|
|
48
|
+
}
|
|
49
|
+
case "webp":
|
|
50
|
+
return (bytes.length >= 12 &&
|
|
51
|
+
bytes.subarray(0, 4).toString("latin1") === "RIFF" &&
|
|
52
|
+
bytes.subarray(8, 12).toString("latin1") === "WEBP");
|
|
53
|
+
case "bmp":
|
|
54
|
+
return bytes.length >= 2 && bytes[0] === 0x42 && bytes[1] === 0x4d;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* Validate an image source and return its canonical form
|
|
59
|
+
* (`data:image/<type>;base64,<base64 without whitespace>`), or `null` if the
|
|
60
|
+
* source must not be embedded.
|
|
61
|
+
*/
|
|
62
|
+
export function normalizeEmbeddableImageSrc(src, options = {}) {
|
|
63
|
+
if (typeof src !== "string")
|
|
64
|
+
return null;
|
|
65
|
+
const memo = options.memo;
|
|
66
|
+
if (memo?.has(src))
|
|
67
|
+
return memo.get(src) ?? null;
|
|
68
|
+
const result = normalizeImageSrcUncached(src, options);
|
|
69
|
+
memo?.set(src, result);
|
|
70
|
+
if (result !== null)
|
|
71
|
+
memo?.set(result, result);
|
|
72
|
+
return result;
|
|
73
|
+
}
|
|
74
|
+
function normalizeImageSrcUncached(src, options) {
|
|
75
|
+
const allowedTypes = options.allowedTypes ?? ALL_EMBEDDABLE_IMAGE_TYPES;
|
|
76
|
+
const maxBytes = options.maxImageBytes ?? MAX_EMBEDDED_IMAGE_BYTES;
|
|
77
|
+
// Cheap upper bound before any further work on very large strings:
|
|
78
|
+
// base64 is 4 chars per 3 bytes; allow generous slack for whitespace.
|
|
79
|
+
if (src.length > Math.ceil(maxBytes / 3) * 4 * 2 + 1024)
|
|
80
|
+
return null;
|
|
81
|
+
const trimmed = src.replace(/^[\t\n\f\r ]+|[\t\n\f\r ]+$/g, "");
|
|
82
|
+
const prefix = DATA_URL_PREFIX_RE.exec(trimmed);
|
|
83
|
+
if (!prefix)
|
|
84
|
+
return null;
|
|
85
|
+
const declared = prefix[1].toLowerCase();
|
|
86
|
+
const type = declared === "jpg" ? "jpeg" : declared;
|
|
87
|
+
if (!allowedTypes.includes(type))
|
|
88
|
+
return null;
|
|
89
|
+
const body = trimmed.slice(prefix[0].length).replace(HTML_WHITESPACE_RE, "");
|
|
90
|
+
if (body.length === 0 || body.length % 4 !== 0)
|
|
91
|
+
return null;
|
|
92
|
+
if (!BASE64_BODY_RE.test(body))
|
|
93
|
+
return null;
|
|
94
|
+
const padding = body.endsWith("==") ? 2 : body.endsWith("=") ? 1 : 0;
|
|
95
|
+
const decodedBytes = (body.length / 4) * 3 - padding;
|
|
96
|
+
if (decodedBytes > maxBytes)
|
|
97
|
+
return null;
|
|
98
|
+
// The declared type must match the actual content.
|
|
99
|
+
const head = Buffer.from(body.slice(0, 16), "base64");
|
|
100
|
+
if (!hasSignature(head, type))
|
|
101
|
+
return null;
|
|
102
|
+
if (type === "png" && options.verifyPngForPdfkit) {
|
|
103
|
+
const maxPixels = options.maxDecodedPixels ?? MAX_DECODED_IMAGE_PIXELS;
|
|
104
|
+
if (!isPngSafeForPdfkit(Buffer.from(body, "base64"), maxPixels)) {
|
|
105
|
+
return null;
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
return `data:image/${type};base64,${body}`;
|
|
109
|
+
}
|
|
110
|
+
// ---------------------------------------------------------------------------
|
|
111
|
+
// PNG validation for pdfkit
|
|
112
|
+
// ---------------------------------------------------------------------------
|
|
113
|
+
const PNG_ALLOWED_BIT_DEPTHS = {
|
|
114
|
+
0: [1, 2, 4, 8, 16], // greyscale
|
|
115
|
+
2: [8, 16], // truecolor
|
|
116
|
+
3: [1, 2, 4, 8], // indexed
|
|
117
|
+
4: [8, 16], // greyscale + alpha
|
|
118
|
+
6: [8, 16], // truecolor + alpha
|
|
119
|
+
};
|
|
120
|
+
/**
|
|
121
|
+
* pdfkit (bundled in pdfmake) embeds most PNGs as-is, but PNGs with an alpha
|
|
122
|
+
* channel, indexed transparency (tRNS) or interlacing are decoded with png-js
|
|
123
|
+
* inside an asynchronous zlib callback. Any error there (corrupt deflate
|
|
124
|
+
* stream, invalid scanline filter) is thrown from that callback as an
|
|
125
|
+
* uncaught exception, which terminates the server. Such PNGs are therefore
|
|
126
|
+
* decoded here first, synchronously and with Node's zlib, replicating png-js'
|
|
127
|
+
* scanline walk, so only images that png-js can decode without throwing are
|
|
128
|
+
* embedded. Also bounds the memory png-js would allocate.
|
|
129
|
+
*/
|
|
130
|
+
export function isPngSafeForPdfkit(png, maxPixels) {
|
|
131
|
+
try {
|
|
132
|
+
let pos = 8;
|
|
133
|
+
let ihdr = null;
|
|
134
|
+
let hasPalette = false;
|
|
135
|
+
let indexedTransparency = false;
|
|
136
|
+
const idat = [];
|
|
137
|
+
let sawEnd = false;
|
|
138
|
+
while (!sawEnd) {
|
|
139
|
+
if (pos + 8 > png.length)
|
|
140
|
+
return false;
|
|
141
|
+
const length = png.readUInt32BE(pos);
|
|
142
|
+
const type = png.toString("latin1", pos + 4, pos + 8);
|
|
143
|
+
pos += 8;
|
|
144
|
+
if (length > 0x7fffffff || pos + length + 4 > png.length)
|
|
145
|
+
return false;
|
|
146
|
+
const data = png.subarray(pos, pos + length);
|
|
147
|
+
if (ihdr === null && type !== "IHDR")
|
|
148
|
+
return false;
|
|
149
|
+
switch (type) {
|
|
150
|
+
case "IHDR":
|
|
151
|
+
if (ihdr !== null || length !== 13)
|
|
152
|
+
return false;
|
|
153
|
+
ihdr = {
|
|
154
|
+
width: data.readUInt32BE(0),
|
|
155
|
+
height: data.readUInt32BE(4),
|
|
156
|
+
bits: data[8],
|
|
157
|
+
colorType: data[9],
|
|
158
|
+
interlace: data[12],
|
|
159
|
+
};
|
|
160
|
+
if (data[10] !== 0 || data[11] !== 0)
|
|
161
|
+
return false;
|
|
162
|
+
break;
|
|
163
|
+
case "PLTE":
|
|
164
|
+
hasPalette = length > 0 && length % 3 === 0;
|
|
165
|
+
break;
|
|
166
|
+
case "IDAT":
|
|
167
|
+
idat.push(data);
|
|
168
|
+
break;
|
|
169
|
+
case "tRNS":
|
|
170
|
+
if (ihdr?.colorType === 3)
|
|
171
|
+
indexedTransparency = true;
|
|
172
|
+
break;
|
|
173
|
+
case "IEND":
|
|
174
|
+
sawEnd = true;
|
|
175
|
+
break;
|
|
176
|
+
}
|
|
177
|
+
pos += length + 4; // data + CRC
|
|
178
|
+
}
|
|
179
|
+
if (!ihdr || idat.length === 0)
|
|
180
|
+
return false;
|
|
181
|
+
const { width, height, bits, colorType, interlace } = ihdr;
|
|
182
|
+
if (width < 1 || height < 1 || width > 0x7fffffff || height > 0x7fffffff) {
|
|
183
|
+
return false;
|
|
184
|
+
}
|
|
185
|
+
const allowedBits = PNG_ALLOWED_BIT_DEPTHS[colorType];
|
|
186
|
+
if (!allowedBits || !allowedBits.includes(bits))
|
|
187
|
+
return false;
|
|
188
|
+
if (interlace !== 0 && interlace !== 1)
|
|
189
|
+
return false;
|
|
190
|
+
if (colorType === 3 && !hasPalette)
|
|
191
|
+
return false;
|
|
192
|
+
const hasAlpha = colorType === 4 || colorType === 6;
|
|
193
|
+
const needsDecode = hasAlpha || indexedTransparency || interlace === 1;
|
|
194
|
+
if (!needsDecode)
|
|
195
|
+
return true; // embedded verbatim, never decoded
|
|
196
|
+
if (width * height > maxPixels)
|
|
197
|
+
return false;
|
|
198
|
+
// Same arithmetic as png-js (including its fractional pixelBytes).
|
|
199
|
+
const colors = colorType === 2 || colorType === 6 ? 3 : 1;
|
|
200
|
+
const pixelBytes = (bits * (colors + (hasAlpha ? 1 : 0))) / 8;
|
|
201
|
+
const passes = interlace === 1
|
|
202
|
+
? [
|
|
203
|
+
[0, 0, 8, 8],
|
|
204
|
+
[4, 0, 8, 8],
|
|
205
|
+
[0, 4, 4, 8],
|
|
206
|
+
[2, 0, 4, 4],
|
|
207
|
+
[0, 2, 2, 4],
|
|
208
|
+
[1, 0, 2, 2],
|
|
209
|
+
[0, 1, 1, 2],
|
|
210
|
+
]
|
|
211
|
+
: [[0, 0, 1, 1]];
|
|
212
|
+
let expected = 0;
|
|
213
|
+
for (const [x0, y0, dx, dy] of passes) {
|
|
214
|
+
const w = Math.ceil((width - x0) / dx);
|
|
215
|
+
const h = Math.ceil((height - y0) / dy);
|
|
216
|
+
if (h > 0)
|
|
217
|
+
expected += h * (1 + Math.max(0, Math.ceil(pixelBytes * w)));
|
|
218
|
+
}
|
|
219
|
+
const compressed = Buffer.concat(idat);
|
|
220
|
+
const { buffer: inflated, engine } = zlib.inflateSync(compressed, {
|
|
221
|
+
info: true,
|
|
222
|
+
maxOutputLength: expected + 1024,
|
|
223
|
+
});
|
|
224
|
+
// Reject trailing data after the zlib stream (strict parity with pako).
|
|
225
|
+
if (engine.bytesWritten !== compressed.length)
|
|
226
|
+
return false;
|
|
227
|
+
// Walk the scanlines exactly like png-js and check every filter byte.
|
|
228
|
+
let p = 0;
|
|
229
|
+
const length = inflated.length;
|
|
230
|
+
for (const [x0, y0, dx, dy] of passes) {
|
|
231
|
+
const w = Math.ceil((width - x0) / dx);
|
|
232
|
+
const h = Math.ceil((height - y0) / dy);
|
|
233
|
+
const rowBytes = Math.max(0, Math.ceil(pixelBytes * w));
|
|
234
|
+
for (let row = 0; row < h && p < length; row++) {
|
|
235
|
+
if (inflated[p++] > 4)
|
|
236
|
+
return false;
|
|
237
|
+
p += rowBytes;
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
return true;
|
|
241
|
+
}
|
|
242
|
+
catch {
|
|
243
|
+
return false;
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
const isTagWhitespace = (ch) => ch === "\t" || ch === "\n" || ch === "\f" || ch === "\r" || ch === " ";
|
|
247
|
+
/**
|
|
248
|
+
* Parse the attributes of a start tag whose name ends at `pos`.
|
|
249
|
+
* Follows the HTML tokenizer states "before attribute name" through
|
|
250
|
+
* "after attribute value (quoted)".
|
|
251
|
+
*/
|
|
252
|
+
function parseTagAttributes(html, pos) {
|
|
253
|
+
const attributes = new Map();
|
|
254
|
+
const len = html.length;
|
|
255
|
+
let i = pos;
|
|
256
|
+
while (i < len) {
|
|
257
|
+
// Before attribute name: skip whitespace and stray solidus.
|
|
258
|
+
const ch = html[i];
|
|
259
|
+
if (isTagWhitespace(ch) || ch === "/") {
|
|
260
|
+
i++;
|
|
261
|
+
continue;
|
|
262
|
+
}
|
|
263
|
+
if (ch === ">") {
|
|
264
|
+
return { attributes, end: i + 1 };
|
|
265
|
+
}
|
|
266
|
+
// Attribute name (a leading "=" is part of the name per spec).
|
|
267
|
+
let nameStart = i;
|
|
268
|
+
i++;
|
|
269
|
+
while (i < len) {
|
|
270
|
+
const c = html[i];
|
|
271
|
+
if (isTagWhitespace(c) || c === "/" || c === ">" || c === "=")
|
|
272
|
+
break;
|
|
273
|
+
i++;
|
|
274
|
+
}
|
|
275
|
+
const name = html.slice(nameStart, i).toLowerCase();
|
|
276
|
+
// After attribute name.
|
|
277
|
+
while (i < len && isTagWhitespace(html[i]))
|
|
278
|
+
i++;
|
|
279
|
+
let value = "";
|
|
280
|
+
if (i < len && html[i] === "=") {
|
|
281
|
+
i++;
|
|
282
|
+
while (i < len && isTagWhitespace(html[i]))
|
|
283
|
+
i++;
|
|
284
|
+
if (i >= len)
|
|
285
|
+
break;
|
|
286
|
+
const q = html[i];
|
|
287
|
+
if (q === '"' || q === "'") {
|
|
288
|
+
const close = html.indexOf(q, i + 1);
|
|
289
|
+
if (close === -1) {
|
|
290
|
+
i = len; // EOF inside quoted value
|
|
291
|
+
break;
|
|
292
|
+
}
|
|
293
|
+
value = html.slice(i + 1, close);
|
|
294
|
+
i = close + 1;
|
|
295
|
+
}
|
|
296
|
+
else if (q !== ">") {
|
|
297
|
+
const valueStart = i;
|
|
298
|
+
while (i < len && !isTagWhitespace(html[i]) && html[i] !== ">")
|
|
299
|
+
i++;
|
|
300
|
+
value = html.slice(valueStart, i);
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
if (!attributes.has(name))
|
|
304
|
+
attributes.set(name, value);
|
|
305
|
+
}
|
|
306
|
+
// EOF inside the tag: per spec the tag is dropped.
|
|
307
|
+
return { attributes, end: -1 };
|
|
308
|
+
}
|
|
309
|
+
const NAMED_REFS = {
|
|
310
|
+
amp: "&",
|
|
311
|
+
lt: "<",
|
|
312
|
+
gt: ">",
|
|
313
|
+
quot: '"',
|
|
314
|
+
apos: "'",
|
|
315
|
+
nbsp: " ",
|
|
316
|
+
};
|
|
317
|
+
/** Decode the character references that matter for alt text. */
|
|
318
|
+
function decodeBasicEntities(text) {
|
|
319
|
+
return text.replace(/&(#[xX][0-9a-fA-F]{1,6}|#[0-9]{1,7}|[a-zA-Z]+);/g, (match, ref) => {
|
|
320
|
+
if (ref[0] === "#") {
|
|
321
|
+
const code = ref[1] === "x" || ref[1] === "X"
|
|
322
|
+
? parseInt(ref.slice(2), 16)
|
|
323
|
+
: parseInt(ref.slice(1), 10);
|
|
324
|
+
if (!Number.isFinite(code) ||
|
|
325
|
+
code <= 0 ||
|
|
326
|
+
code > 0x10ffff ||
|
|
327
|
+
(code >= 0xd800 && code <= 0xdfff)) {
|
|
328
|
+
return "�";
|
|
329
|
+
}
|
|
330
|
+
return String.fromCodePoint(code);
|
|
331
|
+
}
|
|
332
|
+
const named = NAMED_REFS[ref.toLowerCase()];
|
|
333
|
+
return named ?? match;
|
|
334
|
+
});
|
|
335
|
+
}
|
|
336
|
+
export function escapeHtmlText(text) {
|
|
337
|
+
return text
|
|
338
|
+
.replace(/&/g, "&")
|
|
339
|
+
.replace(/</g, "<")
|
|
340
|
+
.replace(/>/g, ">")
|
|
341
|
+
.replace(/"/g, """)
|
|
342
|
+
.replace(/'/g, "'");
|
|
343
|
+
}
|
|
344
|
+
const SAFE_ATTR_NAME_RE = /^[a-zA-Z_:][-a-zA-Z0-9_:.]*$/;
|
|
345
|
+
/** Attributes that can make a converter fetch or reference an image. */
|
|
346
|
+
const IMAGE_SOURCE_ATTRS = new Set(["src", "href", "xlink:href"]);
|
|
347
|
+
const ALWAYS_DROPPED_ATTRS = new Set([
|
|
348
|
+
"srcset",
|
|
349
|
+
"data-src",
|
|
350
|
+
"data-srcset",
|
|
351
|
+
"lowsrc",
|
|
352
|
+
"dynsrc",
|
|
353
|
+
"longdesc",
|
|
354
|
+
"usemap",
|
|
355
|
+
"imagesizes",
|
|
356
|
+
"imagesrcset",
|
|
357
|
+
]);
|
|
358
|
+
function serializeTag(tagName, attributes) {
|
|
359
|
+
let out = `<${tagName}`;
|
|
360
|
+
for (const [name, value] of attributes) {
|
|
361
|
+
if (!SAFE_ATTR_NAME_RE.test(name))
|
|
362
|
+
continue;
|
|
363
|
+
// Keep character references as written (they decode identically when
|
|
364
|
+
// re-parsed inside a double-quoted value); only the quote must be escaped.
|
|
365
|
+
out += ` ${name}="${value.replace(/"/g, """)}"`;
|
|
366
|
+
}
|
|
367
|
+
return out + ">";
|
|
368
|
+
}
|
|
369
|
+
function sanitizeImageTag(tagName, attributes, options) {
|
|
370
|
+
const kept = new Map();
|
|
371
|
+
let hasSource = false;
|
|
372
|
+
for (const [name, value] of attributes) {
|
|
373
|
+
if (ALWAYS_DROPPED_ATTRS.has(name))
|
|
374
|
+
continue;
|
|
375
|
+
if (IMAGE_SOURCE_ATTRS.has(name)) {
|
|
376
|
+
const normalized = normalizeEmbeddableImageSrc(value, options);
|
|
377
|
+
if (normalized === null)
|
|
378
|
+
continue;
|
|
379
|
+
kept.set(name, normalized);
|
|
380
|
+
hasSource = true;
|
|
381
|
+
continue;
|
|
382
|
+
}
|
|
383
|
+
kept.set(name, value);
|
|
384
|
+
}
|
|
385
|
+
// A disallowed source is not "partially" kept: if any source attribute was
|
|
386
|
+
// rejected, the whole image is replaced (avoids parser disagreements over
|
|
387
|
+
// which of src/href wins).
|
|
388
|
+
const rejectedSource = [...attributes.keys()].some((name) => IMAGE_SOURCE_ATTRS.has(name) && !kept.has(name));
|
|
389
|
+
if (hasSource && !rejectedSource) {
|
|
390
|
+
return serializeTag(tagName, kept);
|
|
391
|
+
}
|
|
392
|
+
const alt = attributes.get("alt");
|
|
393
|
+
return alt ? escapeHtmlText(decodeBasicEntities(alt)) : "";
|
|
394
|
+
}
|
|
395
|
+
function sanitizeSourceTag(attributes) {
|
|
396
|
+
const kept = new Map();
|
|
397
|
+
for (const [name, value] of attributes) {
|
|
398
|
+
if (name === "src" || name === "srcset" || ALWAYS_DROPPED_ATTRS.has(name)) {
|
|
399
|
+
continue;
|
|
400
|
+
}
|
|
401
|
+
kept.set(name, value);
|
|
402
|
+
}
|
|
403
|
+
return serializeTag("source", kept);
|
|
404
|
+
}
|
|
405
|
+
const IMAGE_TAG_START_RE = /<(img|image|source)(?=[\t\n\f\r />]|$)/giy;
|
|
406
|
+
/**
|
|
407
|
+
* Rewrite every image-bearing tag in `html` so that only validated inline
|
|
408
|
+
* `data:` images remain. Non-embeddable images are replaced with their
|
|
409
|
+
* escaped `alt` text (or removed when there is none).
|
|
410
|
+
*/
|
|
411
|
+
export function sanitizeHtmlImages(html, options = {}) {
|
|
412
|
+
let out = "";
|
|
413
|
+
let last = 0;
|
|
414
|
+
let searchFrom = 0;
|
|
415
|
+
while (searchFrom < html.length) {
|
|
416
|
+
const lt = html.indexOf("<", searchFrom);
|
|
417
|
+
if (lt === -1)
|
|
418
|
+
break;
|
|
419
|
+
IMAGE_TAG_START_RE.lastIndex = lt;
|
|
420
|
+
const match = IMAGE_TAG_START_RE.exec(html);
|
|
421
|
+
if (!match) {
|
|
422
|
+
searchFrom = lt + 1;
|
|
423
|
+
continue;
|
|
424
|
+
}
|
|
425
|
+
const tagName = match[1].toLowerCase();
|
|
426
|
+
const parsed = parseTagAttributes(html, lt + match[0].length);
|
|
427
|
+
out += html.slice(last, lt);
|
|
428
|
+
if (parsed.end === -1) {
|
|
429
|
+
// Unterminated tag at end of input: drop it.
|
|
430
|
+
last = html.length;
|
|
431
|
+
break;
|
|
432
|
+
}
|
|
433
|
+
out +=
|
|
434
|
+
tagName === "source"
|
|
435
|
+
? sanitizeSourceTag(parsed.attributes)
|
|
436
|
+
: sanitizeImageTag(tagName, parsed.attributes, options);
|
|
437
|
+
last = parsed.end;
|
|
438
|
+
searchFrom = parsed.end;
|
|
439
|
+
}
|
|
440
|
+
return out + html.slice(last);
|
|
441
|
+
}
|
|
442
|
+
/**
|
|
443
|
+
* Defense in depth for the PDF path: walk a pdfmake content tree and replace
|
|
444
|
+
* any `image` node whose source is not an embeddable data URL (e.g. one that
|
|
445
|
+
* entered through html-to-pdfmake's `data-pdfmake` attribute) with empty text.
|
|
446
|
+
* Mutates and returns the tree.
|
|
447
|
+
*/
|
|
448
|
+
export function scrubPdfmakeImages(content, options = {}) {
|
|
449
|
+
const seen = new WeakSet();
|
|
450
|
+
const visit = (node) => {
|
|
451
|
+
if (Array.isArray(node)) {
|
|
452
|
+
if (seen.has(node))
|
|
453
|
+
return node;
|
|
454
|
+
seen.add(node);
|
|
455
|
+
for (let i = 0; i < node.length; i++)
|
|
456
|
+
node[i] = visit(node[i]);
|
|
457
|
+
return node;
|
|
458
|
+
}
|
|
459
|
+
if (node && typeof node === "object") {
|
|
460
|
+
if (seen.has(node))
|
|
461
|
+
return node;
|
|
462
|
+
seen.add(node);
|
|
463
|
+
const record = node;
|
|
464
|
+
if (Object.prototype.hasOwnProperty.call(record, "image")) {
|
|
465
|
+
const normalized = normalizeEmbeddableImageSrc(record.image, options);
|
|
466
|
+
if (normalized === null)
|
|
467
|
+
return { text: "" };
|
|
468
|
+
record.image = normalized;
|
|
469
|
+
}
|
|
470
|
+
for (const key of Object.keys(record)) {
|
|
471
|
+
record[key] = visit(record[key]);
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
return node;
|
|
475
|
+
};
|
|
476
|
+
return visit(content);
|
|
477
|
+
}
|
|
478
|
+
//# sourceMappingURL=html-image-sanitizer.js.map
|
|
@@ -9,6 +9,8 @@
|
|
|
9
9
|
* - html-to-docx: HTML → DOCX conversion
|
|
10
10
|
* - jsdom: DOM emulation for Node.js
|
|
11
11
|
*/
|
|
12
|
+
import { sanitizeHtmlImages, scrubPdfmakeImages, } from "./html-image-sanitizer.js";
|
|
13
|
+
import { PDF_GENERATION_TIMEOUT_MS } from "./limits.js";
|
|
12
14
|
// Lazy-loaded libraries (imported only when needed)
|
|
13
15
|
let pdfMake = null;
|
|
14
16
|
let pdfFonts = null;
|
|
@@ -61,10 +63,13 @@ function sanitizeHTMLForDOCX(html) {
|
|
|
61
63
|
// Common typographic characters
|
|
62
64
|
.replace(/—/g, "—") // Em dash
|
|
63
65
|
.replace(/–/g, "–") // En dash
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
.
|
|
67
|
-
.replace(
|
|
66
|
+
// Curly quotes only (written as escapes so they cannot be "normalized"
|
|
67
|
+
// into ASCII quotes, which previously turned every quoted attribute
|
|
68
|
+
// value - e.g. src="data:..." or style="..." - into garbage).
|
|
69
|
+
.replace(/“/g, "“") // Left double quote
|
|
70
|
+
.replace(/”/g, "”") // Right double quote
|
|
71
|
+
.replace(/‘/g, "‘") // Left single quote
|
|
72
|
+
.replace(/’/g, "’") // Right single quote
|
|
68
73
|
.replace(/…/g, "…") // Ellipsis
|
|
69
74
|
// Degree and other symbols
|
|
70
75
|
.replace(/°/g, "°") // Degree
|
|
@@ -124,8 +129,14 @@ export async function htmlToPDF(htmlContent, options = {}) {
|
|
|
124
129
|
}
|
|
125
130
|
// Create DOM window for html-to-pdfmake
|
|
126
131
|
const { window } = new jsdom("");
|
|
132
|
+
// Only validated inline data: images may reach pdfmake (VFO-13).
|
|
133
|
+
const imageOptions = {
|
|
134
|
+
...PDF_IMAGE_OPTIONS,
|
|
135
|
+
memo: new Map(),
|
|
136
|
+
};
|
|
137
|
+
const safeHTML = sanitizeHtmlImages(htmlContent, imageOptions);
|
|
127
138
|
// Convert HTML to PDFMake format with styling
|
|
128
|
-
const converted = htmlToPdfmake(
|
|
139
|
+
const converted = htmlToPdfmake(safeHTML, {
|
|
129
140
|
window,
|
|
130
141
|
defaultStyles: {
|
|
131
142
|
// Headings with colors
|
|
@@ -227,16 +238,66 @@ export async function htmlToPDF(htmlContent, options = {}) {
|
|
|
227
238
|
pageSize: "A4",
|
|
228
239
|
pageMargins: [40, 60, 40, 60],
|
|
229
240
|
};
|
|
230
|
-
//
|
|
241
|
+
// Defense in depth: html-to-pdfmake can also emit image nodes from
|
|
242
|
+
// `data-pdfmake` attributes, so validate the final tree as well.
|
|
243
|
+
scrubPdfmakeImages(converted, imageOptions);
|
|
244
|
+
return renderPdfBuffer(docDefinition, options.timeoutMs ?? PDF_GENERATION_TIMEOUT_MS);
|
|
245
|
+
}
|
|
246
|
+
/**
|
|
247
|
+
* pdfkit (used by pdfmake) can only embed PNG and JPEG images; anything else
|
|
248
|
+
* would make the whole PDF fail, so other formats degrade to alt text.
|
|
249
|
+
*/
|
|
250
|
+
const PDF_IMAGE_OPTIONS = {
|
|
251
|
+
allowedTypes: ["png", "jpeg"],
|
|
252
|
+
verifyPngForPdfkit: true,
|
|
253
|
+
};
|
|
254
|
+
const DOCX_IMAGE_OPTIONS = {};
|
|
255
|
+
function toError(error) {
|
|
256
|
+
if (error instanceof Error)
|
|
257
|
+
return error;
|
|
258
|
+
return new Error(typeof error === "string" ? error : String(error));
|
|
259
|
+
}
|
|
260
|
+
/**
|
|
261
|
+
* Render a pdfmake document definition to a Buffer.
|
|
262
|
+
*
|
|
263
|
+
* pdfmake 0.2.x's `getBuffer(cb)` builds the document inside an internal
|
|
264
|
+
* promise chain: if building fails (e.g. an invalid image), the callback is
|
|
265
|
+
* never called and the rejection is unhandled, which hangs the write and
|
|
266
|
+
* terminates Node (VFO-13). pdfmake exposes no promise API to catch that, so
|
|
267
|
+
* instead we use `getStream()` without a callback: it builds the PDFKit
|
|
268
|
+
* document synchronously (errors are thrown to us here) and skips pdfmake's
|
|
269
|
+
* URL resolver entirely (no network fetches of fonts/images). The stream is
|
|
270
|
+
* then drained with error handling and a hard timeout, so this promise always
|
|
271
|
+
* settles.
|
|
272
|
+
*/
|
|
273
|
+
function renderPdfBuffer(docDefinition, timeoutMs) {
|
|
231
274
|
return new Promise((resolve, reject) => {
|
|
275
|
+
let settled = false;
|
|
276
|
+
const timer = setTimeout(() => {
|
|
277
|
+
settle(new Error(`PDF generation timed out after ${timeoutMs} ms`));
|
|
278
|
+
}, timeoutMs);
|
|
279
|
+
timer.unref?.();
|
|
280
|
+
function settle(error, buffer) {
|
|
281
|
+
if (settled)
|
|
282
|
+
return;
|
|
283
|
+
settled = true;
|
|
284
|
+
clearTimeout(timer);
|
|
285
|
+
if (error)
|
|
286
|
+
reject(error);
|
|
287
|
+
else
|
|
288
|
+
resolve(buffer);
|
|
289
|
+
}
|
|
232
290
|
try {
|
|
233
291
|
const pdfDoc = pdfMake.createPdf(docDefinition);
|
|
234
|
-
pdfDoc.
|
|
235
|
-
|
|
236
|
-
|
|
292
|
+
const stream = pdfDoc.getStream();
|
|
293
|
+
const chunks = [];
|
|
294
|
+
stream.on("data", (chunk) => chunks.push(chunk));
|
|
295
|
+
stream.on("end", () => settle(null, Buffer.concat(chunks)));
|
|
296
|
+
stream.on("error", (error) => settle(toError(error)));
|
|
297
|
+
stream.end();
|
|
237
298
|
}
|
|
238
299
|
catch (error) {
|
|
239
|
-
|
|
300
|
+
settle(toError(error));
|
|
240
301
|
}
|
|
241
302
|
});
|
|
242
303
|
}
|
|
@@ -262,8 +323,11 @@ export async function htmlToDOCX(htmlContent, options = {}) {
|
|
|
262
323
|
const module = await import("@turbodocx/html-to-docx");
|
|
263
324
|
HTMLtoDOCX = module.default || module;
|
|
264
325
|
}
|
|
326
|
+
// Only validated inline data: images may reach html-to-docx, which would
|
|
327
|
+
// otherwise download http(s) image URLs (VFO-13).
|
|
328
|
+
const imageSafeHTML = sanitizeHtmlImages(htmlContent, DOCX_IMAGE_OPTIONS);
|
|
265
329
|
// Sanitize HTML to handle problematic Unicode characters
|
|
266
|
-
const sanitizedHTML = sanitizeHTMLForDOCX(
|
|
330
|
+
const sanitizedHTML = sanitizeHTMLForDOCX(imageSafeHTML);
|
|
267
331
|
// DOCX generation options
|
|
268
332
|
const docxOptions = {
|
|
269
333
|
title: options.title || "Document",
|