reamkit 1.26.0 → 1.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -4
- package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
- package/dist/esm/pdf-reader/annot-draw.js +374 -0
- package/dist/esm/pdf-reader/annots.d.ts +3 -1
- package/dist/esm/pdf-reader/annots.js +18 -4
- package/dist/esm/pdf-reader/ccitt.d.ts +18 -0
- package/dist/esm/pdf-reader/ccitt.js +70 -2
- package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
- package/dist/esm/pdf-reader/cie-color.js +112 -0
- package/dist/esm/pdf-reader/content.d.ts +47 -3
- package/dist/esm/pdf-reader/content.js +108 -13
- package/dist/esm/pdf-reader/display.d.ts +1 -1
- package/dist/esm/pdf-reader/display.js +21 -6
- package/dist/esm/pdf-reader/document.d.ts +6 -0
- package/dist/esm/pdf-reader/document.js +29 -1
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +12 -3
- package/dist/esm/pdf-reader/embedded-fonts.js +22 -3
- package/dist/esm/pdf-reader/flow-build.d.ts +12 -4
- package/dist/esm/pdf-reader/flow-build.js +102 -8
- package/dist/esm/pdf-reader/font.js +77 -16
- package/dist/esm/pdf-reader/function.d.ts +16 -0
- package/dist/esm/pdf-reader/function.js +414 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +8 -4
- package/dist/esm/pdf-reader/image-decode.js +192 -11
- package/dist/esm/pdf-reader/images.d.ts +12 -0
- package/dist/esm/pdf-reader/images.js +109 -9
- package/dist/esm/pdf-reader/jbig2.d.ts +23 -0
- package/dist/esm/pdf-reader/jbig2.js +23 -14
- package/dist/esm/pdf-reader/layout.d.ts +32 -0
- package/dist/esm/pdf-reader/layout.js +189 -40
- package/dist/esm/pdf-reader/lexer.d.ts +2 -0
- package/dist/esm/pdf-reader/lexer.js +4 -0
- package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
- package/dist/esm/pdf-reader/optional-content.js +93 -0
- package/dist/esm/pdf-reader/reader.d.ts +5 -2
- package/dist/esm/pdf-reader/reader.js +80 -7
- package/dist/esm/pdf-reader/shading.d.ts +60 -8
- package/dist/esm/pdf-reader/shading.js +124 -20
- package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
- package/dist/esm/pdf-reader/standard-metrics.js +18 -0
- package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
- package/dist/esm/pdf-reader/standard-widths.js +62 -0
- package/dist/esm/pdf-reader/tagged.js +204 -32
- package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
- package/dist/esm/pdf-reader/text-rules.js +112 -0
- package/dist/esm/pdf-reader/text.js +78 -4
- package/dist/esm/pdf-reader/vector.d.ts +5 -0
- package/dist/esm/pdf-reader/vector.js +39 -26
- package/dist/esm/word/docx-writer.js +73 -11
- package/package.json +1 -1
|
@@ -2,6 +2,7 @@ import { PDF_NULL, PdfHexString, PdfName, PdfStream } from "../pdf/objects.js";
|
|
|
2
2
|
import { encodePng } from "../core/png-encode.js";
|
|
3
3
|
import { lzwDecodeMsb } from "../core/lzw.js";
|
|
4
4
|
import { reversePredictor } from "./predictor.js";
|
|
5
|
+
import { labToSrgb } from "./cie-color.js";
|
|
5
6
|
import { decodeCcitt } from "./ccitt.js";
|
|
6
7
|
import { decodeJbig2 } from "./jbig2.js";
|
|
7
8
|
import { decodeJpeg } from "./jpeg.js";
|
|
@@ -16,19 +17,23 @@ var MAX_PIXELS = 4e7;
|
|
|
16
17
|
* Supports DeviceGray/RGB/CMYK, CalGray/CalRGB, ICCBased (by `/N`) and Indexed
|
|
17
18
|
* colour spaces; Flate, LZW (EP12), RunLength, ASCII85, ASCIIHex and CCITT
|
|
18
19
|
* Group 4 / Group 3 1-D fax (EP15) filters; PNG/TIFF predictors; bit depths
|
|
19
|
-
* 1/2/4/8/16;
|
|
20
|
-
*
|
|
21
|
-
* 2-D) return a typed failure so the
|
|
20
|
+
* 1/2/4/8/16; an `/SMask` folded in as the PNG alpha channel; and a stencil
|
|
21
|
+
* `/ImageMask` (§8.9.6.2) given back as RGBA in `fillHex`. Unsupported inputs
|
|
22
|
+
* (Separation/DeviceN/Lab, CCITT Group 3 2-D) return a typed failure so the
|
|
23
|
+
* caller records a loss.
|
|
22
24
|
*
|
|
25
|
+
* @param file The owning file.
|
|
26
|
+
* @param stream The image XObject, or an inline image wrapped as one.
|
|
27
|
+
* @param fillHex §8.9.6.2 — the non-stroking colour a stencil mask paints in.
|
|
23
28
|
* @returns The decoded image, or `{ ok: false }` with the loss severity and reason.
|
|
24
29
|
*/
|
|
25
|
-
function decodePdfImage(file, stream) {
|
|
30
|
+
function decodePdfImage(file, stream, fillHex) {
|
|
26
31
|
const d = stream.dict;
|
|
27
32
|
const width = intOf(file.get(d, "Width")) || intOf(file.get(d, "W"));
|
|
28
33
|
const height = intOf(file.get(d, "Height")) || intOf(file.get(d, "H"));
|
|
29
34
|
if (width <= 0 || height <= 0) return fail("dropped", "image with no dimensions");
|
|
30
35
|
if (width * height > MAX_PIXELS) return fail("dropped", "image too large to decode");
|
|
31
|
-
if (boolOf(d.get("ImageMask")) || boolOf(d.get("IM"))) return
|
|
36
|
+
if (boolOf(d.get("ImageMask")) || boolOf(d.get("IM"))) return stencil(file, stream, width, height, fillHex);
|
|
32
37
|
const filters = filterNames(file, d);
|
|
33
38
|
const last = filters[filters.length - 1];
|
|
34
39
|
if (last === "DCTDecode" || last === "DCT") {
|
|
@@ -116,13 +121,14 @@ function decodeCcittImage(file, stream, filters, width, height) {
|
|
|
116
121
|
byteAlign: parms ? boolOf(file.get(parms, "EncodedByteAlign")) : false
|
|
117
122
|
});
|
|
118
123
|
if (!packed) return "CCITT fax image not decoded (Group 3 2-D or malformed)";
|
|
124
|
+
const blackIs1 = parms ? boolOf(file.get(parms, "BlackIs1")) : false;
|
|
119
125
|
const rowBytes = columns + 7 >> 3;
|
|
120
126
|
const samples = new Uint8Array(width * height);
|
|
121
127
|
for (let y = 0; y < height; y++) {
|
|
122
128
|
const rowOff = y * rowBytes;
|
|
123
129
|
for (let x = 0; x < width; x++) {
|
|
124
|
-
const
|
|
125
|
-
samples[y * width + x] =
|
|
130
|
+
const bit = x < columns ? packed[rowOff + (x >> 3)] >> 7 - (x & 7) & 1 : 0;
|
|
131
|
+
samples[y * width + x] = bit === (blackIs1 ? 0 : 1) ? 0 : 255;
|
|
126
132
|
}
|
|
127
133
|
}
|
|
128
134
|
return {
|
|
@@ -130,6 +136,89 @@ function decodeCcittImage(file, stream, filters, width, height) {
|
|
|
130
136
|
samples
|
|
131
137
|
};
|
|
132
138
|
}
|
|
139
|
+
/**
|
|
140
|
+
* §8.9.6.2 — a stencil mask: one bit per pixel saying WHERE to paint, and the
|
|
141
|
+
* colour is whatever the page's non-stroking colour was at the `Do`.
|
|
142
|
+
*
|
|
143
|
+
* A sample of 0 paints and 1 leaves the page alone, which `/Decode [1 0]`
|
|
144
|
+
* swaps. Given back as an RGBA picture — the fill colour throughout, opaque
|
|
145
|
+
* where the stencil paints and clear where it does not — which is what the mask
|
|
146
|
+
* MEANS and what every format downstream can show.
|
|
147
|
+
* images_1bit_grayscale.pdf draws two of them and they were dropped entirely.
|
|
148
|
+
*/
|
|
149
|
+
function stencil(file, stream, width, height, fillHex) {
|
|
150
|
+
const d = stream.dict;
|
|
151
|
+
const filters = filterNames(file, d);
|
|
152
|
+
const last = filters[filters.length - 1];
|
|
153
|
+
const packed = last === "CCITTFaxDecode" || last === "CCF" ? ccittBits(file, stream, filters, height) : last === "JBIG2Decode" ? jbig2Bits(file, stream, filters, width, height) : decodeChain(file, stream, filters);
|
|
154
|
+
if (!packed) return fail("dropped", "stencil image mask not decoded");
|
|
155
|
+
const decode = file.resolve(d.get("Decode") ?? d.get("D") ?? PDF_NULL);
|
|
156
|
+
const paintBit = Array.isArray(decode) && file.resolve(decode[0] ?? PDF_NULL) === 1 ? 1 : 0;
|
|
157
|
+
const rgb = fillHex !== void 0 ? parseHex(fillHex) : [
|
|
158
|
+
0,
|
|
159
|
+
0,
|
|
160
|
+
0
|
|
161
|
+
];
|
|
162
|
+
const rowBytes = width + 7 >> 3;
|
|
163
|
+
if (packed.length < rowBytes * height) return fail("dropped", "stencil image mask truncated");
|
|
164
|
+
const samples = new Uint8Array(width * height * 4);
|
|
165
|
+
for (let y = 0; y < height; y++) for (let x = 0; x < width; x++) {
|
|
166
|
+
const bit = packed[y * rowBytes + (x >> 3)] >> 7 - (x & 7) & 1;
|
|
167
|
+
const at = (y * width + x) * 4;
|
|
168
|
+
samples[at] = rgb[0];
|
|
169
|
+
samples[at + 1] = rgb[1];
|
|
170
|
+
samples[at + 2] = rgb[2];
|
|
171
|
+
samples[at + 3] = bit === paintBit ? 255 : 0;
|
|
172
|
+
}
|
|
173
|
+
return {
|
|
174
|
+
ok: true,
|
|
175
|
+
bytes: encodePng(width, height, "rgba", samples),
|
|
176
|
+
format: "png",
|
|
177
|
+
widthPx: width,
|
|
178
|
+
heightPx: height
|
|
179
|
+
};
|
|
180
|
+
}
|
|
181
|
+
/** A 6-hex colour as its three bytes. */
|
|
182
|
+
function parseHex(hex) {
|
|
183
|
+
const v = Number.parseInt(hex, 16);
|
|
184
|
+
if (!Number.isFinite(v)) return [
|
|
185
|
+
0,
|
|
186
|
+
0,
|
|
187
|
+
0
|
|
188
|
+
];
|
|
189
|
+
return [
|
|
190
|
+
v >> 16 & 255,
|
|
191
|
+
v >> 8 & 255,
|
|
192
|
+
v & 255
|
|
193
|
+
];
|
|
194
|
+
}
|
|
195
|
+
/** A stencil's bits where the file coded them as fax (§7.4.6). */
|
|
196
|
+
function ccittBits(file, stream, filters, height) {
|
|
197
|
+
const parms = decodeParmsOf(file, stream.dict);
|
|
198
|
+
const columns = (parms ? intOf(file.get(parms, "Columns")) : 0) || 1728;
|
|
199
|
+
const packed = decodeCcitt(applyChainExceptLast(filters, stream.data), {
|
|
200
|
+
k: parms ? intOf(file.get(parms, "K")) : 0,
|
|
201
|
+
columns,
|
|
202
|
+
rows: height,
|
|
203
|
+
byteAlign: parms ? boolOf(file.get(parms, "EncodedByteAlign")) : false
|
|
204
|
+
});
|
|
205
|
+
if (!packed) return void 0;
|
|
206
|
+
if (parms ? boolOf(file.get(parms, "BlackIs1")) : false) return packed;
|
|
207
|
+
const flipped = new Uint8Array(packed.length);
|
|
208
|
+
for (let i = 0; i < packed.length; i++) flipped[i] = ~packed[i] & 255;
|
|
209
|
+
return flipped;
|
|
210
|
+
}
|
|
211
|
+
/** …or as JBIG2, which codes 1 as black and so paints on 1. */
|
|
212
|
+
function jbig2Bits(file, stream, filters, width, height) {
|
|
213
|
+
const parms = decodeParmsOf(file, stream.dict);
|
|
214
|
+
const globalsVal = parms ? file.get(parms, "JBIG2Globals") : void 0;
|
|
215
|
+
const globals = globalsVal instanceof PdfStream ? file.streamData(globalsVal) : void 0;
|
|
216
|
+
const packed = decodeJbig2(applyChainExceptLast(filters, stream.data), globals, width, height);
|
|
217
|
+
if (!packed) return void 0;
|
|
218
|
+
const flipped = new Uint8Array(packed.length);
|
|
219
|
+
for (let i = 0; i < packed.length; i++) flipped[i] = ~packed[i] & 255;
|
|
220
|
+
return flipped;
|
|
221
|
+
}
|
|
133
222
|
function resolveColorSpace(file, csVal) {
|
|
134
223
|
const cs = file.resolve(csVal ?? PDF_NULL);
|
|
135
224
|
if (cs instanceof PdfName) return namedColorSpace(cs.value);
|
|
@@ -178,6 +267,39 @@ function resolveColorSpace(file, csVal) {
|
|
|
178
267
|
lookup
|
|
179
268
|
};
|
|
180
269
|
}
|
|
270
|
+
if (tag === "Lab") {
|
|
271
|
+
const params = file.resolve(cs[1] ?? PDF_NULL);
|
|
272
|
+
const dict = params instanceof Map ? params : void 0;
|
|
273
|
+
const nums = (v) => {
|
|
274
|
+
const r = v !== void 0 ? file.resolve(v) : void 0;
|
|
275
|
+
return Array.isArray(r) ? r.map((x) => typeof file.resolve(x) === "number" ? file.resolve(x) : 0) : [];
|
|
276
|
+
};
|
|
277
|
+
const white = dict ? nums(file.get(dict, "WhitePoint")) : [];
|
|
278
|
+
if (white.length < 3 || !(white[1] > 0)) return void 0;
|
|
279
|
+
const range = dict ? nums(file.get(dict, "Range")) : [];
|
|
280
|
+
return {
|
|
281
|
+
kind: "lab",
|
|
282
|
+
components: 3,
|
|
283
|
+
lab: {
|
|
284
|
+
white: [
|
|
285
|
+
white[0],
|
|
286
|
+
white[1],
|
|
287
|
+
white[2]
|
|
288
|
+
],
|
|
289
|
+
range: range.length === 4 ? [
|
|
290
|
+
range[0],
|
|
291
|
+
range[1],
|
|
292
|
+
range[2],
|
|
293
|
+
range[3]
|
|
294
|
+
] : [
|
|
295
|
+
-100,
|
|
296
|
+
100,
|
|
297
|
+
-100,
|
|
298
|
+
100
|
|
299
|
+
]
|
|
300
|
+
}
|
|
301
|
+
};
|
|
302
|
+
}
|
|
181
303
|
}
|
|
182
304
|
function namedColorSpace(name) {
|
|
183
305
|
if (name === "DeviceGray" || name === "G" || name === "CalGray") return {
|
|
@@ -241,13 +363,48 @@ function toColor(cs, s, px, bpc, decode) {
|
|
|
241
363
|
samples: out
|
|
242
364
|
};
|
|
243
365
|
}
|
|
366
|
+
if (cs.kind === "lab") {
|
|
367
|
+
const lab = cs.lab;
|
|
368
|
+
const out = new Uint8Array(px * 3);
|
|
369
|
+
const span = (i) => decode && decode.length >= 6 ? [0, 1] : i === 0 ? [0, 100] : [lab.range[(i - 1) * 2], lab.range[(i - 1) * 2 + 1]];
|
|
370
|
+
for (let i = 0; i < px; i++) {
|
|
371
|
+
const comp = [
|
|
372
|
+
0,
|
|
373
|
+
1,
|
|
374
|
+
2
|
|
375
|
+
].map((k) => {
|
|
376
|
+
const t = c01(s[i * 3 + k], k);
|
|
377
|
+
const [lo, hi] = span(k);
|
|
378
|
+
return lo + t * (hi - lo);
|
|
379
|
+
});
|
|
380
|
+
const [r, g, b] = labToSrgb(lab.white, [
|
|
381
|
+
comp[0],
|
|
382
|
+
comp[1],
|
|
383
|
+
comp[2]
|
|
384
|
+
]);
|
|
385
|
+
out[i * 3] = Math.round(r * 255);
|
|
386
|
+
out[i * 3 + 1] = Math.round(g * 255);
|
|
387
|
+
out[i * 3 + 2] = Math.round(b * 255);
|
|
388
|
+
}
|
|
389
|
+
return {
|
|
390
|
+
color: "rgb",
|
|
391
|
+
samples: out
|
|
392
|
+
};
|
|
393
|
+
}
|
|
244
394
|
const base = cs.base;
|
|
245
395
|
const lookup = cs.lookup;
|
|
246
396
|
const hival = cs.hival ?? 255;
|
|
247
397
|
const bn = base.components;
|
|
398
|
+
const index = (v) => {
|
|
399
|
+
if (!decode || decode.length < 2) return Math.min(v, hival);
|
|
400
|
+
const lo = decode[0];
|
|
401
|
+
const hi = decode[1];
|
|
402
|
+
const at = maxv > 0 ? lo + v / maxv * (hi - lo) : lo;
|
|
403
|
+
return Math.min(Math.max(0, Math.round(at)), hival);
|
|
404
|
+
};
|
|
248
405
|
if (base.kind === "gray") {
|
|
249
406
|
const out = new Uint8Array(px);
|
|
250
|
-
for (let i = 0; i < px; i++) out[i] = lookup[
|
|
407
|
+
for (let i = 0; i < px; i++) out[i] = lookup[index(s[i]) * bn] ?? 0;
|
|
251
408
|
return {
|
|
252
409
|
color: "gray",
|
|
253
410
|
samples: out
|
|
@@ -255,7 +412,7 @@ function toColor(cs, s, px, bpc, decode) {
|
|
|
255
412
|
}
|
|
256
413
|
const out = new Uint8Array(px * 3);
|
|
257
414
|
for (let i = 0; i < px; i++) {
|
|
258
|
-
const [r, g, b] = paletteRgb(base, lookup,
|
|
415
|
+
const [r, g, b] = paletteRgb(base, lookup, index(s[i]) * bn);
|
|
259
416
|
out[i * 3] = r;
|
|
260
417
|
out[i * 3 + 1] = g;
|
|
261
418
|
out[i * 3 + 2] = b;
|
|
@@ -271,6 +428,20 @@ function paletteRgb(base, lookup, off) {
|
|
|
271
428
|
lookup[off + 1] ?? 0,
|
|
272
429
|
lookup[off + 2] ?? 0
|
|
273
430
|
];
|
|
431
|
+
if (base.kind === "lab" && base.lab) {
|
|
432
|
+
const r = base.lab.range;
|
|
433
|
+
const at = (i, lo, hi) => lo + (lookup[off + i] ?? 0) / 255 * (hi - lo);
|
|
434
|
+
const [sr, sg, sb] = labToSrgb(base.lab.white, [
|
|
435
|
+
at(0, 0, 100),
|
|
436
|
+
at(1, r[0], r[1]),
|
|
437
|
+
at(2, r[2], r[3])
|
|
438
|
+
]);
|
|
439
|
+
return [
|
|
440
|
+
Math.round(sr * 255),
|
|
441
|
+
Math.round(sg * 255),
|
|
442
|
+
Math.round(sb * 255)
|
|
443
|
+
];
|
|
444
|
+
}
|
|
274
445
|
if (base.kind === "cmyk") {
|
|
275
446
|
const c = (lookup[off] ?? 0) / 255;
|
|
276
447
|
const m = (lookup[off + 1] ?? 0) / 255;
|
|
@@ -406,15 +577,25 @@ function combineAlpha(color, alpha) {
|
|
|
406
577
|
};
|
|
407
578
|
}
|
|
408
579
|
function filterNames(file, d) {
|
|
409
|
-
const f = file.resolve(d.get("Filter") ?? PDF_NULL);
|
|
580
|
+
const f = file.resolve(d.get("Filter") ?? d.get("F") ?? PDF_NULL);
|
|
410
581
|
const arr = Array.isArray(f) ? f : [f];
|
|
411
582
|
const out = [];
|
|
412
583
|
for (const x of arr) {
|
|
413
584
|
const r = file.resolve(x);
|
|
414
|
-
if (r instanceof PdfName) out.push(r.value);
|
|
585
|
+
if (r instanceof PdfName) out.push(FILTER_ABBREVIATIONS[r.value] ?? r.value);
|
|
415
586
|
}
|
|
416
587
|
return out;
|
|
417
588
|
}
|
|
589
|
+
/** §8.9.7 Table 93 — the short name an inline image writes a filter under. */
|
|
590
|
+
var FILTER_ABBREVIATIONS = {
|
|
591
|
+
AHx: "ASCIIHexDecode",
|
|
592
|
+
A85: "ASCII85Decode",
|
|
593
|
+
LZW: "LZWDecode",
|
|
594
|
+
Fl: "FlateDecode",
|
|
595
|
+
RL: "RunLengthDecode",
|
|
596
|
+
CCF: "CCITTFaxDecode",
|
|
597
|
+
DCT: "DCTDecode"
|
|
598
|
+
};
|
|
418
599
|
function decodeChain(file, stream, filters) {
|
|
419
600
|
let data = stream.data;
|
|
420
601
|
let mayPredict = false;
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { ImageCrop } from '../core/document-model/types.js';
|
|
1
2
|
import { Loss } from '../core/ir/index.js';
|
|
2
3
|
import { PdfFile, PdfPage } from './document.js';
|
|
3
4
|
/**
|
|
@@ -15,6 +16,17 @@ export interface PdfImage {
|
|
|
15
16
|
/** Page-space lower-left corner (points, y-up). */
|
|
16
17
|
readonly x: number;
|
|
17
18
|
readonly y: number;
|
|
19
|
+
/**
|
|
20
|
+
* §8.9.5 — how far the CTM turns the picture, in degrees counter-clockwise.
|
|
21
|
+
* The box above is the picture's own, unturned; a turn spins it about its
|
|
22
|
+
* centre, which is what every downstream format does with one.
|
|
23
|
+
*/
|
|
24
|
+
readonly rotationDeg?: number;
|
|
25
|
+
/**
|
|
26
|
+
* §8.5.4 — the fraction of each of the picture's OWN edges a clip cut away,
|
|
27
|
+
* where one bounded it: `a:srcRect` in DrawingML terms. Absent is whole.
|
|
28
|
+
*/
|
|
29
|
+
readonly crop?: ImageCrop;
|
|
18
30
|
/** Enclosing marked-content id, if the placement was inside a `/Figure`. */
|
|
19
31
|
readonly mcid?: number;
|
|
20
32
|
/**
|
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
2
|
import { FEATURES } from "../core/ir/features.js";
|
|
3
3
|
import { interpretContent, multiply } from "./content.js";
|
|
4
|
-
import {
|
|
4
|
+
import { buildAlphaMap } from "./shading.js";
|
|
5
5
|
import { collectPageAppearances } from "./annots.js";
|
|
6
|
+
import { hiddenProperties, hiddenXObject } from "./optional-content.js";
|
|
7
|
+
import { decodePdfImage } from "./image-decode.js";
|
|
6
8
|
//#region src/pdf-reader/images.ts
|
|
7
9
|
var NO_FONTS = /* @__PURE__ */ new Map();
|
|
8
10
|
var MAX_FORM_DEPTH = 12;
|
|
@@ -20,6 +22,7 @@ function collectPageImages(file, page) {
|
|
|
20
22
|
const images = [];
|
|
21
23
|
const lossByDetail = /* @__PURE__ */ new Map();
|
|
22
24
|
const visiting = /* @__PURE__ */ new Set();
|
|
25
|
+
const alphaCache = /* @__PURE__ */ new Map();
|
|
23
26
|
const addLoss = (severity, detail) => {
|
|
24
27
|
if (!lossByDetail.has(detail)) lossByDetail.set(detail, {
|
|
25
28
|
severity,
|
|
@@ -30,7 +33,12 @@ function collectPageImages(file, page) {
|
|
|
30
33
|
const walk = (resources, content, baseCtm, depth, inheritedMcid, prefix) => {
|
|
31
34
|
const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
|
|
32
35
|
const xobjDict = xobjects instanceof Map ? xobjects : void 0;
|
|
33
|
-
|
|
36
|
+
let paints = alphaCache.get(resources);
|
|
37
|
+
if (!paints) {
|
|
38
|
+
paints = buildAlphaMap(file, resources);
|
|
39
|
+
alphaCache.set(resources, paints);
|
|
40
|
+
}
|
|
41
|
+
const result = interpretContent(content, NO_FONTS, baseCtm, void 0, paints, void 0, hiddenProperties(file, resources));
|
|
34
42
|
const patterns = resources ? file.get(resources, "Pattern") : PDF_NULL;
|
|
35
43
|
const patternDict = patterns instanceof Map ? patterns : void 0;
|
|
36
44
|
for (const vector of result.vectors) {
|
|
@@ -45,18 +53,32 @@ function collectPageImages(file, page) {
|
|
|
45
53
|
}
|
|
46
54
|
for (const placement of result.images) {
|
|
47
55
|
if (images.length >= MAX_IMAGES) return;
|
|
56
|
+
if (placement.inline) {
|
|
57
|
+
const decoded = decodePdfImage(file, new PdfStream(placement.inline.dict, placement.inline.data), placement.fillHex);
|
|
58
|
+
if (decoded.ok) {
|
|
59
|
+
images.push({
|
|
60
|
+
...geometry(placement.ctm, decoded, placement.mcid ?? inheritedMcid, placement.clip),
|
|
61
|
+
orderKey: [...prefix, placement.order]
|
|
62
|
+
});
|
|
63
|
+
if (decoded.degraded) addLoss("degraded", decoded.degraded);
|
|
64
|
+
} else addLoss(decoded.severity, decoded.detail);
|
|
65
|
+
continue;
|
|
66
|
+
}
|
|
48
67
|
const stream = xobjDict ? file.resolve(xobjDict.get(placement.name) ?? PDF_NULL) : PDF_NULL;
|
|
49
68
|
if (!(stream instanceof PdfStream)) continue;
|
|
69
|
+
if (hiddenXObject(file, stream)) continue;
|
|
50
70
|
const subtype = nameOf(file.get(stream.dict, "Subtype"));
|
|
51
71
|
const mcid = placement.mcid ?? inheritedMcid;
|
|
52
72
|
if (subtype === "Image") {
|
|
53
|
-
const decoded = decodePdfImage(file, stream);
|
|
73
|
+
const decoded = decodePdfImage(file, stream, placement.fillHex);
|
|
54
74
|
if (decoded.ok) {
|
|
55
75
|
images.push({
|
|
56
|
-
...geometry(placement.ctm, decoded, mcid),
|
|
76
|
+
...geometry(placement.ctm, decoded, mcid, placement.clip),
|
|
57
77
|
orderKey: [...prefix, placement.order]
|
|
58
78
|
});
|
|
59
79
|
if (decoded.degraded) addLoss("degraded", decoded.degraded);
|
|
80
|
+
if (placement.blend !== void 0) addLoss("degraded", `PDF blend mode /${placement.blend} is not performed; the picture is drawn over what it was to blend with`);
|
|
81
|
+
if (placement.masked === true) addLoss("degraded", "PDF soft mask (/SMask in the graphics state) is not applied; the picture is drawn at full opacity throughout");
|
|
60
82
|
} else addLoss(decoded.severity, decoded.detail);
|
|
61
83
|
} else if (subtype === "Form" && depth < MAX_FORM_DEPTH && !visiting.has(stream)) {
|
|
62
84
|
visiting.add(stream);
|
|
@@ -82,17 +104,95 @@ function collectPageImages(file, page) {
|
|
|
82
104
|
losses: [...lossByDetail.values()]
|
|
83
105
|
};
|
|
84
106
|
}
|
|
85
|
-
function geometry(ctm, decoded, mcid) {
|
|
107
|
+
function geometry(ctm, decoded, mcid, clip) {
|
|
108
|
+
const widthPt = Math.hypot(ctm[0], ctm[1]) || 1;
|
|
109
|
+
const heightPt = Math.hypot(ctm[2], ctm[3]) || 1;
|
|
110
|
+
const angle = Math.atan2(ctm[1], ctm[0]) * 180 / Math.PI;
|
|
111
|
+
const box = clippedUnitBox(ctm, clip);
|
|
112
|
+
const u = (box.u0 + box.u1) / 2;
|
|
113
|
+
const v = (box.v0 + box.v1) / 2;
|
|
114
|
+
const cx = ctm[0] * u + ctm[2] * v + ctm[4];
|
|
115
|
+
const cy = ctm[1] * u + ctm[3] * v + ctm[5];
|
|
116
|
+
const shownW = widthPt * (box.u1 - box.u0);
|
|
117
|
+
const shownH = heightPt * (box.v1 - box.v0);
|
|
118
|
+
const crop = box.u0 > 0 || box.v0 > 0 || box.u1 < 1 || box.v1 < 1 ? {
|
|
119
|
+
left: box.u0,
|
|
120
|
+
right: 1 - box.u1,
|
|
121
|
+
top: 1 - box.v1,
|
|
122
|
+
bottom: box.v0
|
|
123
|
+
} : void 0;
|
|
86
124
|
return {
|
|
87
125
|
bytes: decoded.bytes,
|
|
88
126
|
format: decoded.format,
|
|
89
|
-
widthPt:
|
|
90
|
-
heightPt:
|
|
91
|
-
x:
|
|
92
|
-
y:
|
|
127
|
+
widthPt: shownW,
|
|
128
|
+
heightPt: shownH,
|
|
129
|
+
x: cx - shownW / 2,
|
|
130
|
+
y: cy - shownH / 2,
|
|
131
|
+
...crop ? { crop } : {},
|
|
132
|
+
...Math.abs(angle) > .5 ? { rotationDeg: angle } : {},
|
|
93
133
|
...mcid !== void 0 ? { mcid } : {}
|
|
94
134
|
};
|
|
95
135
|
}
|
|
136
|
+
/** Below this the clip took a sliver off an edge and is not worth a crop. */
|
|
137
|
+
var CROPS = .01;
|
|
138
|
+
/**
|
|
139
|
+
* §8.5.4 — the part of the unit square a clip leaves showing, in the PICTURE's
|
|
140
|
+
* own axes.
|
|
141
|
+
*
|
|
142
|
+
* A clip is stated in the page's coordinates and a picture may be turned within
|
|
143
|
+
* them, so intersecting the two boxes as they stand crops along the wrong axes.
|
|
144
|
+
* Carried back through the placement matrix the clip lands in the unit square
|
|
145
|
+
* the image is drawn into, where a crop is what `a:srcRect` means: the fraction
|
|
146
|
+
* off each of the picture's own edges. image-rotated-black-white-ratio.pdf
|
|
147
|
+
* turns picture and clip together by thirty-one degrees, and in that space the
|
|
148
|
+
* clip is square on and takes the middle 53% of both sides.
|
|
149
|
+
*
|
|
150
|
+
* Where the clip is turned differently from the picture this bounds it rather
|
|
151
|
+
* than cutting it exactly — a rectangle is all `a:srcRect` can say.
|
|
152
|
+
*/
|
|
153
|
+
function clippedUnitBox(ctm, clip) {
|
|
154
|
+
const whole = {
|
|
155
|
+
u0: 0,
|
|
156
|
+
v0: 0,
|
|
157
|
+
u1: 1,
|
|
158
|
+
v1: 1
|
|
159
|
+
};
|
|
160
|
+
if (!clip) return whole;
|
|
161
|
+
const det = ctm[0] * ctm[3] - ctm[1] * ctm[2];
|
|
162
|
+
if (!Number.isFinite(det) || Math.abs(det) < 1e-9) return whole;
|
|
163
|
+
let u0 = Infinity;
|
|
164
|
+
let v0 = Infinity;
|
|
165
|
+
let u1 = -Infinity;
|
|
166
|
+
let v1 = -Infinity;
|
|
167
|
+
const add = (x, y) => {
|
|
168
|
+
const dx = x - ctm[4];
|
|
169
|
+
const dy = y - ctm[5];
|
|
170
|
+
const u = (ctm[3] * dx - ctm[2] * dy) / det;
|
|
171
|
+
const v = (ctm[0] * dy - ctm[1] * dx) / det;
|
|
172
|
+
u0 = Math.min(u0, u);
|
|
173
|
+
v0 = Math.min(v0, v);
|
|
174
|
+
u1 = Math.max(u1, u);
|
|
175
|
+
v1 = Math.max(v1, v);
|
|
176
|
+
};
|
|
177
|
+
for (const seg of clip.segs) {
|
|
178
|
+
if (seg.op === "close") continue;
|
|
179
|
+
if (seg.op === "cubic") {
|
|
180
|
+
add(seg.x1, seg.y1);
|
|
181
|
+
add(seg.x2, seg.y2);
|
|
182
|
+
}
|
|
183
|
+
add(seg.x, seg.y);
|
|
184
|
+
}
|
|
185
|
+
if (!Number.isFinite(u0) || !Number.isFinite(v0)) return whole;
|
|
186
|
+
const cut = {
|
|
187
|
+
u0: Math.min(1, Math.max(0, u0)),
|
|
188
|
+
v0: Math.min(1, Math.max(0, v0)),
|
|
189
|
+
u1: Math.min(1, Math.max(0, u1)),
|
|
190
|
+
v1: Math.min(1, Math.max(0, v1))
|
|
191
|
+
};
|
|
192
|
+
if (!(cut.u1 - cut.u0 > CROPS) || !(cut.v1 - cut.v0 > CROPS)) return whole;
|
|
193
|
+
if (cut.u1 - cut.u0 > 1 - CROPS && cut.v1 - cut.v0 > 1 - CROPS) return whole;
|
|
194
|
+
return cut;
|
|
195
|
+
}
|
|
96
196
|
function matrixOf(file, dict) {
|
|
97
197
|
const m = file.resolve(dict.get("Matrix") ?? PDF_NULL);
|
|
98
198
|
if (Array.isArray(m) && m.length >= 6 && m.every((v) => typeof v === "number")) return [
|
|
@@ -71,6 +71,29 @@ export declare function decodeGenericRegion(mq: MQDecoder, cx: Cx, width: number
|
|
|
71
71
|
* @param dy Likewise, vertically.
|
|
72
72
|
*/
|
|
73
73
|
export declare function decodeRefinement(mq: MQDecoder, cx: Cx, width: number, height: number, template: number, reference: Jbig2Bitmap, dx: number, dy: number, at: ReadonlyArray<At>, tpgron: boolean): Jbig2Bitmap;
|
|
74
|
+
/**
|
|
75
|
+
* §6.4.5 — draw a text region: a run of strips, each holding instances of the
|
|
76
|
+
* symbols in `symbols`, placed by running coordinates rather than absolute ones.
|
|
77
|
+
*
|
|
78
|
+
* This is what JBIG2 is for. A scanned page is not stored as pixels but as "the
|
|
79
|
+
* shape called 37, here; the shape called 12, four pixels on" — so the letter
|
|
80
|
+
* "e" costs its bitmap once and a few bits per occurrence after that.
|
|
81
|
+
*/
|
|
82
|
+
/**
|
|
83
|
+
* §6.4.5 Table 34 — SBSYMCODELEN, how many bits a text region spends naming
|
|
84
|
+
* one of its symbols.
|
|
85
|
+
*
|
|
86
|
+
* `ceil(log2(SBNUMSYMS))`, and for ONE symbol that is ZERO: there is nothing to
|
|
87
|
+
* choose, so no bits are read and the id is always that symbol.
|
|
88
|
+
* Rounded up to one — which the HUFFMAN side of the same table does need — the
|
|
89
|
+
* decoder takes a bit belonging to the next field and reads the id as 1, which
|
|
90
|
+
* is no symbol at all: bitmap-symbol-big-segmentid.pdf places one instance of
|
|
91
|
+
* one symbol in each of two regions, and both came back empty.
|
|
92
|
+
*
|
|
93
|
+
* @param symbols How many symbols the region has to choose between.
|
|
94
|
+
* @returns The number of bits an id takes.
|
|
95
|
+
*/
|
|
96
|
+
export declare function symbolCodeLength(symbols: number): number;
|
|
74
97
|
/**
|
|
75
98
|
* Decode an embedded JBIG2 image.
|
|
76
99
|
*
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { decodeCcitt } from "./ccitt.js";
|
|
1
|
+
import { decodeCcitt, decodeCcittPlanes } from "./ccitt.js";
|
|
2
2
|
//#region src/pdf-reader/jbig2.ts
|
|
3
3
|
function makeBitmap(width, height, fill = 0) {
|
|
4
4
|
const data = new Uint8Array(width * height);
|
|
@@ -1792,10 +1792,27 @@ var Corner = /* @__PURE__ */ function(Corner) {
|
|
|
1792
1792
|
* shape called 37, here; the shape called 12, four pixels on" — so the letter
|
|
1793
1793
|
* "e" costs its bitmap once and a few bits per occurrence after that.
|
|
1794
1794
|
*/
|
|
1795
|
+
/**
|
|
1796
|
+
* §6.4.5 Table 34 — SBSYMCODELEN, how many bits a text region spends naming
|
|
1797
|
+
* one of its symbols.
|
|
1798
|
+
*
|
|
1799
|
+
* `ceil(log2(SBNUMSYMS))`, and for ONE symbol that is ZERO: there is nothing to
|
|
1800
|
+
* choose, so no bits are read and the id is always that symbol.
|
|
1801
|
+
* Rounded up to one — which the HUFFMAN side of the same table does need — the
|
|
1802
|
+
* decoder takes a bit belonging to the next field and reads the id as 1, which
|
|
1803
|
+
* is no symbol at all: bitmap-symbol-big-segmentid.pdf places one instance of
|
|
1804
|
+
* one symbol in each of two regions, and both came back empty.
|
|
1805
|
+
*
|
|
1806
|
+
* @param symbols How many symbols the region has to choose between.
|
|
1807
|
+
* @returns The number of bits an id takes.
|
|
1808
|
+
*/
|
|
1809
|
+
function symbolCodeLength(symbols) {
|
|
1810
|
+
return Math.ceil(Math.log2(Math.max(symbols, 1)));
|
|
1811
|
+
}
|
|
1795
1812
|
function decodeTextRegion(mq, symbols, p, stripsLog) {
|
|
1796
1813
|
const bmp = makeBitmap(p.width, p.height, p.defPixel);
|
|
1797
1814
|
const strips = 1 << stripsLog;
|
|
1798
|
-
const codeLen =
|
|
1815
|
+
const codeLen = symbolCodeLength(symbols.length);
|
|
1799
1816
|
const iadt = new IntDecoder(mq);
|
|
1800
1817
|
const iafs = new IntDecoder(mq);
|
|
1801
1818
|
const iads = new IntDecoder(mq);
|
|
@@ -2164,21 +2181,13 @@ function decodePatternDictionary(data, mmr, template, pw, ph, grayMax) {
|
|
|
2164
2181
|
function decodeGrayScale(mq, data, mmr, template, at, width, height, bits, skip) {
|
|
2165
2182
|
const cx = newContexts(65536);
|
|
2166
2183
|
const planes = new Array(bits);
|
|
2167
|
-
|
|
2184
|
+
const packed = mmr ? decodeCcittPlanes(data, width, height, bits) : void 0;
|
|
2185
|
+
const rowBytes = width + 7 >> 3;
|
|
2168
2186
|
const decoder = mq ?? new MQDecoder(data);
|
|
2169
2187
|
for (let j = bits - 1; j >= 0; j--) if (mmr) {
|
|
2170
|
-
const
|
|
2171
|
-
k: -1,
|
|
2172
|
-
columns: width,
|
|
2173
|
-
rows: height,
|
|
2174
|
-
byteAlign: false
|
|
2175
|
-
});
|
|
2188
|
+
const bytes = packed?.[bits - 1 - j];
|
|
2176
2189
|
const plane = makeBitmap(width, height);
|
|
2177
|
-
if (
|
|
2178
|
-
const rowBytes = width + 7 >> 3;
|
|
2179
|
-
for (let y = 0; y < height; y++) for (let x = 0; x < width; x++) plane.data[y * width + x] = packed[y * rowBytes + (x >> 3)] >> 7 - (x & 7) & 1;
|
|
2180
|
-
mmrOffset += packed.length;
|
|
2181
|
-
}
|
|
2190
|
+
if (bytes) for (let y = 0; y < height; y++) for (let x = 0; x < width; x++) plane.data[y * width + x] = bytes[y * rowBytes + (x >> 3)] >> 7 - (x & 7) & 1;
|
|
2182
2191
|
planes[j] = plane;
|
|
2183
2192
|
} else planes[j] = decodeGenericRegion(decoder, cx, width, height, template, at, false, skip);
|
|
2184
2193
|
const values = new Array(width * height).fill(0);
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { PdfFile } from './document.js';
|
|
2
2
|
import { Reconstruction } from './flow-build.js';
|
|
3
|
+
/** §9.10.2 — a glyph the face maps to no character (see `./font`). */
|
|
4
|
+
export declare const UNMAPPED = "\uFFFD";
|
|
3
5
|
/**
|
|
4
6
|
* Heuristically reconstruct an untagged PDF into a {@link Reconstruction}
|
|
5
7
|
* (E-PDF EP4). With no structure tree there is only positioned content, so
|
|
@@ -16,3 +18,33 @@ import { Reconstruction } from './flow-build.js';
|
|
|
16
18
|
* @returns The reconstructed {@link FlowDoc} plus any read-time losses.
|
|
17
19
|
*/
|
|
18
20
|
export declare function reconstructByLayout(file: PdfFile, mode?: 'flow' | 'positional'): Reconstruction;
|
|
21
|
+
/**
|
|
22
|
+
* Whether a line ENDED a paragraph, rather than wrapping into the next.
|
|
23
|
+
*
|
|
24
|
+
* Leading alone cannot tell the two apart: five labels stacked at 15pt with a
|
|
25
|
+
* 12pt face look exactly like five wrapped lines, and alphatrans.pdf's five are
|
|
26
|
+
* read as one paragraph and re-wrapped into two. But a wrapping engine pulls
|
|
27
|
+
* the next word UP — so a line that stops well short of the measure stopped
|
|
28
|
+
* because its author stopped it, and the line after it begins something new.
|
|
29
|
+
* The same rule separates two paragraphs set with no extra space between them,
|
|
30
|
+
* which used to run together for the same reason.
|
|
31
|
+
*
|
|
32
|
+
* Only where both lines start at the same edge. Where they do not, the block is
|
|
33
|
+
* placed rather than set — a centred title's every line is short of the measure
|
|
34
|
+
* and none of them ends anything.
|
|
35
|
+
*
|
|
36
|
+
* @param prev The line before: where it starts, how wide it is, its face.
|
|
37
|
+
* @param next The line after — only where it starts matters.
|
|
38
|
+
* @param column The measure both were set in, when it is known.
|
|
39
|
+
* @returns Whether the first line ended a paragraph.
|
|
40
|
+
*/
|
|
41
|
+
export declare function endedParagraph(prev: {
|
|
42
|
+
x: number;
|
|
43
|
+
width: number;
|
|
44
|
+
fontSize: number;
|
|
45
|
+
}, next: {
|
|
46
|
+
x: number;
|
|
47
|
+
}, column: {
|
|
48
|
+
left: number;
|
|
49
|
+
right: number;
|
|
50
|
+
} | undefined): boolean;
|