reamkit 1.28.0 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/core/converter/ream.d.ts +8 -0
- package/dist/esm/core/document-model/types.d.ts +2 -0
- package/dist/esm/core/drawingml/shape-render.js +13 -1
- package/dist/esm/core/fonts/index.d.ts +1 -1
- package/dist/esm/core/fonts/remote-fonts.d.ts +8 -0
- package/dist/esm/core/fonts/remote-fonts.js +100 -17
- package/dist/esm/core/ir/flow.d.ts +17 -0
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/pdf-reader/annot-draw.js +93 -1
- package/dist/esm/pdf-reader/annots.d.ts +18 -0
- package/dist/esm/pdf-reader/annots.js +86 -9
- package/dist/esm/pdf-reader/cmap.js +13 -3
- package/dist/esm/pdf-reader/content.d.ts +32 -4
- package/dist/esm/pdf-reader/content.js +177 -12
- package/dist/esm/pdf-reader/display.js +16 -0
- package/dist/esm/pdf-reader/document.d.ts +8 -0
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +24 -0
- package/dist/esm/pdf-reader/embedded-fonts.js +48 -9
- package/dist/esm/pdf-reader/flow-build.d.ts +101 -5
- package/dist/esm/pdf-reader/flow-build.js +219 -25
- package/dist/esm/pdf-reader/font.d.ts +27 -1
- package/dist/esm/pdf-reader/font.js +313 -29
- package/dist/esm/pdf-reader/glyf-outline.js +13 -2
- package/dist/esm/pdf-reader/glyph-shapes.d.ts +18 -0
- package/dist/esm/pdf-reader/glyph-shapes.js +57 -0
- package/dist/esm/pdf-reader/image-decode.js +70 -4
- package/dist/esm/pdf-reader/images.d.ts +5 -0
- package/dist/esm/pdf-reader/images.js +4 -2
- package/dist/esm/pdf-reader/jbig2.d.ts +40 -1
- package/dist/esm/pdf-reader/jbig2.js +78 -16
- package/dist/esm/pdf-reader/jpeg.d.ts +6 -3
- package/dist/esm/pdf-reader/jpeg.js +21 -1
- package/dist/esm/pdf-reader/layout.d.ts +26 -2
- package/dist/esm/pdf-reader/layout.js +974 -55
- package/dist/esm/pdf-reader/lexer.d.ts +10 -0
- package/dist/esm/pdf-reader/lexer.js +17 -0
- package/dist/esm/pdf-reader/pattern-tint.d.ts +11 -1
- package/dist/esm/pdf-reader/pattern-tint.js +21 -3
- package/dist/esm/pdf-reader/regions.d.ts +25 -0
- package/dist/esm/pdf-reader/regions.js +167 -0
- package/dist/esm/pdf-reader/shading.d.ts +58 -2
- package/dist/esm/pdf-reader/shading.js +181 -11
- package/dist/esm/pdf-reader/struct-tree.js +112 -8
- package/dist/esm/pdf-reader/tagged.js +27 -7
- package/dist/esm/pdf-reader/text-rules.js +1 -1
- package/dist/esm/pdf-reader/text.js +9 -14
- package/dist/esm/pdf-reader/vector.d.ts +10 -2
- package/dist/esm/pdf-reader/vector.js +35 -5
- package/dist/esm/word/docx-writer.js +156 -39
- package/dist/esm/word/drawing-parser.js +7 -2
- package/dist/esm/word/paragraph-properties.js +2 -0
- package/package.json +5 -3
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
import { PDF_NULL, PdfHexString, PdfName } from "../pdf/objects.js";
|
|
2
|
+
import { bidiClass } from "../core/bidi/char-types.js";
|
|
3
|
+
import { reorderVisual } from "../core/bidi/algorithm.js";
|
|
4
|
+
import { analyzeString, hasBidiCharacters } from "../core/bidi/index.js";
|
|
2
5
|
import { Lexer } from "./lexer.js";
|
|
3
|
-
import { cmykHex, grayHex, rgbHex, spaceColor } from "./shading.js";
|
|
6
|
+
import { cmykHex, dashLengths, grayHex, rgbHex, spaceColor } from "./shading.js";
|
|
4
7
|
//#region src/pdf-reader/content.ts
|
|
5
8
|
/**
|
|
6
9
|
* A painted path emitted by the path-painting operators (§8.5.3), captured in
|
|
@@ -35,6 +38,18 @@ function pathBox(segs) {
|
|
|
35
38
|
}
|
|
36
39
|
/** The area of a box, for choosing the smaller of two clip regions. */
|
|
37
40
|
var area = (b) => Math.max(0, b.maxX - b.minX) * Math.max(0, b.maxY - b.minY);
|
|
41
|
+
/**
|
|
42
|
+
* §9.10 — whether the font says what character a glyph stands for: text that is
|
|
43
|
+
* neither empty nor the replacement character a reader writes where nothing is
|
|
44
|
+
* stated.
|
|
45
|
+
*
|
|
46
|
+
* @param text What the font's decode gave for the code.
|
|
47
|
+
* @returns Whether a flowing reading can write it as a letter.
|
|
48
|
+
*/
|
|
49
|
+
function statesACharacter(text) {
|
|
50
|
+
const trimmed = text.trim();
|
|
51
|
+
return trimmed.length > 0 && !trimmed.includes("�");
|
|
52
|
+
}
|
|
38
53
|
/** The identity {@link Matrix} (no transform). */
|
|
39
54
|
var IDENTITY = [
|
|
40
55
|
1,
|
|
@@ -89,9 +104,13 @@ function initialState() {
|
|
|
89
104
|
fillColor: "000000",
|
|
90
105
|
strokeColor: "000000",
|
|
91
106
|
lineWidth: 1,
|
|
107
|
+
dash: [],
|
|
108
|
+
lineCap: 0,
|
|
92
109
|
fillGradient: void 0,
|
|
93
110
|
fillPattern: void 0,
|
|
111
|
+
fillPatternPaint: void 0,
|
|
94
112
|
fillAlpha: 1,
|
|
113
|
+
strokeAlpha: 1,
|
|
95
114
|
fillDarkens: false,
|
|
96
115
|
blendMode: void 0,
|
|
97
116
|
softMask: false,
|
|
@@ -217,6 +236,25 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
217
236
|
y
|
|
218
237
|
});
|
|
219
238
|
};
|
|
239
|
+
const currentPoint = () => {
|
|
240
|
+
for (let i = path.length - 1; i >= 0; i--) {
|
|
241
|
+
const seg = path[i];
|
|
242
|
+
if (seg.op === "close") {
|
|
243
|
+
for (let k = i - 1; k >= 0; k--) {
|
|
244
|
+
const start = path[k];
|
|
245
|
+
if (start.op === "move") return {
|
|
246
|
+
x: start.x,
|
|
247
|
+
y: start.y
|
|
248
|
+
};
|
|
249
|
+
}
|
|
250
|
+
return;
|
|
251
|
+
}
|
|
252
|
+
return {
|
|
253
|
+
x: seg.x,
|
|
254
|
+
y: seg.y
|
|
255
|
+
};
|
|
256
|
+
}
|
|
257
|
+
};
|
|
220
258
|
const rectTo = (x, y, w, h) => {
|
|
221
259
|
moveTo(x, y);
|
|
222
260
|
lineTo(x + w, y);
|
|
@@ -229,6 +267,11 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
229
267
|
const scale = Math.sqrt(Math.abs(m[0] * m[3] - m[1] * m[2])) || 1;
|
|
230
268
|
return state.lineWidth * scale;
|
|
231
269
|
};
|
|
270
|
+
const ctmDash = () => {
|
|
271
|
+
const m = state.ctm;
|
|
272
|
+
const scale = Math.sqrt(Math.abs(m[0] * m[3] - m[1] * m[2])) || 1;
|
|
273
|
+
return state.dash.map((n) => n * scale);
|
|
274
|
+
};
|
|
232
275
|
const paintPath = (fill, stroke) => {
|
|
233
276
|
if (path.length >= 2 && (fill || stroke) && visible()) {
|
|
234
277
|
const mcid = mcStack.length > 0 ? mcStack[mcStack.length - 1] : void 0;
|
|
@@ -237,6 +280,7 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
237
280
|
segs: path,
|
|
238
281
|
...state.clip ? { clip: state.clip } : {},
|
|
239
282
|
...fill && state.fillPattern !== void 0 ? { patternName: state.fillPattern } : {},
|
|
283
|
+
...fill && state.fillPattern !== void 0 && state.fillPatternPaint !== void 0 ? { patternPaint: state.fillPatternPaint } : {},
|
|
240
284
|
...fill ? { fillHex: state.fillColor } : {},
|
|
241
285
|
...fill && state.fillAlpha < 1 ? { alpha: state.fillAlpha } : {},
|
|
242
286
|
...fill && state.fillDarkens ? { darkens: true } : {},
|
|
@@ -247,6 +291,10 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
247
291
|
strokeHex: state.strokeColor,
|
|
248
292
|
lineWidth: ctmLineWidth()
|
|
249
293
|
} : {},
|
|
294
|
+
...stroke && state.dash.length > 0 ? { dash: ctmDash() } : {},
|
|
295
|
+
...stroke && state.lineCap === 1 ? { cap: "round" } : {},
|
|
296
|
+
...stroke && state.lineCap === 2 ? { cap: "square" } : {},
|
|
297
|
+
...stroke && state.strokeAlpha < 1 ? { strokeAlpha: state.strokeAlpha } : {},
|
|
250
298
|
...mcid !== void 0 ? { mcid } : {}
|
|
251
299
|
});
|
|
252
300
|
}
|
|
@@ -284,7 +332,8 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
284
332
|
stream,
|
|
285
333
|
resources: type3.resources,
|
|
286
334
|
ctm: multiply(type3.matrix, multiply(scale, multiply(tm, state.ctm))),
|
|
287
|
-
order: paintOrder
|
|
335
|
+
order: paintOrder++,
|
|
336
|
+
readable: statesACharacter(state.font.decode([code]))
|
|
288
337
|
});
|
|
289
338
|
}
|
|
290
339
|
}
|
|
@@ -506,6 +555,8 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
506
555
|
if (named !== void 0) {
|
|
507
556
|
state.fillGradient = shadings.get(named);
|
|
508
557
|
state.fillPattern = state.fillGradient ? void 0 : named;
|
|
558
|
+
const components = operands.filter((o) => typeof o === "number");
|
|
559
|
+
state.fillPatternPaint = components.length > 0 ? spaceColor(components, void 0) : void 0;
|
|
509
560
|
break;
|
|
510
561
|
}
|
|
511
562
|
const hex = colorOfOperands(operands, state.fillSpace);
|
|
@@ -518,7 +569,12 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
518
569
|
}
|
|
519
570
|
case "SCN":
|
|
520
571
|
case "SC": {
|
|
521
|
-
|
|
572
|
+
const last = operands[operands.length - 1];
|
|
573
|
+
if (last instanceof PdfName) {
|
|
574
|
+
const sweep = shadings.get(last.value);
|
|
575
|
+
if (sweep) state.strokeColor = midGradient(sweep);
|
|
576
|
+
break;
|
|
577
|
+
}
|
|
522
578
|
const hex = colorOfOperands(operands, state.strokeSpace);
|
|
523
579
|
if (hex !== void 0) state.strokeColor = hex;
|
|
524
580
|
break;
|
|
@@ -535,6 +591,14 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
535
591
|
case "w":
|
|
536
592
|
state.lineWidth = num(0);
|
|
537
593
|
break;
|
|
594
|
+
case "d":
|
|
595
|
+
state.dash = dashLengths(operands[0]);
|
|
596
|
+
break;
|
|
597
|
+
case "J": {
|
|
598
|
+
const cap = num(0);
|
|
599
|
+
if (cap === 0 || cap === 1 || cap === 2) state.lineCap = cap;
|
|
600
|
+
break;
|
|
601
|
+
}
|
|
538
602
|
case "m":
|
|
539
603
|
moveTo(num(0), num(1));
|
|
540
604
|
break;
|
|
@@ -544,6 +608,26 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
544
608
|
case "c":
|
|
545
609
|
curveTo(num(0), num(1), num(2), num(3), num(4), num(5));
|
|
546
610
|
break;
|
|
611
|
+
case "v": {
|
|
612
|
+
const at = currentPoint();
|
|
613
|
+
if (at) {
|
|
614
|
+
const [x2, y2] = toPage(num(0), num(1));
|
|
615
|
+
const [x, y] = toPage(num(2), num(3));
|
|
616
|
+
path.push({
|
|
617
|
+
op: "cubic",
|
|
618
|
+
x1: at.x,
|
|
619
|
+
y1: at.y,
|
|
620
|
+
x2,
|
|
621
|
+
y2,
|
|
622
|
+
x,
|
|
623
|
+
y
|
|
624
|
+
});
|
|
625
|
+
}
|
|
626
|
+
break;
|
|
627
|
+
}
|
|
628
|
+
case "y":
|
|
629
|
+
curveTo(num(0), num(1), num(2), num(3), num(2), num(3));
|
|
630
|
+
break;
|
|
547
631
|
case "re":
|
|
548
632
|
rectTo(num(0), num(1), num(2), num(3));
|
|
549
633
|
break;
|
|
@@ -597,6 +681,10 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
597
681
|
state.blendMode = paint.blend;
|
|
598
682
|
}
|
|
599
683
|
if (paint.masked !== void 0) state.softMask = paint.masked;
|
|
684
|
+
if (paint.lineWidth !== void 0) state.lineWidth = paint.lineWidth;
|
|
685
|
+
if (paint.dash !== void 0) state.dash = paint.dash;
|
|
686
|
+
if (paint.lineCap !== void 0) state.lineCap = paint.lineCap;
|
|
687
|
+
if (paint.strokeAlpha !== void 0) state.strokeAlpha = paint.strokeAlpha;
|
|
600
688
|
}
|
|
601
689
|
break;
|
|
602
690
|
}
|
|
@@ -700,12 +788,27 @@ function strokesText(mode) {
|
|
|
700
788
|
* ArabicCIDTrueType.pdf came out mirrored for exactly that reason: the reader
|
|
701
789
|
* passed visual order through and the layout reversed it a second time.
|
|
702
790
|
*
|
|
703
|
-
* A run
|
|
704
|
-
* number inside an Arabic sentence
|
|
705
|
-
*
|
|
791
|
+
* A run wholly right-to-left is simply reversed. A MIXED one — a full stop or
|
|
792
|
+
* a number inside an Arabic sentence — is put through the bidi algorithm's
|
|
793
|
+
* reordering (UAX #9 L2), which for the runs a line holds undoes itself: the
|
|
794
|
+
* painted order reordered is the reading order. Left alone, a note of
|
|
795
|
+
* freetext_no_appearance.pdf came back with every line that held a full stop
|
|
796
|
+
* reading back to front. The direction it is reordered in is the one most of
|
|
797
|
+
* its letters run in — the painted order says nothing about which comes first.
|
|
706
798
|
*/
|
|
707
799
|
function logicalOrder(text) {
|
|
708
|
-
|
|
800
|
+
if (isRightToLeft(text)) return [...text].reverse().join("");
|
|
801
|
+
if (!hasBidiCharacters(text)) return text;
|
|
802
|
+
const chars = [...text];
|
|
803
|
+
let rtl = 0;
|
|
804
|
+
let strong = 0;
|
|
805
|
+
for (const ch of chars) {
|
|
806
|
+
const type = bidiClass(ch.codePointAt(0) ?? 0);
|
|
807
|
+
if (type === "R" || type === "AL") rtl++;
|
|
808
|
+
if (type === "R" || type === "AL" || type === "L") strong++;
|
|
809
|
+
}
|
|
810
|
+
const { levels } = analyzeString(text, rtl * 2 >= strong ? "rtl" : "ltr");
|
|
811
|
+
return reorderVisual(levels).map((i) => chars[i] ?? "").join("");
|
|
709
812
|
}
|
|
710
813
|
/**
|
|
711
814
|
* Whether a string is wholly right-to-left: at least one letter of an RTL
|
|
@@ -824,8 +927,8 @@ function readValue(lexer) {
|
|
|
824
927
|
* The bytes are binary and may hold `EI` themselves, so the end is found by
|
|
825
928
|
* MEASURING where the dictionary says how much there is — an unfiltered image
|
|
826
929
|
* is exactly `ceil(W · BPC · components / 8) · H` bytes — and only searched for
|
|
827
|
-
* where a filter makes the length unknowable
|
|
828
|
-
* two of them and both were skipped over.
|
|
930
|
+
* where a filter makes the length unknowable (see {@link filteredEnd}).
|
|
931
|
+
* images_1bit_grayscale.pdf draws two of them and both were skipped over.
|
|
829
932
|
*/
|
|
830
933
|
function readInlineImage(lexer) {
|
|
831
934
|
const dict = /* @__PURE__ */ new Map();
|
|
@@ -835,24 +938,86 @@ function readInlineImage(lexer) {
|
|
|
835
938
|
if (tok.kind === "keyword" && tok.value === "ID") break;
|
|
836
939
|
if (tok.kind === "name") dict.set(tok.value, readValue(lexer));
|
|
837
940
|
}
|
|
838
|
-
const start = lexer.pos + 1;
|
|
839
941
|
const measured = inlineLength(dict);
|
|
942
|
+
const crlf = lexer.byteAt(lexer.pos) === 13 && lexer.byteAt(lexer.pos + 1) === 10;
|
|
943
|
+
const start = lexer.pos + (crlf && measured === void 0 ? 2 : 1);
|
|
840
944
|
let end = measured !== void 0 ? start + measured : -1;
|
|
841
945
|
if (end < 0 || end > lexer.length) {
|
|
842
|
-
end = lexer
|
|
946
|
+
end = filteredEnd(lexer, start, dict);
|
|
843
947
|
if (end < 0) {
|
|
844
948
|
lexer.pos = lexer.length;
|
|
845
949
|
return;
|
|
846
950
|
}
|
|
847
951
|
}
|
|
848
952
|
const data = lexer.slice(start, Math.min(end, lexer.length));
|
|
849
|
-
const ei = lexer.
|
|
953
|
+
const ei = lexer.indexOfKeyword("EI", end);
|
|
850
954
|
lexer.pos = ei < 0 ? lexer.length : ei + 2;
|
|
851
955
|
return {
|
|
852
956
|
dict,
|
|
853
957
|
data
|
|
854
958
|
};
|
|
855
959
|
}
|
|
960
|
+
/**
|
|
961
|
+
* §8.9.7 — where a FILTERED inline image's bytes end, which the dictionary
|
|
962
|
+
* cannot say.
|
|
963
|
+
*
|
|
964
|
+
* Searched for as the first two bytes `EI`, the end fell inside the picture
|
|
965
|
+
* wherever its own bytes spelled them: a JPEG is dense with them, and
|
|
966
|
+
* bug1065245.pdf's three banners were each cut off a few hundred bytes in,
|
|
967
|
+
* would not decode, and were dropped — a blank sheet. So the data is read to
|
|
968
|
+
* where its OWN encoding ends: a JPEG at its end-of-image marker, hex text at
|
|
969
|
+
* `>`, base-85 at `~>`. Anything else ends at the first `EI` standing as a
|
|
970
|
+
* word of its own.
|
|
971
|
+
*/
|
|
972
|
+
function filteredEnd(lexer, start, dict) {
|
|
973
|
+
const f = dict.get("F") ?? dict.get("Filter");
|
|
974
|
+
const first = Array.isArray(f) ? f[0] : f;
|
|
975
|
+
const filter = first instanceof PdfName ? first.value : "";
|
|
976
|
+
if (filter === "DCT" || filter === "DCTDecode") {
|
|
977
|
+
const eoi = jpegEnd(lexer, start);
|
|
978
|
+
if (eoi !== void 0) return eoi;
|
|
979
|
+
} else if (filter === "AHx" || filter === "ASCIIHexDecode") {
|
|
980
|
+
const gt = lexer.indexOfAscii(">", start);
|
|
981
|
+
if (gt >= 0) return gt + 1;
|
|
982
|
+
} else if (filter === "A85" || filter === "ASCII85Decode") {
|
|
983
|
+
const tilde = lexer.indexOfAscii("~>", start);
|
|
984
|
+
if (tilde >= 0) return tilde + 2;
|
|
985
|
+
}
|
|
986
|
+
return lexer.indexOfKeyword("EI", start);
|
|
987
|
+
}
|
|
988
|
+
/**
|
|
989
|
+
* Where a JPEG that starts at `start` ends: past its end-of-image marker,
|
|
990
|
+
* found by walking its segments — a length-prefixed marker segment is stepped
|
|
991
|
+
* over whole, and the coded data after a start-of-scan is read to the next
|
|
992
|
+
* marker, where `FF` stands before anything but a stuffed `00` or a restart.
|
|
993
|
+
*/
|
|
994
|
+
function jpegEnd(lexer, start) {
|
|
995
|
+
if (lexer.byteAt(start) !== 255 || lexer.byteAt(start + 1) !== 216) return void 0;
|
|
996
|
+
let i = start + 2;
|
|
997
|
+
for (;;) {
|
|
998
|
+
if (lexer.byteAt(i) !== 255) return void 0;
|
|
999
|
+
let marker = lexer.byteAt(i + 1);
|
|
1000
|
+
while (marker === 255) marker = lexer.byteAt(++i + 1);
|
|
1001
|
+
if (marker < 0) return void 0;
|
|
1002
|
+
if (marker === 217) return i + 2;
|
|
1003
|
+
if (marker >= 208 && marker <= 215 || marker === 1) {
|
|
1004
|
+
i += 2;
|
|
1005
|
+
continue;
|
|
1006
|
+
}
|
|
1007
|
+
const hi = lexer.byteAt(i + 2);
|
|
1008
|
+
const lo = lexer.byteAt(i + 3);
|
|
1009
|
+
if (hi < 0 || lo < 0 || (hi << 8) + lo < 2) return void 0;
|
|
1010
|
+
i += 2 + (hi << 8) + lo;
|
|
1011
|
+
if (marker !== 218) continue;
|
|
1012
|
+
for (;;) {
|
|
1013
|
+
const b = lexer.byteAt(i);
|
|
1014
|
+
if (b < 0) return void 0;
|
|
1015
|
+
const next = lexer.byteAt(i + 1);
|
|
1016
|
+
if (b === 255 && next !== 0 && !(next >= 208 && next <= 215)) break;
|
|
1017
|
+
i++;
|
|
1018
|
+
}
|
|
1019
|
+
}
|
|
1020
|
+
}
|
|
856
1021
|
/** How many bytes an UNFILTERED inline image's samples take, if that is known. */
|
|
857
1022
|
function inlineLength(dict) {
|
|
858
1023
|
if (dict.has("F") || dict.has("Filter")) return void 0;
|
|
@@ -199,12 +199,28 @@ function placeVectors(vectors, d) {
|
|
|
199
199
|
ys.push(s.y);
|
|
200
200
|
if (s.op === "cubic") xs.push(s.x1, s.x2), ys.push(s.y1, s.y2);
|
|
201
201
|
}
|
|
202
|
+
const g = v.gradient;
|
|
203
|
+
const moved = g?.axis !== void 0 ? { gradient: {
|
|
204
|
+
...g,
|
|
205
|
+
axis: ((a) => {
|
|
206
|
+
const p0 = d.place(a.x0, a.y0);
|
|
207
|
+
const p1 = d.place(a.x1, a.y1);
|
|
208
|
+
return {
|
|
209
|
+
x0: p0.x,
|
|
210
|
+
y0: p0.y,
|
|
211
|
+
x1: p1.x,
|
|
212
|
+
y1: p1.y
|
|
213
|
+
};
|
|
214
|
+
})(g.axis)
|
|
215
|
+
} } : {};
|
|
202
216
|
if (xs.length === 0) return {
|
|
203
217
|
...v,
|
|
218
|
+
...moved,
|
|
204
219
|
segs
|
|
205
220
|
};
|
|
206
221
|
return {
|
|
207
222
|
...v,
|
|
223
|
+
...moved,
|
|
208
224
|
segs,
|
|
209
225
|
minX: Math.min(...xs),
|
|
210
226
|
minY: Math.min(...ys),
|
|
@@ -110,6 +110,14 @@ export declare class PdfFile {
|
|
|
110
110
|
* reader takes one from whoever needs it:
|
|
111
111
|
*
|
|
112
112
|
* ```ts
|
|
113
|
+
* // browser: any wasm/JS decoder you ship — brotli-dec-wasm is ~200 KB
|
|
114
|
+
* import brotliPromise from 'brotli-dec-wasm';
|
|
115
|
+
* const brotli = await brotliPromise;
|
|
116
|
+
* Ream.parse(pdf, { filters: { BrotliDecode: (b) => brotli.decompress(b) } });
|
|
117
|
+
* ```
|
|
118
|
+
*
|
|
119
|
+
* ```ts
|
|
120
|
+
* // node, where the runtime already carries one
|
|
113
121
|
* import { brotliDecompressSync } from 'node:zlib';
|
|
114
122
|
* Ream.parse(pdf, { filters: { BrotliDecode: (b) => brotliDecompressSync(b) } });
|
|
115
123
|
* ```
|
|
@@ -29,6 +29,16 @@ import { FontRegistry } from '../core/font/index.js';
|
|
|
29
29
|
* @returns Name → a one-face registry holding that program.
|
|
30
30
|
*/
|
|
31
31
|
export declare function collectEmbeddedFonts(file: PdfFile, pages: ReadonlyArray<PdfPage>, losses?: Array<Loss>): Map<string, FontRegistry>;
|
|
32
|
+
/**
|
|
33
|
+
* Every `/Font` dictionary the pages draw with, each once: the page's own
|
|
34
|
+
* resources and those of every form it paints (§8.8 — a drawing keeps most of
|
|
35
|
+
* its lettering inside them).
|
|
36
|
+
*
|
|
37
|
+
* @param file The owning file.
|
|
38
|
+
* @param pages The pages whose fonts are wanted.
|
|
39
|
+
* @param visit Called once per font dictionary.
|
|
40
|
+
*/
|
|
41
|
+
export declare function eachPageFont(file: PdfFile, pages: ReadonlyArray<PdfPage>, visit: (fontDict: PdfDict) => void): void;
|
|
32
42
|
/**
|
|
33
43
|
* The name a run set in `fontDict` will ask for: its `/BaseFont` without the
|
|
34
44
|
* six-capital subset prefix (§9.6.4), lowercased.
|
|
@@ -49,3 +59,17 @@ export declare function embeddedFontName(file: PdfFile, fontDict: PdfDict): stri
|
|
|
49
59
|
* @param fontDict The font dictionary.
|
|
50
60
|
*/
|
|
51
61
|
export declare function hasLiftableProgram(file: PdfFile, fontDict: PdfDict): boolean;
|
|
62
|
+
/**
|
|
63
|
+
* The style a font's embedded program states for itself — the `head` table's
|
|
64
|
+
* macStyle, bit 0 bold and bit 1 italic — where the file carries a TrueType
|
|
65
|
+
* program to read it from.
|
|
66
|
+
*
|
|
67
|
+
* @param file The document.
|
|
68
|
+
* @param fontDict The font dictionary.
|
|
69
|
+
* @returns The program's own style, or `undefined` where there is no program
|
|
70
|
+
* or it has no header.
|
|
71
|
+
*/
|
|
72
|
+
export declare function programStyle(file: PdfFile, fontDict: PdfDict): {
|
|
73
|
+
bold: boolean;
|
|
74
|
+
italic: boolean;
|
|
75
|
+
} | undefined;
|
|
@@ -32,11 +32,7 @@ var MAX_FORM_DEPTH = 8;
|
|
|
32
32
|
*/
|
|
33
33
|
function collectEmbeddedFonts(file, pages, losses) {
|
|
34
34
|
const out = /* @__PURE__ */ new Map();
|
|
35
|
-
|
|
36
|
-
const visiting = /* @__PURE__ */ new Set();
|
|
37
|
-
const addFont = (fontDict) => {
|
|
38
|
-
if (seen.has(fontDict)) return;
|
|
39
|
-
seen.add(fontDict);
|
|
35
|
+
eachPageFont(file, pages, (fontDict) => {
|
|
40
36
|
const name = embeddedFontName(file, fontDict);
|
|
41
37
|
if (name === void 0 || out.has(name)) return;
|
|
42
38
|
const program = fontProgram(file, fontDict);
|
|
@@ -53,13 +49,29 @@ function collectEmbeddedFonts(file, pages, losses) {
|
|
|
53
49
|
}
|
|
54
50
|
out.set(name, FontRegistry.fromBytes({ regular: program }));
|
|
55
51
|
} catch {}
|
|
56
|
-
};
|
|
52
|
+
});
|
|
53
|
+
return out;
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Every `/Font` dictionary the pages draw with, each once: the page's own
|
|
57
|
+
* resources and those of every form it paints (§8.8 — a drawing keeps most of
|
|
58
|
+
* its lettering inside them).
|
|
59
|
+
*
|
|
60
|
+
* @param file The owning file.
|
|
61
|
+
* @param pages The pages whose fonts are wanted.
|
|
62
|
+
* @param visit Called once per font dictionary.
|
|
63
|
+
*/
|
|
64
|
+
function eachPageFont(file, pages, visit) {
|
|
65
|
+
const seen = /* @__PURE__ */ new Set();
|
|
66
|
+
const visiting = /* @__PURE__ */ new Set();
|
|
57
67
|
const walk = (resources, depth) => {
|
|
58
68
|
if (!resources) return;
|
|
59
69
|
const fonts = file.get(resources, "Font");
|
|
60
70
|
if (fonts instanceof Map) for (const value of fonts.values()) {
|
|
61
71
|
const dict = file.resolve(value);
|
|
62
|
-
if (dict instanceof Map)
|
|
72
|
+
if (!(dict instanceof Map) || seen.has(dict)) continue;
|
|
73
|
+
seen.add(dict);
|
|
74
|
+
visit(dict);
|
|
63
75
|
}
|
|
64
76
|
if (depth >= MAX_FORM_DEPTH) return;
|
|
65
77
|
const xobjects = file.get(resources, "XObject");
|
|
@@ -76,7 +88,6 @@ function collectEmbeddedFonts(file, pages, losses) {
|
|
|
76
88
|
}
|
|
77
89
|
};
|
|
78
90
|
for (const page of pages) walk(page.resources, 0);
|
|
79
|
-
return out;
|
|
80
91
|
}
|
|
81
92
|
/**
|
|
82
93
|
* The name a run set in `fontDict` will ask for: its `/BaseFont` without the
|
|
@@ -105,6 +116,34 @@ function embeddedFontName(file, fontDict) {
|
|
|
105
116
|
function hasLiftableProgram(file, fontDict) {
|
|
106
117
|
return fontProgram(file, fontDict) !== void 0;
|
|
107
118
|
}
|
|
119
|
+
/**
|
|
120
|
+
* The style a font's embedded program states for itself — the `head` table's
|
|
121
|
+
* macStyle, bit 0 bold and bit 1 italic — where the file carries a TrueType
|
|
122
|
+
* program to read it from.
|
|
123
|
+
*
|
|
124
|
+
* @param file The document.
|
|
125
|
+
* @param fontDict The font dictionary.
|
|
126
|
+
* @returns The program's own style, or `undefined` where there is no program
|
|
127
|
+
* or it has no header.
|
|
128
|
+
*/
|
|
129
|
+
function programStyle(file, fontDict) {
|
|
130
|
+
const program = fontProgram(file, fontDict);
|
|
131
|
+
if (!program || program.length < 12) return void 0;
|
|
132
|
+
const view = new DataView(program.buffer, program.byteOffset, program.byteLength);
|
|
133
|
+
const tables = view.getUint16(4);
|
|
134
|
+
for (let i = 0; i < tables; i++) {
|
|
135
|
+
const at = 12 + i * 16;
|
|
136
|
+
if (at + 16 > program.length) return void 0;
|
|
137
|
+
if (String.fromCharCode(...program.subarray(at, at + 4)) !== "head") continue;
|
|
138
|
+
const offset = view.getUint32(at + 8);
|
|
139
|
+
if (offset + 46 > program.length) return void 0;
|
|
140
|
+
const macStyle = view.getUint16(offset + 44);
|
|
141
|
+
return {
|
|
142
|
+
bold: (macStyle & 1) !== 0,
|
|
143
|
+
italic: (macStyle & 2) !== 0
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
}
|
|
108
147
|
/** §9.9 `/FontFile2` — the TrueType program, off the font or its descendant. */
|
|
109
148
|
function fontProgram(file, fontDict) {
|
|
110
149
|
const owner = descendant(file, fontDict) ?? fontDict;
|
|
@@ -123,4 +162,4 @@ function descendant(file, fontDict) {
|
|
|
123
162
|
return first instanceof Map ? first : void 0;
|
|
124
163
|
}
|
|
125
164
|
//#endregion
|
|
126
|
-
export { collectEmbeddedFonts, embeddedFontName, hasLiftableProgram };
|
|
165
|
+
export { collectEmbeddedFonts, eachPageFont, embeddedFontName, hasLiftableProgram, programStyle };
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { BodyElement, ParagraphProperties, Section, SectionProperties, TextOutline } from '../core/document-model/index.js';
|
|
2
|
-
import { FlowDoc } from '../core/ir/flow.js';
|
|
2
|
+
import { FaceFamily, FlowDoc } from '../core/ir/flow.js';
|
|
3
3
|
import { FontRegistry } from '../core/font/index.js';
|
|
4
|
-
import { Loss, ResourceStore } from '../core/ir/index.js';
|
|
4
|
+
import { Loss, Pt, ResourceStore } from '../core/ir/index.js';
|
|
5
5
|
import { PdfImage } from './images.js';
|
|
6
6
|
import { PdfPage } from './document.js';
|
|
7
7
|
import { PdfVector } from './vector.js';
|
|
@@ -36,6 +36,20 @@ export interface PageFrame {
|
|
|
36
36
|
*/
|
|
37
37
|
export declare function paragraphBlock(text: string, outlineLevel?: number): BodyElement;
|
|
38
38
|
/** One piece of reconstructed text, carrying any hyperlink (E-PDF EP8). */
|
|
39
|
+
/**
|
|
40
|
+
* The space set between a span and what follows it, in the span's own size and
|
|
41
|
+
* face: a word space is as wide as the type it stands in.
|
|
42
|
+
*
|
|
43
|
+
* Written bare, the space took the document's default size, whatever the words
|
|
44
|
+
* around it were set at: issue10665_reduced.pdf sets "78" and "110" twenty
|
|
45
|
+
* points apart in 60-point type, and the eleven-point space between them closed
|
|
46
|
+
* them up to "78110"; in 7-point footnotes the same space is half as wide
|
|
47
|
+
* again as the page's, and pushes their lines over.
|
|
48
|
+
*
|
|
49
|
+
* @param before The span the space follows, where there is one.
|
|
50
|
+
* @returns The space.
|
|
51
|
+
*/
|
|
52
|
+
export declare function spaceAfter(before: TextSpan | undefined): TextSpan;
|
|
39
53
|
export interface TextSpan {
|
|
40
54
|
readonly text: string;
|
|
41
55
|
readonly href?: string;
|
|
@@ -66,7 +80,21 @@ export interface TextSpan {
|
|
|
66
80
|
* survives as its own run) and squashing whitespace. With no hrefs this
|
|
67
81
|
* collapses to a single run — the same shape {@link paragraphBlock} produces.
|
|
68
82
|
*/
|
|
69
|
-
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number, placement?: Pick<ParagraphProperties, 'alignment' | 'spacingBefore' | 'indentLeft' | 'indentFirstLine'>): BodyElement;
|
|
83
|
+
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number, placement?: Pick<ParagraphProperties, 'alignment' | 'spacingBefore' | 'spacingLine' | 'spacingLineRule' | 'indentLeft' | 'indentRight' | 'indentFirstLine' | 'tabs'>): BodyElement;
|
|
84
|
+
/**
|
|
85
|
+
* The height of a line that carries nothing to read: one twip, the least
|
|
86
|
+
* §17.3.1.33 can state. Zero is not a height a writer states at all — it is
|
|
87
|
+
* read as "no line spacing given", and the reader's single spacing comes back.
|
|
88
|
+
*/
|
|
89
|
+
export declare const CARRIER_LINE_PT: Pt;
|
|
90
|
+
/**
|
|
91
|
+
* §17.3.1.33 — the paragraph a floating mark is anchored in, which takes no
|
|
92
|
+
* room: the mark stands where the page drew it and the line carrying it is on
|
|
93
|
+
* no page. Left at a reader's single spacing, every rule and fill an invoice
|
|
94
|
+
* draws was a blank line in the flow, and the text under it moved down a line
|
|
95
|
+
* for each.
|
|
96
|
+
*/
|
|
97
|
+
export declare const FLOAT_CARRIER: ParagraphProperties;
|
|
70
98
|
/**
|
|
71
99
|
* Store a {@link PdfImage}'s bytes (content-addressed dedup) and build the image
|
|
72
100
|
* {@link BodyElement} that references them, sized in points from the placement
|
|
@@ -123,10 +151,78 @@ export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: num
|
|
|
123
151
|
* `A4`. Returns `undefined` when there is no usable first-page box.
|
|
124
152
|
*/
|
|
125
153
|
export declare function sectionFromPdfPages(pages: ReadonlyArray<PdfPage>): SectionProperties | undefined;
|
|
154
|
+
/** How many lines a sheet must hold before where they end says where its measure does. */
|
|
155
|
+
export declare const MEASURE_LINES = 3;
|
|
156
|
+
/**
|
|
157
|
+
* How far in, as a share of the sheet, a right margin the lines do not show
|
|
158
|
+
* may come: what a sheet of a line or two is re-set across is at least the
|
|
159
|
+
* rest of it.
|
|
160
|
+
*/
|
|
161
|
+
export declare const GUESSED_MARGIN: number;
|
|
162
|
+
/**
|
|
163
|
+
* §17.3.1.33 — where the baseline of an EXACT line stands in its box, from the
|
|
164
|
+
* top. LibreOffice puts it at four fifths of the height; Word at the height
|
|
165
|
+
* less the face's descent, which for the faces documents use is within a tenth
|
|
166
|
+
* of a line of the same place.
|
|
167
|
+
*/
|
|
168
|
+
export declare const BASELINE_AT = 0.8;
|
|
169
|
+
/** A line's box where no pitch was measured: the ordinary single spacing, in ems. */
|
|
170
|
+
export declare const NATURAL_LINE_EM = 1.2;
|
|
171
|
+
/**
|
|
172
|
+
* The size a document's text is set in: the middle of every size its runs
|
|
173
|
+
* carry.
|
|
174
|
+
*
|
|
175
|
+
* @param pageRuns Each page's runs.
|
|
176
|
+
* @returns The size, in points; 0 where no run states one.
|
|
177
|
+
*/
|
|
178
|
+
export declare function textSizeOf(pageRuns: ReadonlyArray<ReadonlyArray<TextRun>>): number;
|
|
179
|
+
/**
|
|
180
|
+
* Whether a run is set too small to be read, beside the text of its document —
|
|
181
|
+
* a mark the producer leaves on the sheet, not a line of the page.
|
|
182
|
+
*
|
|
183
|
+
* TCPDF signs the last page of everything it makes "Powered by TCPDF
|
|
184
|
+
* (www.tcpdf.org)" in type one point high, three points from the corner of the
|
|
185
|
+
* paper. Taken for text it was the leftmost and the lowest thing on the page:
|
|
186
|
+
* the margins came in at the edge of the sheet and every line of basicapi.pdf
|
|
187
|
+
* and alphatrans.pdf was set against it.
|
|
188
|
+
*
|
|
189
|
+
* @param run The run.
|
|
190
|
+
* @param textSize The size the document's text is set in (see {@link textSizeOf}).
|
|
191
|
+
* @returns `true` where the run is a mark rather than text.
|
|
192
|
+
*/
|
|
193
|
+
export declare function tooSmallToRead(run: TextRun, textSize: number): boolean;
|
|
194
|
+
/**
|
|
195
|
+
* The margins the SOURCE used, measured off where its words actually sit.
|
|
196
|
+
*
|
|
197
|
+
* A PDF states none — text is placed anywhere on the MediaBox — so the reader
|
|
198
|
+
* used to leave them at zero rather than invent an inch. But the words
|
|
199
|
+
* themselves say where the margin was: the leftmost glyph on the page is the
|
|
200
|
+
* left margin, and reflowing inside it keeps the measure the author set instead
|
|
201
|
+
* of running the text from edge to edge.
|
|
202
|
+
*
|
|
203
|
+
* Measured on the MEDIAN page rather than the extreme one, so a single full-
|
|
204
|
+
* bleed rule or a page number in the corner does not collapse the margin for
|
|
205
|
+
* the whole document, and clamped so a strange page cannot leave no text area
|
|
206
|
+
* at all.
|
|
207
|
+
*
|
|
208
|
+
* @param section The section the page box gave, or `undefined`.
|
|
209
|
+
* @param shown Each page as it is shown, for its own width and height.
|
|
210
|
+
* @param pageRuns Each page's runs, already placed on the shown page.
|
|
211
|
+
* @param pageMarks Each page's pictures, which the measure has to hold.
|
|
212
|
+
* @param foot The running foot lifted off the pages, as the first page
|
|
213
|
+
* showed it — the band the text block stands above.
|
|
214
|
+
* @returns The section with measured margins, or `section` when nothing is
|
|
215
|
+
* measurable.
|
|
216
|
+
*/
|
|
126
217
|
export declare function withMeasuredMargins(section: SectionProperties | undefined, shown: ReadonlyArray<{
|
|
127
218
|
width: number;
|
|
128
219
|
height: number;
|
|
129
|
-
}>, pageRuns: ReadonlyArray<ReadonlyArray<TextRun
|
|
220
|
+
}>, pageRuns: ReadonlyArray<ReadonlyArray<TextRun>>, pageMarks?: ReadonlyArray<ReadonlyArray<{
|
|
221
|
+
x: number;
|
|
222
|
+
y: number;
|
|
223
|
+
widthPt: number;
|
|
224
|
+
heightPt: number;
|
|
225
|
+
}>>, foot?: ReadonlyArray<TextRun>): SectionProperties | undefined;
|
|
130
226
|
/**
|
|
131
227
|
* Assemble the final {@link FlowDoc} for a reconstruction: the body elements
|
|
132
228
|
* with their styles resolved against the empty style sheet, the lifted-image
|
|
@@ -134,4 +230,4 @@ export declare function withMeasuredMargins(section: SectionProperties | undefin
|
|
|
134
230
|
* both reconstruction paths (the tagged fast-path EP3 and the heuristic layout
|
|
135
231
|
* path EP4).
|
|
136
232
|
*/
|
|
137
|
-
export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore, section?: SectionProperties, embeddedFonts?: ReadonlyMap<string, FontRegistry>, sections?: ReadonlyArray<Section>, headersFooters?: ReadonlyMap<string, ReadonlyArray<BodyElement
|
|
233
|
+
export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore, section?: SectionProperties, embeddedFonts?: ReadonlyMap<string, FontRegistry>, sections?: ReadonlyArray<Section>, headersFooters?: ReadonlyMap<string, ReadonlyArray<BodyElement>>, faceFamilies?: ReadonlyMap<string, FaceFamily>): FlowDoc;
|