reamkit 1.29.0 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/core/document-model/types.d.ts +2 -0
- package/dist/esm/core/drawingml/shape-render.js +13 -1
- package/dist/esm/core/fonts/index.d.ts +1 -1
- package/dist/esm/core/fonts/remote-fonts.d.ts +8 -0
- package/dist/esm/core/fonts/remote-fonts.js +100 -17
- package/dist/esm/core/ir/flow.d.ts +17 -0
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/pdf-reader/annot-draw.js +93 -1
- package/dist/esm/pdf-reader/annots.d.ts +18 -0
- package/dist/esm/pdf-reader/annots.js +86 -9
- package/dist/esm/pdf-reader/cmap.js +5 -2
- package/dist/esm/pdf-reader/content.d.ts +17 -4
- package/dist/esm/pdf-reader/content.js +163 -11
- package/dist/esm/pdf-reader/display.js +16 -0
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +24 -0
- package/dist/esm/pdf-reader/embedded-fonts.js +48 -9
- package/dist/esm/pdf-reader/flow-build.d.ts +96 -5
- package/dist/esm/pdf-reader/flow-build.js +210 -23
- package/dist/esm/pdf-reader/font.d.ts +27 -1
- package/dist/esm/pdf-reader/font.js +313 -29
- package/dist/esm/pdf-reader/glyf-outline.js +13 -2
- package/dist/esm/pdf-reader/glyph-shapes.d.ts +18 -0
- package/dist/esm/pdf-reader/glyph-shapes.js +57 -0
- package/dist/esm/pdf-reader/image-decode.js +70 -4
- package/dist/esm/pdf-reader/images.d.ts +5 -0
- package/dist/esm/pdf-reader/images.js +4 -2
- package/dist/esm/pdf-reader/jbig2.d.ts +40 -1
- package/dist/esm/pdf-reader/jbig2.js +78 -16
- package/dist/esm/pdf-reader/jpeg.d.ts +6 -3
- package/dist/esm/pdf-reader/jpeg.js +21 -1
- package/dist/esm/pdf-reader/layout.d.ts +26 -2
- package/dist/esm/pdf-reader/layout.js +638 -92
- package/dist/esm/pdf-reader/lexer.d.ts +10 -0
- package/dist/esm/pdf-reader/lexer.js +17 -0
- package/dist/esm/pdf-reader/pattern-tint.d.ts +11 -1
- package/dist/esm/pdf-reader/pattern-tint.js +21 -3
- package/dist/esm/pdf-reader/regions.d.ts +25 -0
- package/dist/esm/pdf-reader/regions.js +167 -0
- package/dist/esm/pdf-reader/shading.d.ts +58 -2
- package/dist/esm/pdf-reader/shading.js +181 -11
- package/dist/esm/pdf-reader/struct-tree.js +112 -8
- package/dist/esm/pdf-reader/tagged.js +27 -7
- package/dist/esm/pdf-reader/text-rules.js +1 -1
- package/dist/esm/pdf-reader/text.js +9 -14
- package/dist/esm/pdf-reader/vector.d.ts +8 -2
- package/dist/esm/pdf-reader/vector.js +34 -5
- package/dist/esm/word/docx-writer.js +137 -36
- package/dist/esm/word/drawing-parser.js +7 -2
- package/dist/esm/word/paragraph-properties.js +2 -0
- package/package.json +5 -3
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
import { PDF_NULL, PdfHexString, PdfName } from "../pdf/objects.js";
|
|
2
|
+
import { bidiClass } from "../core/bidi/char-types.js";
|
|
3
|
+
import { reorderVisual } from "../core/bidi/algorithm.js";
|
|
4
|
+
import { analyzeString, hasBidiCharacters } from "../core/bidi/index.js";
|
|
2
5
|
import { Lexer } from "./lexer.js";
|
|
3
|
-
import { cmykHex, grayHex, rgbHex, spaceColor } from "./shading.js";
|
|
6
|
+
import { cmykHex, dashLengths, grayHex, rgbHex, spaceColor } from "./shading.js";
|
|
4
7
|
//#region src/pdf-reader/content.ts
|
|
5
8
|
/**
|
|
6
9
|
* A painted path emitted by the path-painting operators (§8.5.3), captured in
|
|
@@ -101,9 +104,13 @@ function initialState() {
|
|
|
101
104
|
fillColor: "000000",
|
|
102
105
|
strokeColor: "000000",
|
|
103
106
|
lineWidth: 1,
|
|
107
|
+
dash: [],
|
|
108
|
+
lineCap: 0,
|
|
104
109
|
fillGradient: void 0,
|
|
105
110
|
fillPattern: void 0,
|
|
111
|
+
fillPatternPaint: void 0,
|
|
106
112
|
fillAlpha: 1,
|
|
113
|
+
strokeAlpha: 1,
|
|
107
114
|
fillDarkens: false,
|
|
108
115
|
blendMode: void 0,
|
|
109
116
|
softMask: false,
|
|
@@ -229,6 +236,25 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
229
236
|
y
|
|
230
237
|
});
|
|
231
238
|
};
|
|
239
|
+
const currentPoint = () => {
|
|
240
|
+
for (let i = path.length - 1; i >= 0; i--) {
|
|
241
|
+
const seg = path[i];
|
|
242
|
+
if (seg.op === "close") {
|
|
243
|
+
for (let k = i - 1; k >= 0; k--) {
|
|
244
|
+
const start = path[k];
|
|
245
|
+
if (start.op === "move") return {
|
|
246
|
+
x: start.x,
|
|
247
|
+
y: start.y
|
|
248
|
+
};
|
|
249
|
+
}
|
|
250
|
+
return;
|
|
251
|
+
}
|
|
252
|
+
return {
|
|
253
|
+
x: seg.x,
|
|
254
|
+
y: seg.y
|
|
255
|
+
};
|
|
256
|
+
}
|
|
257
|
+
};
|
|
232
258
|
const rectTo = (x, y, w, h) => {
|
|
233
259
|
moveTo(x, y);
|
|
234
260
|
lineTo(x + w, y);
|
|
@@ -241,6 +267,11 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
241
267
|
const scale = Math.sqrt(Math.abs(m[0] * m[3] - m[1] * m[2])) || 1;
|
|
242
268
|
return state.lineWidth * scale;
|
|
243
269
|
};
|
|
270
|
+
const ctmDash = () => {
|
|
271
|
+
const m = state.ctm;
|
|
272
|
+
const scale = Math.sqrt(Math.abs(m[0] * m[3] - m[1] * m[2])) || 1;
|
|
273
|
+
return state.dash.map((n) => n * scale);
|
|
274
|
+
};
|
|
244
275
|
const paintPath = (fill, stroke) => {
|
|
245
276
|
if (path.length >= 2 && (fill || stroke) && visible()) {
|
|
246
277
|
const mcid = mcStack.length > 0 ? mcStack[mcStack.length - 1] : void 0;
|
|
@@ -249,6 +280,7 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
249
280
|
segs: path,
|
|
250
281
|
...state.clip ? { clip: state.clip } : {},
|
|
251
282
|
...fill && state.fillPattern !== void 0 ? { patternName: state.fillPattern } : {},
|
|
283
|
+
...fill && state.fillPattern !== void 0 && state.fillPatternPaint !== void 0 ? { patternPaint: state.fillPatternPaint } : {},
|
|
252
284
|
...fill ? { fillHex: state.fillColor } : {},
|
|
253
285
|
...fill && state.fillAlpha < 1 ? { alpha: state.fillAlpha } : {},
|
|
254
286
|
...fill && state.fillDarkens ? { darkens: true } : {},
|
|
@@ -259,6 +291,10 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
259
291
|
strokeHex: state.strokeColor,
|
|
260
292
|
lineWidth: ctmLineWidth()
|
|
261
293
|
} : {},
|
|
294
|
+
...stroke && state.dash.length > 0 ? { dash: ctmDash() } : {},
|
|
295
|
+
...stroke && state.lineCap === 1 ? { cap: "round" } : {},
|
|
296
|
+
...stroke && state.lineCap === 2 ? { cap: "square" } : {},
|
|
297
|
+
...stroke && state.strokeAlpha < 1 ? { strokeAlpha: state.strokeAlpha } : {},
|
|
262
298
|
...mcid !== void 0 ? { mcid } : {}
|
|
263
299
|
});
|
|
264
300
|
}
|
|
@@ -519,6 +555,8 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
519
555
|
if (named !== void 0) {
|
|
520
556
|
state.fillGradient = shadings.get(named);
|
|
521
557
|
state.fillPattern = state.fillGradient ? void 0 : named;
|
|
558
|
+
const components = operands.filter((o) => typeof o === "number");
|
|
559
|
+
state.fillPatternPaint = components.length > 0 ? spaceColor(components, void 0) : void 0;
|
|
522
560
|
break;
|
|
523
561
|
}
|
|
524
562
|
const hex = colorOfOperands(operands, state.fillSpace);
|
|
@@ -531,7 +569,12 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
531
569
|
}
|
|
532
570
|
case "SCN":
|
|
533
571
|
case "SC": {
|
|
534
|
-
|
|
572
|
+
const last = operands[operands.length - 1];
|
|
573
|
+
if (last instanceof PdfName) {
|
|
574
|
+
const sweep = shadings.get(last.value);
|
|
575
|
+
if (sweep) state.strokeColor = midGradient(sweep);
|
|
576
|
+
break;
|
|
577
|
+
}
|
|
535
578
|
const hex = colorOfOperands(operands, state.strokeSpace);
|
|
536
579
|
if (hex !== void 0) state.strokeColor = hex;
|
|
537
580
|
break;
|
|
@@ -548,6 +591,14 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
548
591
|
case "w":
|
|
549
592
|
state.lineWidth = num(0);
|
|
550
593
|
break;
|
|
594
|
+
case "d":
|
|
595
|
+
state.dash = dashLengths(operands[0]);
|
|
596
|
+
break;
|
|
597
|
+
case "J": {
|
|
598
|
+
const cap = num(0);
|
|
599
|
+
if (cap === 0 || cap === 1 || cap === 2) state.lineCap = cap;
|
|
600
|
+
break;
|
|
601
|
+
}
|
|
551
602
|
case "m":
|
|
552
603
|
moveTo(num(0), num(1));
|
|
553
604
|
break;
|
|
@@ -557,6 +608,26 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
557
608
|
case "c":
|
|
558
609
|
curveTo(num(0), num(1), num(2), num(3), num(4), num(5));
|
|
559
610
|
break;
|
|
611
|
+
case "v": {
|
|
612
|
+
const at = currentPoint();
|
|
613
|
+
if (at) {
|
|
614
|
+
const [x2, y2] = toPage(num(0), num(1));
|
|
615
|
+
const [x, y] = toPage(num(2), num(3));
|
|
616
|
+
path.push({
|
|
617
|
+
op: "cubic",
|
|
618
|
+
x1: at.x,
|
|
619
|
+
y1: at.y,
|
|
620
|
+
x2,
|
|
621
|
+
y2,
|
|
622
|
+
x,
|
|
623
|
+
y
|
|
624
|
+
});
|
|
625
|
+
}
|
|
626
|
+
break;
|
|
627
|
+
}
|
|
628
|
+
case "y":
|
|
629
|
+
curveTo(num(0), num(1), num(2), num(3), num(2), num(3));
|
|
630
|
+
break;
|
|
560
631
|
case "re":
|
|
561
632
|
rectTo(num(0), num(1), num(2), num(3));
|
|
562
633
|
break;
|
|
@@ -610,6 +681,10 @@ function interpretContent(bytes, fonts, initialCtm = IDENTITY, shadings = /* @__
|
|
|
610
681
|
state.blendMode = paint.blend;
|
|
611
682
|
}
|
|
612
683
|
if (paint.masked !== void 0) state.softMask = paint.masked;
|
|
684
|
+
if (paint.lineWidth !== void 0) state.lineWidth = paint.lineWidth;
|
|
685
|
+
if (paint.dash !== void 0) state.dash = paint.dash;
|
|
686
|
+
if (paint.lineCap !== void 0) state.lineCap = paint.lineCap;
|
|
687
|
+
if (paint.strokeAlpha !== void 0) state.strokeAlpha = paint.strokeAlpha;
|
|
613
688
|
}
|
|
614
689
|
break;
|
|
615
690
|
}
|
|
@@ -713,12 +788,27 @@ function strokesText(mode) {
|
|
|
713
788
|
* ArabicCIDTrueType.pdf came out mirrored for exactly that reason: the reader
|
|
714
789
|
* passed visual order through and the layout reversed it a second time.
|
|
715
790
|
*
|
|
716
|
-
* A run
|
|
717
|
-
* number inside an Arabic sentence
|
|
718
|
-
*
|
|
791
|
+
* A run wholly right-to-left is simply reversed. A MIXED one — a full stop or
|
|
792
|
+
* a number inside an Arabic sentence — is put through the bidi algorithm's
|
|
793
|
+
* reordering (UAX #9 L2), which for the runs a line holds undoes itself: the
|
|
794
|
+
* painted order reordered is the reading order. Left alone, a note of
|
|
795
|
+
* freetext_no_appearance.pdf came back with every line that held a full stop
|
|
796
|
+
* reading back to front. The direction it is reordered in is the one most of
|
|
797
|
+
* its letters run in — the painted order says nothing about which comes first.
|
|
719
798
|
*/
|
|
720
799
|
function logicalOrder(text) {
|
|
721
|
-
|
|
800
|
+
if (isRightToLeft(text)) return [...text].reverse().join("");
|
|
801
|
+
if (!hasBidiCharacters(text)) return text;
|
|
802
|
+
const chars = [...text];
|
|
803
|
+
let rtl = 0;
|
|
804
|
+
let strong = 0;
|
|
805
|
+
for (const ch of chars) {
|
|
806
|
+
const type = bidiClass(ch.codePointAt(0) ?? 0);
|
|
807
|
+
if (type === "R" || type === "AL") rtl++;
|
|
808
|
+
if (type === "R" || type === "AL" || type === "L") strong++;
|
|
809
|
+
}
|
|
810
|
+
const { levels } = analyzeString(text, rtl * 2 >= strong ? "rtl" : "ltr");
|
|
811
|
+
return reorderVisual(levels).map((i) => chars[i] ?? "").join("");
|
|
722
812
|
}
|
|
723
813
|
/**
|
|
724
814
|
* Whether a string is wholly right-to-left: at least one letter of an RTL
|
|
@@ -837,8 +927,8 @@ function readValue(lexer) {
|
|
|
837
927
|
* The bytes are binary and may hold `EI` themselves, so the end is found by
|
|
838
928
|
* MEASURING where the dictionary says how much there is — an unfiltered image
|
|
839
929
|
* is exactly `ceil(W · BPC · components / 8) · H` bytes — and only searched for
|
|
840
|
-
* where a filter makes the length unknowable
|
|
841
|
-
* two of them and both were skipped over.
|
|
930
|
+
* where a filter makes the length unknowable (see {@link filteredEnd}).
|
|
931
|
+
* images_1bit_grayscale.pdf draws two of them and both were skipped over.
|
|
842
932
|
*/
|
|
843
933
|
function readInlineImage(lexer) {
|
|
844
934
|
const dict = /* @__PURE__ */ new Map();
|
|
@@ -848,24 +938,86 @@ function readInlineImage(lexer) {
|
|
|
848
938
|
if (tok.kind === "keyword" && tok.value === "ID") break;
|
|
849
939
|
if (tok.kind === "name") dict.set(tok.value, readValue(lexer));
|
|
850
940
|
}
|
|
851
|
-
const start = lexer.pos + 1;
|
|
852
941
|
const measured = inlineLength(dict);
|
|
942
|
+
const crlf = lexer.byteAt(lexer.pos) === 13 && lexer.byteAt(lexer.pos + 1) === 10;
|
|
943
|
+
const start = lexer.pos + (crlf && measured === void 0 ? 2 : 1);
|
|
853
944
|
let end = measured !== void 0 ? start + measured : -1;
|
|
854
945
|
if (end < 0 || end > lexer.length) {
|
|
855
|
-
end = lexer
|
|
946
|
+
end = filteredEnd(lexer, start, dict);
|
|
856
947
|
if (end < 0) {
|
|
857
948
|
lexer.pos = lexer.length;
|
|
858
949
|
return;
|
|
859
950
|
}
|
|
860
951
|
}
|
|
861
952
|
const data = lexer.slice(start, Math.min(end, lexer.length));
|
|
862
|
-
const ei = lexer.
|
|
953
|
+
const ei = lexer.indexOfKeyword("EI", end);
|
|
863
954
|
lexer.pos = ei < 0 ? lexer.length : ei + 2;
|
|
864
955
|
return {
|
|
865
956
|
dict,
|
|
866
957
|
data
|
|
867
958
|
};
|
|
868
959
|
}
|
|
960
|
+
/**
|
|
961
|
+
* §8.9.7 — where a FILTERED inline image's bytes end, which the dictionary
|
|
962
|
+
* cannot say.
|
|
963
|
+
*
|
|
964
|
+
* Searched for as the first two bytes `EI`, the end fell inside the picture
|
|
965
|
+
* wherever its own bytes spelled them: a JPEG is dense with them, and
|
|
966
|
+
* bug1065245.pdf's three banners were each cut off a few hundred bytes in,
|
|
967
|
+
* would not decode, and were dropped — a blank sheet. So the data is read to
|
|
968
|
+
* where its OWN encoding ends: a JPEG at its end-of-image marker, hex text at
|
|
969
|
+
* `>`, base-85 at `~>`. Anything else ends at the first `EI` standing as a
|
|
970
|
+
* word of its own.
|
|
971
|
+
*/
|
|
972
|
+
function filteredEnd(lexer, start, dict) {
|
|
973
|
+
const f = dict.get("F") ?? dict.get("Filter");
|
|
974
|
+
const first = Array.isArray(f) ? f[0] : f;
|
|
975
|
+
const filter = first instanceof PdfName ? first.value : "";
|
|
976
|
+
if (filter === "DCT" || filter === "DCTDecode") {
|
|
977
|
+
const eoi = jpegEnd(lexer, start);
|
|
978
|
+
if (eoi !== void 0) return eoi;
|
|
979
|
+
} else if (filter === "AHx" || filter === "ASCIIHexDecode") {
|
|
980
|
+
const gt = lexer.indexOfAscii(">", start);
|
|
981
|
+
if (gt >= 0) return gt + 1;
|
|
982
|
+
} else if (filter === "A85" || filter === "ASCII85Decode") {
|
|
983
|
+
const tilde = lexer.indexOfAscii("~>", start);
|
|
984
|
+
if (tilde >= 0) return tilde + 2;
|
|
985
|
+
}
|
|
986
|
+
return lexer.indexOfKeyword("EI", start);
|
|
987
|
+
}
|
|
988
|
+
/**
|
|
989
|
+
* Where a JPEG that starts at `start` ends: past its end-of-image marker,
|
|
990
|
+
* found by walking its segments — a length-prefixed marker segment is stepped
|
|
991
|
+
* over whole, and the coded data after a start-of-scan is read to the next
|
|
992
|
+
* marker, where `FF` stands before anything but a stuffed `00` or a restart.
|
|
993
|
+
*/
|
|
994
|
+
function jpegEnd(lexer, start) {
|
|
995
|
+
if (lexer.byteAt(start) !== 255 || lexer.byteAt(start + 1) !== 216) return void 0;
|
|
996
|
+
let i = start + 2;
|
|
997
|
+
for (;;) {
|
|
998
|
+
if (lexer.byteAt(i) !== 255) return void 0;
|
|
999
|
+
let marker = lexer.byteAt(i + 1);
|
|
1000
|
+
while (marker === 255) marker = lexer.byteAt(++i + 1);
|
|
1001
|
+
if (marker < 0) return void 0;
|
|
1002
|
+
if (marker === 217) return i + 2;
|
|
1003
|
+
if (marker >= 208 && marker <= 215 || marker === 1) {
|
|
1004
|
+
i += 2;
|
|
1005
|
+
continue;
|
|
1006
|
+
}
|
|
1007
|
+
const hi = lexer.byteAt(i + 2);
|
|
1008
|
+
const lo = lexer.byteAt(i + 3);
|
|
1009
|
+
if (hi < 0 || lo < 0 || (hi << 8) + lo < 2) return void 0;
|
|
1010
|
+
i += 2 + (hi << 8) + lo;
|
|
1011
|
+
if (marker !== 218) continue;
|
|
1012
|
+
for (;;) {
|
|
1013
|
+
const b = lexer.byteAt(i);
|
|
1014
|
+
if (b < 0) return void 0;
|
|
1015
|
+
const next = lexer.byteAt(i + 1);
|
|
1016
|
+
if (b === 255 && next !== 0 && !(next >= 208 && next <= 215)) break;
|
|
1017
|
+
i++;
|
|
1018
|
+
}
|
|
1019
|
+
}
|
|
1020
|
+
}
|
|
869
1021
|
/** How many bytes an UNFILTERED inline image's samples take, if that is known. */
|
|
870
1022
|
function inlineLength(dict) {
|
|
871
1023
|
if (dict.has("F") || dict.has("Filter")) return void 0;
|
|
@@ -199,12 +199,28 @@ function placeVectors(vectors, d) {
|
|
|
199
199
|
ys.push(s.y);
|
|
200
200
|
if (s.op === "cubic") xs.push(s.x1, s.x2), ys.push(s.y1, s.y2);
|
|
201
201
|
}
|
|
202
|
+
const g = v.gradient;
|
|
203
|
+
const moved = g?.axis !== void 0 ? { gradient: {
|
|
204
|
+
...g,
|
|
205
|
+
axis: ((a) => {
|
|
206
|
+
const p0 = d.place(a.x0, a.y0);
|
|
207
|
+
const p1 = d.place(a.x1, a.y1);
|
|
208
|
+
return {
|
|
209
|
+
x0: p0.x,
|
|
210
|
+
y0: p0.y,
|
|
211
|
+
x1: p1.x,
|
|
212
|
+
y1: p1.y
|
|
213
|
+
};
|
|
214
|
+
})(g.axis)
|
|
215
|
+
} } : {};
|
|
202
216
|
if (xs.length === 0) return {
|
|
203
217
|
...v,
|
|
218
|
+
...moved,
|
|
204
219
|
segs
|
|
205
220
|
};
|
|
206
221
|
return {
|
|
207
222
|
...v,
|
|
223
|
+
...moved,
|
|
208
224
|
segs,
|
|
209
225
|
minX: Math.min(...xs),
|
|
210
226
|
minY: Math.min(...ys),
|
|
@@ -29,6 +29,16 @@ import { FontRegistry } from '../core/font/index.js';
|
|
|
29
29
|
* @returns Name → a one-face registry holding that program.
|
|
30
30
|
*/
|
|
31
31
|
export declare function collectEmbeddedFonts(file: PdfFile, pages: ReadonlyArray<PdfPage>, losses?: Array<Loss>): Map<string, FontRegistry>;
|
|
32
|
+
/**
|
|
33
|
+
* Every `/Font` dictionary the pages draw with, each once: the page's own
|
|
34
|
+
* resources and those of every form it paints (§8.8 — a drawing keeps most of
|
|
35
|
+
* its lettering inside them).
|
|
36
|
+
*
|
|
37
|
+
* @param file The owning file.
|
|
38
|
+
* @param pages The pages whose fonts are wanted.
|
|
39
|
+
* @param visit Called once per font dictionary.
|
|
40
|
+
*/
|
|
41
|
+
export declare function eachPageFont(file: PdfFile, pages: ReadonlyArray<PdfPage>, visit: (fontDict: PdfDict) => void): void;
|
|
32
42
|
/**
|
|
33
43
|
* The name a run set in `fontDict` will ask for: its `/BaseFont` without the
|
|
34
44
|
* six-capital subset prefix (§9.6.4), lowercased.
|
|
@@ -49,3 +59,17 @@ export declare function embeddedFontName(file: PdfFile, fontDict: PdfDict): stri
|
|
|
49
59
|
* @param fontDict The font dictionary.
|
|
50
60
|
*/
|
|
51
61
|
export declare function hasLiftableProgram(file: PdfFile, fontDict: PdfDict): boolean;
|
|
62
|
+
/**
|
|
63
|
+
* The style a font's embedded program states for itself — the `head` table's
|
|
64
|
+
* macStyle, bit 0 bold and bit 1 italic — where the file carries a TrueType
|
|
65
|
+
* program to read it from.
|
|
66
|
+
*
|
|
67
|
+
* @param file The document.
|
|
68
|
+
* @param fontDict The font dictionary.
|
|
69
|
+
* @returns The program's own style, or `undefined` where there is no program
|
|
70
|
+
* or it has no header.
|
|
71
|
+
*/
|
|
72
|
+
export declare function programStyle(file: PdfFile, fontDict: PdfDict): {
|
|
73
|
+
bold: boolean;
|
|
74
|
+
italic: boolean;
|
|
75
|
+
} | undefined;
|
|
@@ -32,11 +32,7 @@ var MAX_FORM_DEPTH = 8;
|
|
|
32
32
|
*/
|
|
33
33
|
function collectEmbeddedFonts(file, pages, losses) {
|
|
34
34
|
const out = /* @__PURE__ */ new Map();
|
|
35
|
-
|
|
36
|
-
const visiting = /* @__PURE__ */ new Set();
|
|
37
|
-
const addFont = (fontDict) => {
|
|
38
|
-
if (seen.has(fontDict)) return;
|
|
39
|
-
seen.add(fontDict);
|
|
35
|
+
eachPageFont(file, pages, (fontDict) => {
|
|
40
36
|
const name = embeddedFontName(file, fontDict);
|
|
41
37
|
if (name === void 0 || out.has(name)) return;
|
|
42
38
|
const program = fontProgram(file, fontDict);
|
|
@@ -53,13 +49,29 @@ function collectEmbeddedFonts(file, pages, losses) {
|
|
|
53
49
|
}
|
|
54
50
|
out.set(name, FontRegistry.fromBytes({ regular: program }));
|
|
55
51
|
} catch {}
|
|
56
|
-
};
|
|
52
|
+
});
|
|
53
|
+
return out;
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Every `/Font` dictionary the pages draw with, each once: the page's own
|
|
57
|
+
* resources and those of every form it paints (§8.8 — a drawing keeps most of
|
|
58
|
+
* its lettering inside them).
|
|
59
|
+
*
|
|
60
|
+
* @param file The owning file.
|
|
61
|
+
* @param pages The pages whose fonts are wanted.
|
|
62
|
+
* @param visit Called once per font dictionary.
|
|
63
|
+
*/
|
|
64
|
+
function eachPageFont(file, pages, visit) {
|
|
65
|
+
const seen = /* @__PURE__ */ new Set();
|
|
66
|
+
const visiting = /* @__PURE__ */ new Set();
|
|
57
67
|
const walk = (resources, depth) => {
|
|
58
68
|
if (!resources) return;
|
|
59
69
|
const fonts = file.get(resources, "Font");
|
|
60
70
|
if (fonts instanceof Map) for (const value of fonts.values()) {
|
|
61
71
|
const dict = file.resolve(value);
|
|
62
|
-
if (dict instanceof Map)
|
|
72
|
+
if (!(dict instanceof Map) || seen.has(dict)) continue;
|
|
73
|
+
seen.add(dict);
|
|
74
|
+
visit(dict);
|
|
63
75
|
}
|
|
64
76
|
if (depth >= MAX_FORM_DEPTH) return;
|
|
65
77
|
const xobjects = file.get(resources, "XObject");
|
|
@@ -76,7 +88,6 @@ function collectEmbeddedFonts(file, pages, losses) {
|
|
|
76
88
|
}
|
|
77
89
|
};
|
|
78
90
|
for (const page of pages) walk(page.resources, 0);
|
|
79
|
-
return out;
|
|
80
91
|
}
|
|
81
92
|
/**
|
|
82
93
|
* The name a run set in `fontDict` will ask for: its `/BaseFont` without the
|
|
@@ -105,6 +116,34 @@ function embeddedFontName(file, fontDict) {
|
|
|
105
116
|
function hasLiftableProgram(file, fontDict) {
|
|
106
117
|
return fontProgram(file, fontDict) !== void 0;
|
|
107
118
|
}
|
|
119
|
+
/**
|
|
120
|
+
* The style a font's embedded program states for itself — the `head` table's
|
|
121
|
+
* macStyle, bit 0 bold and bit 1 italic — where the file carries a TrueType
|
|
122
|
+
* program to read it from.
|
|
123
|
+
*
|
|
124
|
+
* @param file The document.
|
|
125
|
+
* @param fontDict The font dictionary.
|
|
126
|
+
* @returns The program's own style, or `undefined` where there is no program
|
|
127
|
+
* or it has no header.
|
|
128
|
+
*/
|
|
129
|
+
function programStyle(file, fontDict) {
|
|
130
|
+
const program = fontProgram(file, fontDict);
|
|
131
|
+
if (!program || program.length < 12) return void 0;
|
|
132
|
+
const view = new DataView(program.buffer, program.byteOffset, program.byteLength);
|
|
133
|
+
const tables = view.getUint16(4);
|
|
134
|
+
for (let i = 0; i < tables; i++) {
|
|
135
|
+
const at = 12 + i * 16;
|
|
136
|
+
if (at + 16 > program.length) return void 0;
|
|
137
|
+
if (String.fromCharCode(...program.subarray(at, at + 4)) !== "head") continue;
|
|
138
|
+
const offset = view.getUint32(at + 8);
|
|
139
|
+
if (offset + 46 > program.length) return void 0;
|
|
140
|
+
const macStyle = view.getUint16(offset + 44);
|
|
141
|
+
return {
|
|
142
|
+
bold: (macStyle & 1) !== 0,
|
|
143
|
+
italic: (macStyle & 2) !== 0
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
}
|
|
108
147
|
/** §9.9 `/FontFile2` — the TrueType program, off the font or its descendant. */
|
|
109
148
|
function fontProgram(file, fontDict) {
|
|
110
149
|
const owner = descendant(file, fontDict) ?? fontDict;
|
|
@@ -123,4 +162,4 @@ function descendant(file, fontDict) {
|
|
|
123
162
|
return first instanceof Map ? first : void 0;
|
|
124
163
|
}
|
|
125
164
|
//#endregion
|
|
126
|
-
export { collectEmbeddedFonts, embeddedFontName, hasLiftableProgram };
|
|
165
|
+
export { collectEmbeddedFonts, eachPageFont, embeddedFontName, hasLiftableProgram, programStyle };
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { BodyElement, ParagraphProperties, Section, SectionProperties, TextOutline } from '../core/document-model/index.js';
|
|
2
|
-
import { FlowDoc } from '../core/ir/flow.js';
|
|
2
|
+
import { FaceFamily, FlowDoc } from '../core/ir/flow.js';
|
|
3
3
|
import { FontRegistry } from '../core/font/index.js';
|
|
4
|
-
import { Loss, ResourceStore } from '../core/ir/index.js';
|
|
4
|
+
import { Loss, Pt, ResourceStore } from '../core/ir/index.js';
|
|
5
5
|
import { PdfImage } from './images.js';
|
|
6
6
|
import { PdfPage } from './document.js';
|
|
7
7
|
import { PdfVector } from './vector.js';
|
|
@@ -36,6 +36,20 @@ export interface PageFrame {
|
|
|
36
36
|
*/
|
|
37
37
|
export declare function paragraphBlock(text: string, outlineLevel?: number): BodyElement;
|
|
38
38
|
/** One piece of reconstructed text, carrying any hyperlink (E-PDF EP8). */
|
|
39
|
+
/**
|
|
40
|
+
* The space set between a span and what follows it, in the span's own size and
|
|
41
|
+
* face: a word space is as wide as the type it stands in.
|
|
42
|
+
*
|
|
43
|
+
* Written bare, the space took the document's default size, whatever the words
|
|
44
|
+
* around it were set at: issue10665_reduced.pdf sets "78" and "110" twenty
|
|
45
|
+
* points apart in 60-point type, and the eleven-point space between them closed
|
|
46
|
+
* them up to "78110"; in 7-point footnotes the same space is half as wide
|
|
47
|
+
* again as the page's, and pushes their lines over.
|
|
48
|
+
*
|
|
49
|
+
* @param before The span the space follows, where there is one.
|
|
50
|
+
* @returns The space.
|
|
51
|
+
*/
|
|
52
|
+
export declare function spaceAfter(before: TextSpan | undefined): TextSpan;
|
|
39
53
|
export interface TextSpan {
|
|
40
54
|
readonly text: string;
|
|
41
55
|
readonly href?: string;
|
|
@@ -66,7 +80,21 @@ export interface TextSpan {
|
|
|
66
80
|
* survives as its own run) and squashing whitespace. With no hrefs this
|
|
67
81
|
* collapses to a single run — the same shape {@link paragraphBlock} produces.
|
|
68
82
|
*/
|
|
69
|
-
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number, placement?: Pick<ParagraphProperties, 'alignment' | 'spacingBefore' | 'indentLeft' | 'indentFirstLine' | 'tabs'>): BodyElement;
|
|
83
|
+
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number, placement?: Pick<ParagraphProperties, 'alignment' | 'spacingBefore' | 'spacingLine' | 'spacingLineRule' | 'indentLeft' | 'indentRight' | 'indentFirstLine' | 'tabs'>): BodyElement;
|
|
84
|
+
/**
|
|
85
|
+
* The height of a line that carries nothing to read: one twip, the least
|
|
86
|
+
* §17.3.1.33 can state. Zero is not a height a writer states at all — it is
|
|
87
|
+
* read as "no line spacing given", and the reader's single spacing comes back.
|
|
88
|
+
*/
|
|
89
|
+
export declare const CARRIER_LINE_PT: Pt;
|
|
90
|
+
/**
|
|
91
|
+
* §17.3.1.33 — the paragraph a floating mark is anchored in, which takes no
|
|
92
|
+
* room: the mark stands where the page drew it and the line carrying it is on
|
|
93
|
+
* no page. Left at a reader's single spacing, every rule and fill an invoice
|
|
94
|
+
* draws was a blank line in the flow, and the text under it moved down a line
|
|
95
|
+
* for each.
|
|
96
|
+
*/
|
|
97
|
+
export declare const FLOAT_CARRIER: ParagraphProperties;
|
|
70
98
|
/**
|
|
71
99
|
* Store a {@link PdfImage}'s bytes (content-addressed dedup) and build the image
|
|
72
100
|
* {@link BodyElement} that references them, sized in points from the placement
|
|
@@ -123,6 +151,69 @@ export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: num
|
|
|
123
151
|
* `A4`. Returns `undefined` when there is no usable first-page box.
|
|
124
152
|
*/
|
|
125
153
|
export declare function sectionFromPdfPages(pages: ReadonlyArray<PdfPage>): SectionProperties | undefined;
|
|
154
|
+
/** How many lines a sheet must hold before where they end says where its measure does. */
|
|
155
|
+
export declare const MEASURE_LINES = 3;
|
|
156
|
+
/**
|
|
157
|
+
* How far in, as a share of the sheet, a right margin the lines do not show
|
|
158
|
+
* may come: what a sheet of a line or two is re-set across is at least the
|
|
159
|
+
* rest of it.
|
|
160
|
+
*/
|
|
161
|
+
export declare const GUESSED_MARGIN: number;
|
|
162
|
+
/**
|
|
163
|
+
* §17.3.1.33 — where the baseline of an EXACT line stands in its box, from the
|
|
164
|
+
* top. LibreOffice puts it at four fifths of the height; Word at the height
|
|
165
|
+
* less the face's descent, which for the faces documents use is within a tenth
|
|
166
|
+
* of a line of the same place.
|
|
167
|
+
*/
|
|
168
|
+
export declare const BASELINE_AT = 0.8;
|
|
169
|
+
/** A line's box where no pitch was measured: the ordinary single spacing, in ems. */
|
|
170
|
+
export declare const NATURAL_LINE_EM = 1.2;
|
|
171
|
+
/**
|
|
172
|
+
* The size a document's text is set in: the middle of every size its runs
|
|
173
|
+
* carry.
|
|
174
|
+
*
|
|
175
|
+
* @param pageRuns Each page's runs.
|
|
176
|
+
* @returns The size, in points; 0 where no run states one.
|
|
177
|
+
*/
|
|
178
|
+
export declare function textSizeOf(pageRuns: ReadonlyArray<ReadonlyArray<TextRun>>): number;
|
|
179
|
+
/**
|
|
180
|
+
* Whether a run is set too small to be read, beside the text of its document —
|
|
181
|
+
* a mark the producer leaves on the sheet, not a line of the page.
|
|
182
|
+
*
|
|
183
|
+
* TCPDF signs the last page of everything it makes "Powered by TCPDF
|
|
184
|
+
* (www.tcpdf.org)" in type one point high, three points from the corner of the
|
|
185
|
+
* paper. Taken for text it was the leftmost and the lowest thing on the page:
|
|
186
|
+
* the margins came in at the edge of the sheet and every line of basicapi.pdf
|
|
187
|
+
* and alphatrans.pdf was set against it.
|
|
188
|
+
*
|
|
189
|
+
* @param run The run.
|
|
190
|
+
* @param textSize The size the document's text is set in (see {@link textSizeOf}).
|
|
191
|
+
* @returns `true` where the run is a mark rather than text.
|
|
192
|
+
*/
|
|
193
|
+
export declare function tooSmallToRead(run: TextRun, textSize: number): boolean;
|
|
194
|
+
/**
|
|
195
|
+
* The margins the SOURCE used, measured off where its words actually sit.
|
|
196
|
+
*
|
|
197
|
+
* A PDF states none — text is placed anywhere on the MediaBox — so the reader
|
|
198
|
+
* used to leave them at zero rather than invent an inch. But the words
|
|
199
|
+
* themselves say where the margin was: the leftmost glyph on the page is the
|
|
200
|
+
* left margin, and reflowing inside it keeps the measure the author set instead
|
|
201
|
+
* of running the text from edge to edge.
|
|
202
|
+
*
|
|
203
|
+
* Measured on the MEDIAN page rather than the extreme one, so a single full-
|
|
204
|
+
* bleed rule or a page number in the corner does not collapse the margin for
|
|
205
|
+
* the whole document, and clamped so a strange page cannot leave no text area
|
|
206
|
+
* at all.
|
|
207
|
+
*
|
|
208
|
+
* @param section The section the page box gave, or `undefined`.
|
|
209
|
+
* @param shown Each page as it is shown, for its own width and height.
|
|
210
|
+
* @param pageRuns Each page's runs, already placed on the shown page.
|
|
211
|
+
* @param pageMarks Each page's pictures, which the measure has to hold.
|
|
212
|
+
* @param foot The running foot lifted off the pages, as the first page
|
|
213
|
+
* showed it — the band the text block stands above.
|
|
214
|
+
* @returns The section with measured margins, or `section` when nothing is
|
|
215
|
+
* measurable.
|
|
216
|
+
*/
|
|
126
217
|
export declare function withMeasuredMargins(section: SectionProperties | undefined, shown: ReadonlyArray<{
|
|
127
218
|
width: number;
|
|
128
219
|
height: number;
|
|
@@ -131,7 +222,7 @@ export declare function withMeasuredMargins(section: SectionProperties | undefin
|
|
|
131
222
|
y: number;
|
|
132
223
|
widthPt: number;
|
|
133
224
|
heightPt: number;
|
|
134
|
-
}
|
|
225
|
+
}>>, foot?: ReadonlyArray<TextRun>): SectionProperties | undefined;
|
|
135
226
|
/**
|
|
136
227
|
* Assemble the final {@link FlowDoc} for a reconstruction: the body elements
|
|
137
228
|
* with their styles resolved against the empty style sheet, the lifted-image
|
|
@@ -139,4 +230,4 @@ export declare function withMeasuredMargins(section: SectionProperties | undefin
|
|
|
139
230
|
* both reconstruction paths (the tagged fast-path EP3 and the heuristic layout
|
|
140
231
|
* path EP4).
|
|
141
232
|
*/
|
|
142
|
-
export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore, section?: SectionProperties, embeddedFonts?: ReadonlyMap<string, FontRegistry>, sections?: ReadonlyArray<Section>, headersFooters?: ReadonlyMap<string, ReadonlyArray<BodyElement
|
|
233
|
+
export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore, section?: SectionProperties, embeddedFonts?: ReadonlyMap<string, FontRegistry>, sections?: ReadonlyArray<Section>, headersFooters?: ReadonlyMap<string, ReadonlyArray<BodyElement>>, faceFamilies?: ReadonlyMap<string, FaceFamily>): FlowDoc;
|