reamkit 1.25.1 → 1.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -5
- package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
- package/dist/esm/pdf-reader/annot-draw.js +374 -0
- package/dist/esm/pdf-reader/annots.d.ts +3 -1
- package/dist/esm/pdf-reader/annots.js +18 -4
- package/dist/esm/pdf-reader/ccitt.d.ts +18 -0
- package/dist/esm/pdf-reader/ccitt.js +70 -2
- package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
- package/dist/esm/pdf-reader/cie-color.js +112 -0
- package/dist/esm/pdf-reader/content.d.ts +47 -17
- package/dist/esm/pdf-reader/content.js +175 -14
- package/dist/esm/pdf-reader/display.d.ts +1 -1
- package/dist/esm/pdf-reader/display.js +21 -6
- package/dist/esm/pdf-reader/document.d.ts +6 -0
- package/dist/esm/pdf-reader/document.js +29 -1
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +12 -3
- package/dist/esm/pdf-reader/embedded-fonts.js +22 -3
- package/dist/esm/pdf-reader/flow-build.d.ts +12 -4
- package/dist/esm/pdf-reader/flow-build.js +102 -8
- package/dist/esm/pdf-reader/font.js +97 -16
- package/dist/esm/pdf-reader/function.d.ts +16 -0
- package/dist/esm/pdf-reader/function.js +414 -0
- package/dist/esm/pdf-reader/glyph-names.d.ts +6 -0
- package/dist/esm/pdf-reader/glyph-names.js +408 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +8 -4
- package/dist/esm/pdf-reader/image-decode.js +224 -12
- package/dist/esm/pdf-reader/images.d.ts +12 -0
- package/dist/esm/pdf-reader/images.js +109 -9
- package/dist/esm/pdf-reader/jbig2.d.ts +109 -0
- package/dist/esm/pdf-reader/jbig2.js +2606 -0
- package/dist/esm/pdf-reader/layout.d.ts +32 -0
- package/dist/esm/pdf-reader/layout.js +189 -40
- package/dist/esm/pdf-reader/lexer.d.ts +2 -0
- package/dist/esm/pdf-reader/lexer.js +4 -0
- package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
- package/dist/esm/pdf-reader/optional-content.js +93 -0
- package/dist/esm/pdf-reader/reader.d.ts +5 -2
- package/dist/esm/pdf-reader/reader.js +80 -7
- package/dist/esm/pdf-reader/shading.d.ts +71 -6
- package/dist/esm/pdf-reader/shading.js +185 -15
- package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
- package/dist/esm/pdf-reader/standard-metrics.js +18 -0
- package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
- package/dist/esm/pdf-reader/standard-widths.js +62 -0
- package/dist/esm/pdf-reader/tagged.js +204 -32
- package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
- package/dist/esm/pdf-reader/text-rules.js +112 -0
- package/dist/esm/pdf-reader/text.js +78 -4
- package/dist/esm/pdf-reader/vector.d.ts +5 -0
- package/dist/esm/pdf-reader/vector.js +39 -25
- package/dist/esm/word/docx-writer.js +213 -23
- package/package.json +1 -1
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { PdfDict } from '../pdf/objects.js';
|
|
2
2
|
import { PdfFile, PdfPage } from './document.js';
|
|
3
|
+
import { Loss } from '../core/ir/index.js';
|
|
3
4
|
import { FontRegistry } from '../core/font/index.js';
|
|
4
5
|
/**
|
|
5
6
|
* Every embedded TrueType program the pages use, by the name a run will ask for
|
|
@@ -15,11 +16,19 @@ import { FontRegistry } from '../core/font/index.js';
|
|
|
15
16
|
* (`/FontFile`) are different formats that the layout's parser does not take;
|
|
16
17
|
* those keep their substitute.
|
|
17
18
|
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
19
|
+
* And only one whose OUTLINES this pipeline can carry: a program with no
|
|
20
|
+
* `glyf`/`loca` cannot be subset, and a face offered here is one the writer
|
|
21
|
+
* will be asked to embed. bug1186827.pdf ships an OpenType/CFF program under
|
|
22
|
+
* `/FontFile2`, which parses like any other and then killed the whole
|
|
23
|
+
* conversion — "Subsetting requires a TrueType font with glyf+loca tables" —
|
|
24
|
+
* rather than losing one face.
|
|
25
|
+
*
|
|
26
|
+
* @param file The owning file.
|
|
27
|
+
* @param pages The pages whose fonts are wanted.
|
|
28
|
+
* @param losses Appended to for a face the page embeds and this cannot carry.
|
|
20
29
|
* @returns Name → a one-face registry holding that program.
|
|
21
30
|
*/
|
|
22
|
-
export declare function collectEmbeddedFonts(file: PdfFile, pages: ReadonlyArray<PdfPage>): Map<string, FontRegistry>;
|
|
31
|
+
export declare function collectEmbeddedFonts(file: PdfFile, pages: ReadonlyArray<PdfPage>, losses?: Array<Loss>): Map<string, FontRegistry>;
|
|
23
32
|
/**
|
|
24
33
|
* The name a run set in `fontDict` will ask for: its `/BaseFont` without the
|
|
25
34
|
* six-capital subset prefix (§9.6.4), lowercased.
|
|
@@ -1,5 +1,7 @@
|
|
|
1
|
+
import { parseTtf } from "../core/font/ttf-parser.js";
|
|
1
2
|
import { FontRegistry } from "../core/font/font-registry.js";
|
|
2
3
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
4
|
+
import { FEATURES } from "../core/ir/features.js";
|
|
3
5
|
//#region src/pdf-reader/embedded-fonts.ts
|
|
4
6
|
var MAX_FORM_DEPTH = 8;
|
|
5
7
|
/**
|
|
@@ -16,11 +18,19 @@ var MAX_FORM_DEPTH = 8;
|
|
|
16
18
|
* (`/FontFile`) are different formats that the layout's parser does not take;
|
|
17
19
|
* those keep their substitute.
|
|
18
20
|
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
+
* And only one whose OUTLINES this pipeline can carry: a program with no
|
|
22
|
+
* `glyf`/`loca` cannot be subset, and a face offered here is one the writer
|
|
23
|
+
* will be asked to embed. bug1186827.pdf ships an OpenType/CFF program under
|
|
24
|
+
* `/FontFile2`, which parses like any other and then killed the whole
|
|
25
|
+
* conversion — "Subsetting requires a TrueType font with glyf+loca tables" —
|
|
26
|
+
* rather than losing one face.
|
|
27
|
+
*
|
|
28
|
+
* @param file The owning file.
|
|
29
|
+
* @param pages The pages whose fonts are wanted.
|
|
30
|
+
* @param losses Appended to for a face the page embeds and this cannot carry.
|
|
21
31
|
* @returns Name → a one-face registry holding that program.
|
|
22
32
|
*/
|
|
23
|
-
function collectEmbeddedFonts(file, pages) {
|
|
33
|
+
function collectEmbeddedFonts(file, pages, losses) {
|
|
24
34
|
const out = /* @__PURE__ */ new Map();
|
|
25
35
|
const seen = /* @__PURE__ */ new Set();
|
|
26
36
|
const visiting = /* @__PURE__ */ new Set();
|
|
@@ -32,6 +42,15 @@ function collectEmbeddedFonts(file, pages) {
|
|
|
32
42
|
const program = fontProgram(file, fontDict);
|
|
33
43
|
if (!program) return;
|
|
34
44
|
try {
|
|
45
|
+
const parsed = parseTtf(program);
|
|
46
|
+
if (!parsed.tables.has("glyf") || !parsed.tables.has("loca")) {
|
|
47
|
+
losses?.push({
|
|
48
|
+
severity: "degraded",
|
|
49
|
+
feature: FEATURES.text,
|
|
50
|
+
detail: `embedded font ${name} has no TrueType outlines (a CFF program under /FontFile2); its text is re-set in a substitute face`
|
|
51
|
+
});
|
|
52
|
+
return;
|
|
53
|
+
}
|
|
35
54
|
out.set(name, FontRegistry.fromBytes({ regular: program }));
|
|
36
55
|
} catch {}
|
|
37
56
|
};
|
|
@@ -1,10 +1,12 @@
|
|
|
1
|
-
import { BodyElement, SectionProperties, TextOutline } from '../core/document-model/index.js';
|
|
1
|
+
import { BodyElement, ParagraphProperties, SectionProperties, TextOutline } from '../core/document-model/index.js';
|
|
2
2
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
3
3
|
import { FontRegistry } from '../core/font/index.js';
|
|
4
4
|
import { Loss, ResourceStore } from '../core/ir/index.js';
|
|
5
5
|
import { PdfImage } from './images.js';
|
|
6
6
|
import { PdfPage } from './document.js';
|
|
7
7
|
import { PdfVector } from './vector.js';
|
|
8
|
+
import { TextMarkup } from './annot-draw.js';
|
|
9
|
+
import { TextRun } from './content.js';
|
|
8
10
|
/**
|
|
9
11
|
* A reconstruction's document plus the losses incurred reading it (e.g. an
|
|
10
12
|
* undecodable image colour space) — surfaced through the reader's `LossReport`.
|
|
@@ -49,6 +51,8 @@ export interface TextSpan {
|
|
|
49
51
|
readonly bold?: boolean;
|
|
50
52
|
/** §9.8.1 — the face was a slanted one. */
|
|
51
53
|
readonly italic?: boolean;
|
|
54
|
+
/** §12.5.6.10 — a text-markup annotation marks these words. */
|
|
55
|
+
readonly markup?: TextMarkup;
|
|
52
56
|
}
|
|
53
57
|
/**
|
|
54
58
|
* Build a paragraph {@link BodyElement} from positioned {@link TextSpan}s,
|
|
@@ -56,13 +60,13 @@ export interface TextSpan {
|
|
|
56
60
|
* survives as its own run) and squashing whitespace. With no hrefs this
|
|
57
61
|
* collapses to a single run — the same shape {@link paragraphBlock} produces.
|
|
58
62
|
*/
|
|
59
|
-
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number): BodyElement;
|
|
63
|
+
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number, placement?: Pick<ParagraphProperties, 'alignment' | 'spacingBefore'>): BodyElement;
|
|
60
64
|
/**
|
|
61
65
|
* Store a {@link PdfImage}'s bytes (content-addressed dedup) and build the image
|
|
62
66
|
* {@link BodyElement} that references them, sized in points from the placement
|
|
63
67
|
* CTM. `alt` becomes the block's alt text when given.
|
|
64
68
|
*/
|
|
65
|
-
export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string, frame?: PageFrame, zOrder?: number): BodyElement;
|
|
69
|
+
export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string, frame?: PageFrame, zOrder?: number, behind?: boolean): BodyElement;
|
|
66
70
|
/**
|
|
67
71
|
* A line of text as an anchored box, standing where the page set it.
|
|
68
72
|
*
|
|
@@ -100,7 +104,7 @@ export declare function dedupeLosses(losses: ReadonlyArray<Loss>): Array<Loss>;
|
|
|
100
104
|
* a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
|
|
101
105
|
* its forty-nine paths one under another spilled it onto a second page.
|
|
102
106
|
*/
|
|
103
|
-
export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: number): BodyElement;
|
|
107
|
+
export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: number, behind?: boolean): BodyElement;
|
|
104
108
|
/**
|
|
105
109
|
* Derive the {@link SectionProperties} geometry from the source pages so a
|
|
106
110
|
* reconstructed PDF re-renders at its real page size and orientation rather than
|
|
@@ -113,6 +117,10 @@ export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: num
|
|
|
113
117
|
* `A4`. Returns `undefined` when there is no usable first-page box.
|
|
114
118
|
*/
|
|
115
119
|
export declare function sectionFromPdfPages(pages: ReadonlyArray<PdfPage>): SectionProperties | undefined;
|
|
120
|
+
export declare function withMeasuredMargins(section: SectionProperties | undefined, shown: ReadonlyArray<{
|
|
121
|
+
width: number;
|
|
122
|
+
height: number;
|
|
123
|
+
}>, pageRuns: ReadonlyArray<ReadonlyArray<TextRun>>): SectionProperties | undefined;
|
|
116
124
|
/**
|
|
117
125
|
* Assemble the final {@link FlowDoc} for a reconstruction: the body elements
|
|
118
126
|
* with their styles resolved against the empty style sheet, the lifted-image
|
|
@@ -27,11 +27,11 @@ function paragraphBlock(text, outlineLevel) {
|
|
|
27
27
|
* survives as its own run) and squashing whitespace. With no hrefs this
|
|
28
28
|
* collapses to a single run — the same shape {@link paragraphBlock} produces.
|
|
29
29
|
*/
|
|
30
|
-
function paragraphFromRuns(spans, outlineLevel) {
|
|
30
|
+
function paragraphFromRuns(spans, outlineLevel, placement) {
|
|
31
31
|
const merged = [];
|
|
32
32
|
for (const s of spans) {
|
|
33
33
|
const last = merged[merged.length - 1];
|
|
34
|
-
if (last && last.href === s.href && last.sizePt === s.sizePt && last.colorHex === s.colorHex && last.fontName === s.fontName && last.outline?.colorHex === s.outline?.colorHex && last.outline?.widthPt === s.outline?.widthPt && last.bold === s.bold && last.italic === s.italic) last.text += s.text;
|
|
34
|
+
if (last && last.href === s.href && last.sizePt === s.sizePt && last.colorHex === s.colorHex && last.fontName === s.fontName && last.outline?.colorHex === s.outline?.colorHex && last.outline?.widthPt === s.outline?.widthPt && last.bold === s.bold && last.italic === s.italic && sameMarkup(last.markup, s.markup)) last.text += s.text;
|
|
35
35
|
else merged.push({
|
|
36
36
|
text: s.text,
|
|
37
37
|
...s.href !== void 0 ? { href: s.href } : {},
|
|
@@ -40,7 +40,8 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
40
40
|
...s.fontName !== void 0 ? { fontName: s.fontName } : {},
|
|
41
41
|
...s.outline !== void 0 ? { outline: s.outline } : {},
|
|
42
42
|
...s.bold !== void 0 ? { bold: s.bold } : {},
|
|
43
|
-
...s.italic !== void 0 ? { italic: s.italic } : {}
|
|
43
|
+
...s.italic !== void 0 ? { italic: s.italic } : {},
|
|
44
|
+
...s.markup !== void 0 ? { markup: s.markup } : {}
|
|
44
45
|
});
|
|
45
46
|
}
|
|
46
47
|
const runs = merged.map((m) => ({
|
|
@@ -54,7 +55,10 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
54
55
|
return {
|
|
55
56
|
kind: "paragraph",
|
|
56
57
|
paragraph: {
|
|
57
|
-
properties:
|
|
58
|
+
properties: {
|
|
59
|
+
...outlineLevel !== void 0 ? { outlineLevel } : {},
|
|
60
|
+
...placement
|
|
61
|
+
},
|
|
58
62
|
runs: runs.filter((r) => r.text.length > 0).map((r) => ({
|
|
59
63
|
text: r.text,
|
|
60
64
|
properties: {
|
|
@@ -63,22 +67,31 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
63
67
|
...r.fontName !== void 0 ? { fontFamily: { ascii: r.fontName } } : {},
|
|
64
68
|
...r.outline !== void 0 ? { textOutline: r.outline } : {},
|
|
65
69
|
...r.bold ? { bold: true } : {},
|
|
66
|
-
...r.italic ? { italic: true } : {}
|
|
70
|
+
...r.italic ? { italic: true } : {},
|
|
71
|
+
...r.markup?.highlightHex !== void 0 ? { shadingColorHex: r.markup.highlightHex } : {},
|
|
72
|
+
...r.markup?.underline !== void 0 ? { underline: r.markup.underline } : {},
|
|
73
|
+
...r.markup?.underlineHex !== void 0 ? { underlineColorHex: r.markup.underlineHex } : {},
|
|
74
|
+
...r.markup?.strike === true ? { strike: true } : {}
|
|
67
75
|
},
|
|
68
76
|
...r.href ? { href: r.href } : {}
|
|
69
77
|
}))
|
|
70
78
|
}
|
|
71
79
|
};
|
|
72
80
|
}
|
|
81
|
+
/** Whether two runs are marked the same way, so they may join into one. */
|
|
82
|
+
function sameMarkup(a, b) {
|
|
83
|
+
return a?.highlightHex === b?.highlightHex && a?.underline === b?.underline && a?.underlineHex === b?.underlineHex && a?.strike === b?.strike;
|
|
84
|
+
}
|
|
73
85
|
/**
|
|
74
86
|
* Store a {@link PdfImage}'s bytes (content-addressed dedup) and build the image
|
|
75
87
|
* {@link BodyElement} that references them, sized in points from the placement
|
|
76
88
|
* CTM. `alt` becomes the block's alt text when given.
|
|
77
89
|
*/
|
|
78
|
-
function imageBlock(image, resources, alt, frame, zOrder) {
|
|
90
|
+
function imageBlock(image, resources, alt, frame, zOrder, behind = false) {
|
|
79
91
|
const resource = resources.put(image.bytes);
|
|
80
92
|
const float = frame !== void 0 ? {
|
|
81
93
|
wrap: "none",
|
|
94
|
+
...behind ? { behind: true } : {},
|
|
82
95
|
...zOrder !== void 0 ? { zOrder } : {},
|
|
83
96
|
posH: {
|
|
84
97
|
relativeFrom: "page",
|
|
@@ -96,6 +109,8 @@ function imageBlock(image, resources, alt, frame, zOrder) {
|
|
|
96
109
|
resource,
|
|
97
110
|
width: pt(image.widthPt),
|
|
98
111
|
height: pt(image.heightPt),
|
|
112
|
+
...image.rotationDeg !== void 0 ? { rotation60k: Math.round(-image.rotationDeg * 6e4) } : {},
|
|
113
|
+
...image.crop ? { crop: image.crop } : {},
|
|
99
114
|
paragraphProperties: {},
|
|
100
115
|
...alt ? { altText: alt } : {}
|
|
101
116
|
}
|
|
@@ -172,7 +187,7 @@ function dedupeLosses(losses) {
|
|
|
172
187
|
* a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
|
|
173
188
|
* its forty-nine paths one under another spilled it onto a second page.
|
|
174
189
|
*/
|
|
175
|
-
function shapeBlock(v, frame, zOrder) {
|
|
190
|
+
function shapeBlock(v, frame, zOrder, behind = false) {
|
|
176
191
|
const w = v.maxX - v.minX;
|
|
177
192
|
const h = v.maxY - v.minY;
|
|
178
193
|
const fx = (x) => x - v.minX;
|
|
@@ -222,6 +237,7 @@ function shapeBlock(v, frame, zOrder) {
|
|
|
222
237
|
} : void 0;
|
|
223
238
|
const float = frame !== void 0 ? {
|
|
224
239
|
wrap: "none",
|
|
240
|
+
...behind || v.darkens === true ? { behind: true } : {},
|
|
225
241
|
...zOrder !== void 0 ? { zOrder } : {},
|
|
226
242
|
posH: {
|
|
227
243
|
relativeFrom: "page",
|
|
@@ -287,6 +303,84 @@ function sectionFromPdfPages(pages) {
|
|
|
287
303
|
};
|
|
288
304
|
}
|
|
289
305
|
/**
|
|
306
|
+
* The margins the SOURCE used, measured off where its words actually sit.
|
|
307
|
+
*
|
|
308
|
+
* A PDF states none — text is placed anywhere on the MediaBox — so the reader
|
|
309
|
+
* used to leave them at zero rather than invent an inch. But the words
|
|
310
|
+
* themselves say where the margin was: the leftmost glyph on the page is the
|
|
311
|
+
* left margin, and reflowing inside it keeps the measure the author set instead
|
|
312
|
+
* of running the text from edge to edge.
|
|
313
|
+
*
|
|
314
|
+
* Measured on the MEDIAN page rather than the extreme one, so a single full-
|
|
315
|
+
* bleed rule or a page number in the corner does not collapse the margin for
|
|
316
|
+
* the whole document, and clamped so a strange page cannot leave no text area
|
|
317
|
+
* at all.
|
|
318
|
+
*
|
|
319
|
+
* @param section The section the page box gave, or `undefined`.
|
|
320
|
+
* @param shown Each page as it is shown, for its own width and height.
|
|
321
|
+
* @param pageRuns Each page's runs, already placed on the shown page.
|
|
322
|
+
* @returns The section with measured margins, or `section` when nothing is
|
|
323
|
+
* measurable.
|
|
324
|
+
*/
|
|
325
|
+
/** What the measure gives back, so the widest line still fits when re-set. */
|
|
326
|
+
var SLACK = .01;
|
|
327
|
+
/** How far a face's ascender stands above its baseline, as a fraction of the size. */
|
|
328
|
+
var ASCENDER = .8;
|
|
329
|
+
/** And its descender below — the two together are a little over one em. */
|
|
330
|
+
var DESCENDER = .22;
|
|
331
|
+
function withMeasuredMargins(section, shown, pageRuns) {
|
|
332
|
+
if (!section?.pageSize) return section;
|
|
333
|
+
const width = section.pageSize.width;
|
|
334
|
+
const height = section.pageSize.height;
|
|
335
|
+
const lefts = [];
|
|
336
|
+
const rights = [];
|
|
337
|
+
const tops = [];
|
|
338
|
+
const bottoms = [];
|
|
339
|
+
pageRuns.forEach((runs, i) => {
|
|
340
|
+
const page = shown[i];
|
|
341
|
+
if (!page || runs.length === 0) return;
|
|
342
|
+
let minX = Infinity;
|
|
343
|
+
let maxX = -Infinity;
|
|
344
|
+
let minY = Infinity;
|
|
345
|
+
let maxY = -Infinity;
|
|
346
|
+
let topSize = 0;
|
|
347
|
+
let bottomSize = 0;
|
|
348
|
+
for (const r of runs) {
|
|
349
|
+
if (!Number.isFinite(r.x) || !Number.isFinite(r.y)) continue;
|
|
350
|
+
minX = Math.min(minX, r.x);
|
|
351
|
+
maxX = Math.max(maxX, r.endX);
|
|
352
|
+
if (r.y < minY) {
|
|
353
|
+
minY = r.y;
|
|
354
|
+
bottomSize = r.fontSizePt;
|
|
355
|
+
}
|
|
356
|
+
if (r.y > maxY) {
|
|
357
|
+
maxY = r.y;
|
|
358
|
+
topSize = r.fontSizePt;
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
if (!Number.isFinite(minX) || !Number.isFinite(minY)) return;
|
|
362
|
+
lefts.push(minX);
|
|
363
|
+
rights.push(page.width - maxX);
|
|
364
|
+
tops.push(page.height - maxY - topSize * ASCENDER);
|
|
365
|
+
bottoms.push(minY - bottomSize * DESCENDER);
|
|
366
|
+
});
|
|
367
|
+
if (lefts.length === 0) return section;
|
|
368
|
+
const median = (xs) => {
|
|
369
|
+
const s = [...xs].sort((a, b) => a - b);
|
|
370
|
+
return s[Math.floor(s.length / 2)] ?? 0;
|
|
371
|
+
};
|
|
372
|
+
const clamp = (v, span) => pt(Math.max(0, Math.min(v, span / 3)));
|
|
373
|
+
return {
|
|
374
|
+
...section,
|
|
375
|
+
margins: {
|
|
376
|
+
left: clamp(median(lefts), width),
|
|
377
|
+
right: clamp(median(rights) - width * SLACK, width),
|
|
378
|
+
top: clamp(median(tops), height),
|
|
379
|
+
bottom: clamp(median(bottoms), height)
|
|
380
|
+
}
|
|
381
|
+
};
|
|
382
|
+
}
|
|
383
|
+
/**
|
|
290
384
|
* Assemble the final {@link FlowDoc} for a reconstruction: the body elements
|
|
291
385
|
* with their styles resolved against the empty style sheet, the lifted-image
|
|
292
386
|
* resource store, and the optional page {@link SectionProperties}. Shared by
|
|
@@ -305,4 +399,4 @@ function buildFlowDoc(body, resources = new ResourceStore(), section, embeddedFo
|
|
|
305
399
|
};
|
|
306
400
|
}
|
|
307
401
|
//#endregion
|
|
308
|
-
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock };
|
|
402
|
+
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock, withMeasuredMargins };
|
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
import { parseTtf } from "../core/font/ttf-parser.js";
|
|
2
2
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
3
|
-
import { embeddedFontName } from "./embedded-fonts.js";
|
|
4
3
|
import { parseToUnicodeCMap } from "./cmap.js";
|
|
4
|
+
import { textForGlyphName } from "./glyph-names.js";
|
|
5
|
+
import { standardFace, standardWidth } from "./standard-widths.js";
|
|
6
|
+
import { embeddedFontName } from "./embedded-fonts.js";
|
|
5
7
|
//#region src/pdf-reader/font.ts
|
|
6
8
|
/**
|
|
7
9
|
* Build a {@link ContentFont} (the interpreter's decode + advance hooks) from a
|
|
@@ -25,22 +27,59 @@ function buildContentFont(file, fontDict) {
|
|
|
25
27
|
toUnicode = parsed.map;
|
|
26
28
|
if (isType0) codeBytes = parsed.codeBytes;
|
|
27
29
|
}
|
|
28
|
-
const
|
|
30
|
+
const fromProgram = isType0 && toUnicode.size === 0 ? embeddedCmap(file, fontDict) : void 0;
|
|
31
|
+
const fromNames = !isType0 ? namedGlyphs(file, fontDict) : void 0;
|
|
32
|
+
const unicode = fromProgram ?? (toUnicode.size > 0 ? toUnicode : fromNames ?? toUnicode);
|
|
29
33
|
const bytesPerCode = codeBytes;
|
|
30
34
|
const style = faceStyle(file, fontDict, isType0);
|
|
31
35
|
const name = embeddedFontName(file, fontDict);
|
|
32
36
|
const type3 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type3" ? type3Face(file, fontDict) : void 0;
|
|
33
|
-
const
|
|
37
|
+
const speechless = isType0 && unicode.size === 0;
|
|
38
|
+
const decodeOne = (code) => unicode.get(code) ?? (bytesPerCode === 1 ? latin1(code) : speechless ? "�" : "");
|
|
39
|
+
const simple = simpleWidths(file, fontDict, decodeOne);
|
|
34
40
|
const width = isType0 ? cidWidths(file, fontDict) : type3 ? (code) => simple(code) * type3.matrix[0] * 1e3 : simple;
|
|
35
41
|
return {
|
|
36
42
|
bytesPerCode,
|
|
37
43
|
...type3 ? { type3 } : {},
|
|
38
44
|
...name !== void 0 ? { name } : {},
|
|
39
|
-
decode: (codes) => codes.map((c) =>
|
|
45
|
+
decode: (codes) => codes.map((c) => readable(decodeOne(c))).join(""),
|
|
40
46
|
width,
|
|
41
47
|
...style
|
|
42
48
|
};
|
|
43
49
|
}
|
|
50
|
+
/**
|
|
51
|
+
* The character a code stands for when nothing in the font says.
|
|
52
|
+
*
|
|
53
|
+
* Latin-1 is the only guess there is, and it is a good one for a text font —
|
|
54
|
+
* but a C0 control is not a glyph. §9.4.3 shows GLYPHS, and a code falling back
|
|
55
|
+
* to one has landed there by accident: issue11549_reduced.pdf's one line came
|
|
56
|
+
* back as U+0007 through U+0011 and was drawn as six empty boxes over a page
|
|
57
|
+
* that shows nothing at all. A `/ToUnicode` that STATES a control is stating
|
|
58
|
+
* something and is left alone.
|
|
59
|
+
*/
|
|
60
|
+
function latin1(code) {
|
|
61
|
+
return code < 32 || code === 127 ? "�" : String.fromCharCode(code);
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* What a `/ToUnicode` gives, less what Unicode says is not a character.
|
|
65
|
+
*
|
|
66
|
+
* `U+FFFE` and `U+FFFF` are noncharacters and the `U+FDD0`–`U+FDEF` block with
|
|
67
|
+
* them; a lone surrogate is half of a pair that never came. A producer that
|
|
68
|
+
* maps its glyphs to any of these has said "no text here" in the only way the
|
|
69
|
+
* format lets it, and carrying them on writes bytes no reader can show —
|
|
70
|
+
* arial_unicode_ab_cidfont.pdf maps its four Arabic letters to `U+FFFF` and the
|
|
71
|
+
* page came back holding four of them. They become `U+FFFD`, which the
|
|
72
|
+
* reconstruction counts and reports rather than passing along.
|
|
73
|
+
*/
|
|
74
|
+
function readable(text) {
|
|
75
|
+
let out = "";
|
|
76
|
+
for (const ch of text) {
|
|
77
|
+
const cp = ch.codePointAt(0) ?? 0;
|
|
78
|
+
const noncharacter = (cp & 65534) === 65534 || cp >= 64976 && cp <= 65007 || cp === 0 || cp >= SURROGATE_FIRST && cp <= SURROGATE_LAST;
|
|
79
|
+
out += noncharacter ? "�" : ch;
|
|
80
|
+
}
|
|
81
|
+
return out;
|
|
82
|
+
}
|
|
44
83
|
/** The last Unicode code point in the BMP, and the surrogate block inside it. */
|
|
45
84
|
var BMP_END = 65535;
|
|
46
85
|
var SURROGATE_FIRST = 55296;
|
|
@@ -148,6 +187,27 @@ function fontMatrix(file, fontDict) {
|
|
|
148
187
|
n[5]
|
|
149
188
|
];
|
|
150
189
|
}
|
|
190
|
+
/**
|
|
191
|
+
* §9.6.6.1 — code → text for a simple font, out of the glyph names its
|
|
192
|
+
* `/Encoding /Differences` gives.
|
|
193
|
+
*
|
|
194
|
+
* Returns `undefined` where the font names nothing, so the caller keeps its
|
|
195
|
+
* Latin-1 reading rather than replacing it with an empty map.
|
|
196
|
+
*/
|
|
197
|
+
function namedGlyphs(file, fontDict) {
|
|
198
|
+
const names = differences(file, fontDict);
|
|
199
|
+
if (names.size === 0) return void 0;
|
|
200
|
+
const out = /* @__PURE__ */ new Map();
|
|
201
|
+
for (const [code, name] of names) {
|
|
202
|
+
const text = textForGlyphName(name);
|
|
203
|
+
if (text !== void 0) out.set(code, text);
|
|
204
|
+
}
|
|
205
|
+
if (out.size > 0) return out;
|
|
206
|
+
if (asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) !== "Type3") return void 0;
|
|
207
|
+
const unreadable = /* @__PURE__ */ new Map();
|
|
208
|
+
for (const code of names.keys()) unreadable.set(code, "�");
|
|
209
|
+
return unreadable;
|
|
210
|
+
}
|
|
151
211
|
/** §9.6.6.1 `/Encoding` `/Differences` — code → glyph name, as the array runs. */
|
|
152
212
|
function differences(file, fontDict) {
|
|
153
213
|
const out = /* @__PURE__ */ new Map();
|
|
@@ -166,6 +226,8 @@ function differences(file, fontDict) {
|
|
|
166
226
|
/** §9.8.2 `/Flags` — bit 7 is Italic, bit 19 ForceBold (bits numbered from 1). */
|
|
167
227
|
var FLAG_ITALIC = 64;
|
|
168
228
|
var FLAG_FORCE_BOLD = 1 << 18;
|
|
229
|
+
/** §9.8.1 `/FontWeight` — the lightest a face may state; below it is no weight. */
|
|
230
|
+
var LIGHTEST_WEIGHT = 100;
|
|
169
231
|
/** §9.8.1 `/FontWeight` — 400 is normal, 700 bold; 600 is where "bold" begins. */
|
|
170
232
|
var BOLD_WEIGHT = 600;
|
|
171
233
|
/**
|
|
@@ -187,23 +249,38 @@ var BOLD_WEIGHT = 600;
|
|
|
187
249
|
function faceStyle(file, fontDict, isType0) {
|
|
188
250
|
const owner = isType0 ? descendantFont(file, fontDict) : fontDict;
|
|
189
251
|
const descriptor = file.resolve(owner.get("FontDescriptor") ?? PDF_NULL);
|
|
252
|
+
const named = styleFromName(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
|
|
190
253
|
if (descriptor instanceof Map) {
|
|
191
254
|
const flags = asNumber(file.resolve(descriptor.get("Flags") ?? PDF_NULL), 0);
|
|
192
|
-
const
|
|
255
|
+
const weightVal = file.resolve(descriptor.get("FontWeight") ?? PDF_NULL);
|
|
193
256
|
const slant = asNumber(file.resolve(descriptor.get("ItalicAngle") ?? PDF_NULL), 0);
|
|
194
|
-
const bold =
|
|
195
|
-
const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0;
|
|
257
|
+
const bold = typeof weightVal === "number" && weightVal >= LIGHTEST_WEIGHT ? asNumber(weightVal, 0) >= BOLD_WEIGHT : (flags & FLAG_FORCE_BOLD) !== 0 || named.bold;
|
|
258
|
+
const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0 || named.italic;
|
|
196
259
|
return {
|
|
197
|
-
...bold ? { bold } : {},
|
|
198
|
-
...italic ? { italic } : {}
|
|
260
|
+
...bold ? { bold: true } : {},
|
|
261
|
+
...italic ? { italic: true } : {}
|
|
199
262
|
};
|
|
200
263
|
}
|
|
201
|
-
const name = asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)).replace(/^[A-Z]{6}\+/u, "");
|
|
202
|
-
const bold = /bold|black|heavy/iu.test(name);
|
|
203
|
-
const italic = /italic|oblique/iu.test(name);
|
|
204
264
|
return {
|
|
205
|
-
...bold ? { bold } : {},
|
|
206
|
-
...italic ? { italic } : {}
|
|
265
|
+
...named.bold ? { bold: true } : {},
|
|
266
|
+
...named.italic ? { italic: true } : {}
|
|
267
|
+
};
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* §9.6.2.2 — the style a font's NAME states, by the PostScript convention:
|
|
271
|
+
* `Family-Style`, or `Family,Style` as Word writes it.
|
|
272
|
+
*
|
|
273
|
+
* The separator is what makes this safe. A family whose name merely CONTAINS
|
|
274
|
+
* the word — "New Basrah Bold", "Damascus Bold", both real faces in
|
|
275
|
+
* ArabicCIDTrueType.pdf — is not a bold cut of anything, and reading it as one
|
|
276
|
+
* set two lines heavy that no reader sets heavy. `Times-Bold` is.
|
|
277
|
+
*/
|
|
278
|
+
function styleFromName(baseFont) {
|
|
279
|
+
const name = baseFont.replace(/^[A-Z]{6}\+/u, "");
|
|
280
|
+
const style = /[-,]([A-Za-z]+)$/u.exec(name)?.[1] ?? "";
|
|
281
|
+
return {
|
|
282
|
+
bold: /bold|black|heavy|semib|demi/iu.test(style),
|
|
283
|
+
italic: /italic|oblique/iu.test(style)
|
|
207
284
|
};
|
|
208
285
|
}
|
|
209
286
|
/** §9.7.4 — a `/Type0` font's one descendant CIDFont, which owns the descriptor. */
|
|
@@ -212,15 +289,19 @@ function descendantFont(file, fontDict) {
|
|
|
212
289
|
const first = Array.isArray(descFonts) ? file.resolve(descFonts[0] ?? PDF_NULL) : PDF_NULL;
|
|
213
290
|
return first instanceof Map ? first : /* @__PURE__ */ new Map();
|
|
214
291
|
}
|
|
215
|
-
function simpleWidths(file, fontDict) {
|
|
292
|
+
function simpleWidths(file, fontDict, decodeOne) {
|
|
216
293
|
const first = asNumber(file.resolve(fontDict.get("FirstChar") ?? PDF_NULL), 0);
|
|
217
294
|
const widthsVal = file.resolve(fontDict.get("Widths") ?? PDF_NULL);
|
|
218
295
|
const widths = Array.isArray(widthsVal) ? widthsVal : [];
|
|
219
296
|
const descriptor = file.resolve(fontDict.get("FontDescriptor") ?? PDF_NULL);
|
|
220
297
|
const missing = descriptor instanceof Map ? asNumber(file.resolve(descriptor.get("MissingWidth") ?? PDF_NULL), 0) : 0;
|
|
298
|
+
const face = standardFace(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
|
|
221
299
|
return (code) => {
|
|
222
300
|
const w = widths[code - first];
|
|
223
|
-
|
|
301
|
+
if (typeof w === "number") return w;
|
|
302
|
+
const built = face === void 0 ? void 0 : standardWidth(face, code, decodeOne(code));
|
|
303
|
+
if (built !== void 0) return built;
|
|
304
|
+
return missing > 0 ? missing : 500;
|
|
224
305
|
};
|
|
225
306
|
}
|
|
226
307
|
function cidWidths(file, fontDict) {
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { PdfValue } from '../pdf/objects.js';
|
|
2
|
+
import { PdfFile } from './document.js';
|
|
3
|
+
/** §7.10 — m numbers in, n numbers out. */
|
|
4
|
+
export type PdfFunction = (inputs: ReadonlyArray<number>) => Array<number>;
|
|
5
|
+
/**
|
|
6
|
+
* Read a `/Function` entry into something callable (§7.10).
|
|
7
|
+
*
|
|
8
|
+
* The entry may also be an ARRAY of functions, each giving one output, which is
|
|
9
|
+
* what a `/DeviceN` with a per-colorant transform states; that comes back as
|
|
10
|
+
* one function returning all of them in order.
|
|
11
|
+
*
|
|
12
|
+
* @param file The owning file.
|
|
13
|
+
* @param value The `/Function` (or `/TintTransform`) entry, unresolved.
|
|
14
|
+
* @returns The function, or `undefined` for one this cannot run.
|
|
15
|
+
*/
|
|
16
|
+
export declare function readFunction(file: PdfFile, value: PdfValue | undefined): PdfFunction | undefined;
|