reamkit 1.26.0 → 1.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -4
- package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
- package/dist/esm/pdf-reader/annot-draw.js +374 -0
- package/dist/esm/pdf-reader/annots.d.ts +3 -1
- package/dist/esm/pdf-reader/annots.js +18 -4
- package/dist/esm/pdf-reader/ccitt.d.ts +18 -0
- package/dist/esm/pdf-reader/ccitt.js +70 -2
- package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
- package/dist/esm/pdf-reader/cie-color.js +112 -0
- package/dist/esm/pdf-reader/content.d.ts +47 -3
- package/dist/esm/pdf-reader/content.js +108 -13
- package/dist/esm/pdf-reader/display.d.ts +1 -1
- package/dist/esm/pdf-reader/display.js +21 -6
- package/dist/esm/pdf-reader/document.d.ts +6 -0
- package/dist/esm/pdf-reader/document.js +29 -1
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +12 -3
- package/dist/esm/pdf-reader/embedded-fonts.js +22 -3
- package/dist/esm/pdf-reader/flow-build.d.ts +12 -4
- package/dist/esm/pdf-reader/flow-build.js +102 -8
- package/dist/esm/pdf-reader/font.js +77 -16
- package/dist/esm/pdf-reader/function.d.ts +16 -0
- package/dist/esm/pdf-reader/function.js +414 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +8 -4
- package/dist/esm/pdf-reader/image-decode.js +192 -11
- package/dist/esm/pdf-reader/images.d.ts +12 -0
- package/dist/esm/pdf-reader/images.js +109 -9
- package/dist/esm/pdf-reader/jbig2.d.ts +23 -0
- package/dist/esm/pdf-reader/jbig2.js +23 -14
- package/dist/esm/pdf-reader/layout.d.ts +32 -0
- package/dist/esm/pdf-reader/layout.js +189 -40
- package/dist/esm/pdf-reader/lexer.d.ts +2 -0
- package/dist/esm/pdf-reader/lexer.js +4 -0
- package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
- package/dist/esm/pdf-reader/optional-content.js +93 -0
- package/dist/esm/pdf-reader/reader.d.ts +5 -2
- package/dist/esm/pdf-reader/reader.js +80 -7
- package/dist/esm/pdf-reader/shading.d.ts +60 -8
- package/dist/esm/pdf-reader/shading.js +124 -20
- package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
- package/dist/esm/pdf-reader/standard-metrics.js +18 -0
- package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
- package/dist/esm/pdf-reader/standard-widths.js +62 -0
- package/dist/esm/pdf-reader/tagged.js +204 -32
- package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
- package/dist/esm/pdf-reader/text-rules.js +112 -0
- package/dist/esm/pdf-reader/text.js +78 -4
- package/dist/esm/pdf-reader/vector.d.ts +5 -0
- package/dist/esm/pdf-reader/vector.js +39 -26
- package/dist/esm/word/docx-writer.js +73 -11
- package/package.json +1 -1
|
@@ -1,10 +1,12 @@
|
|
|
1
|
-
import { BodyElement, SectionProperties, TextOutline } from '../core/document-model/index.js';
|
|
1
|
+
import { BodyElement, ParagraphProperties, SectionProperties, TextOutline } from '../core/document-model/index.js';
|
|
2
2
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
3
3
|
import { FontRegistry } from '../core/font/index.js';
|
|
4
4
|
import { Loss, ResourceStore } from '../core/ir/index.js';
|
|
5
5
|
import { PdfImage } from './images.js';
|
|
6
6
|
import { PdfPage } from './document.js';
|
|
7
7
|
import { PdfVector } from './vector.js';
|
|
8
|
+
import { TextMarkup } from './annot-draw.js';
|
|
9
|
+
import { TextRun } from './content.js';
|
|
8
10
|
/**
|
|
9
11
|
* A reconstruction's document plus the losses incurred reading it (e.g. an
|
|
10
12
|
* undecodable image colour space) — surfaced through the reader's `LossReport`.
|
|
@@ -49,6 +51,8 @@ export interface TextSpan {
|
|
|
49
51
|
readonly bold?: boolean;
|
|
50
52
|
/** §9.8.1 — the face was a slanted one. */
|
|
51
53
|
readonly italic?: boolean;
|
|
54
|
+
/** §12.5.6.10 — a text-markup annotation marks these words. */
|
|
55
|
+
readonly markup?: TextMarkup;
|
|
52
56
|
}
|
|
53
57
|
/**
|
|
54
58
|
* Build a paragraph {@link BodyElement} from positioned {@link TextSpan}s,
|
|
@@ -56,13 +60,13 @@ export interface TextSpan {
|
|
|
56
60
|
* survives as its own run) and squashing whitespace. With no hrefs this
|
|
57
61
|
* collapses to a single run — the same shape {@link paragraphBlock} produces.
|
|
58
62
|
*/
|
|
59
|
-
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number): BodyElement;
|
|
63
|
+
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number, placement?: Pick<ParagraphProperties, 'alignment' | 'spacingBefore'>): BodyElement;
|
|
60
64
|
/**
|
|
61
65
|
* Store a {@link PdfImage}'s bytes (content-addressed dedup) and build the image
|
|
62
66
|
* {@link BodyElement} that references them, sized in points from the placement
|
|
63
67
|
* CTM. `alt` becomes the block's alt text when given.
|
|
64
68
|
*/
|
|
65
|
-
export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string, frame?: PageFrame, zOrder?: number): BodyElement;
|
|
69
|
+
export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string, frame?: PageFrame, zOrder?: number, behind?: boolean): BodyElement;
|
|
66
70
|
/**
|
|
67
71
|
* A line of text as an anchored box, standing where the page set it.
|
|
68
72
|
*
|
|
@@ -100,7 +104,7 @@ export declare function dedupeLosses(losses: ReadonlyArray<Loss>): Array<Loss>;
|
|
|
100
104
|
* a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
|
|
101
105
|
* its forty-nine paths one under another spilled it onto a second page.
|
|
102
106
|
*/
|
|
103
|
-
export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: number): BodyElement;
|
|
107
|
+
export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: number, behind?: boolean): BodyElement;
|
|
104
108
|
/**
|
|
105
109
|
* Derive the {@link SectionProperties} geometry from the source pages so a
|
|
106
110
|
* reconstructed PDF re-renders at its real page size and orientation rather than
|
|
@@ -113,6 +117,10 @@ export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: num
|
|
|
113
117
|
* `A4`. Returns `undefined` when there is no usable first-page box.
|
|
114
118
|
*/
|
|
115
119
|
export declare function sectionFromPdfPages(pages: ReadonlyArray<PdfPage>): SectionProperties | undefined;
|
|
120
|
+
export declare function withMeasuredMargins(section: SectionProperties | undefined, shown: ReadonlyArray<{
|
|
121
|
+
width: number;
|
|
122
|
+
height: number;
|
|
123
|
+
}>, pageRuns: ReadonlyArray<ReadonlyArray<TextRun>>): SectionProperties | undefined;
|
|
116
124
|
/**
|
|
117
125
|
* Assemble the final {@link FlowDoc} for a reconstruction: the body elements
|
|
118
126
|
* with their styles resolved against the empty style sheet, the lifted-image
|
|
@@ -27,11 +27,11 @@ function paragraphBlock(text, outlineLevel) {
|
|
|
27
27
|
* survives as its own run) and squashing whitespace. With no hrefs this
|
|
28
28
|
* collapses to a single run — the same shape {@link paragraphBlock} produces.
|
|
29
29
|
*/
|
|
30
|
-
function paragraphFromRuns(spans, outlineLevel) {
|
|
30
|
+
function paragraphFromRuns(spans, outlineLevel, placement) {
|
|
31
31
|
const merged = [];
|
|
32
32
|
for (const s of spans) {
|
|
33
33
|
const last = merged[merged.length - 1];
|
|
34
|
-
if (last && last.href === s.href && last.sizePt === s.sizePt && last.colorHex === s.colorHex && last.fontName === s.fontName && last.outline?.colorHex === s.outline?.colorHex && last.outline?.widthPt === s.outline?.widthPt && last.bold === s.bold && last.italic === s.italic) last.text += s.text;
|
|
34
|
+
if (last && last.href === s.href && last.sizePt === s.sizePt && last.colorHex === s.colorHex && last.fontName === s.fontName && last.outline?.colorHex === s.outline?.colorHex && last.outline?.widthPt === s.outline?.widthPt && last.bold === s.bold && last.italic === s.italic && sameMarkup(last.markup, s.markup)) last.text += s.text;
|
|
35
35
|
else merged.push({
|
|
36
36
|
text: s.text,
|
|
37
37
|
...s.href !== void 0 ? { href: s.href } : {},
|
|
@@ -40,7 +40,8 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
40
40
|
...s.fontName !== void 0 ? { fontName: s.fontName } : {},
|
|
41
41
|
...s.outline !== void 0 ? { outline: s.outline } : {},
|
|
42
42
|
...s.bold !== void 0 ? { bold: s.bold } : {},
|
|
43
|
-
...s.italic !== void 0 ? { italic: s.italic } : {}
|
|
43
|
+
...s.italic !== void 0 ? { italic: s.italic } : {},
|
|
44
|
+
...s.markup !== void 0 ? { markup: s.markup } : {}
|
|
44
45
|
});
|
|
45
46
|
}
|
|
46
47
|
const runs = merged.map((m) => ({
|
|
@@ -54,7 +55,10 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
54
55
|
return {
|
|
55
56
|
kind: "paragraph",
|
|
56
57
|
paragraph: {
|
|
57
|
-
properties:
|
|
58
|
+
properties: {
|
|
59
|
+
...outlineLevel !== void 0 ? { outlineLevel } : {},
|
|
60
|
+
...placement
|
|
61
|
+
},
|
|
58
62
|
runs: runs.filter((r) => r.text.length > 0).map((r) => ({
|
|
59
63
|
text: r.text,
|
|
60
64
|
properties: {
|
|
@@ -63,22 +67,31 @@ function paragraphFromRuns(spans, outlineLevel) {
|
|
|
63
67
|
...r.fontName !== void 0 ? { fontFamily: { ascii: r.fontName } } : {},
|
|
64
68
|
...r.outline !== void 0 ? { textOutline: r.outline } : {},
|
|
65
69
|
...r.bold ? { bold: true } : {},
|
|
66
|
-
...r.italic ? { italic: true } : {}
|
|
70
|
+
...r.italic ? { italic: true } : {},
|
|
71
|
+
...r.markup?.highlightHex !== void 0 ? { shadingColorHex: r.markup.highlightHex } : {},
|
|
72
|
+
...r.markup?.underline !== void 0 ? { underline: r.markup.underline } : {},
|
|
73
|
+
...r.markup?.underlineHex !== void 0 ? { underlineColorHex: r.markup.underlineHex } : {},
|
|
74
|
+
...r.markup?.strike === true ? { strike: true } : {}
|
|
67
75
|
},
|
|
68
76
|
...r.href ? { href: r.href } : {}
|
|
69
77
|
}))
|
|
70
78
|
}
|
|
71
79
|
};
|
|
72
80
|
}
|
|
81
|
+
/** Whether two runs are marked the same way, so they may join into one. */
|
|
82
|
+
function sameMarkup(a, b) {
|
|
83
|
+
return a?.highlightHex === b?.highlightHex && a?.underline === b?.underline && a?.underlineHex === b?.underlineHex && a?.strike === b?.strike;
|
|
84
|
+
}
|
|
73
85
|
/**
|
|
74
86
|
* Store a {@link PdfImage}'s bytes (content-addressed dedup) and build the image
|
|
75
87
|
* {@link BodyElement} that references them, sized in points from the placement
|
|
76
88
|
* CTM. `alt` becomes the block's alt text when given.
|
|
77
89
|
*/
|
|
78
|
-
function imageBlock(image, resources, alt, frame, zOrder) {
|
|
90
|
+
function imageBlock(image, resources, alt, frame, zOrder, behind = false) {
|
|
79
91
|
const resource = resources.put(image.bytes);
|
|
80
92
|
const float = frame !== void 0 ? {
|
|
81
93
|
wrap: "none",
|
|
94
|
+
...behind ? { behind: true } : {},
|
|
82
95
|
...zOrder !== void 0 ? { zOrder } : {},
|
|
83
96
|
posH: {
|
|
84
97
|
relativeFrom: "page",
|
|
@@ -96,6 +109,8 @@ function imageBlock(image, resources, alt, frame, zOrder) {
|
|
|
96
109
|
resource,
|
|
97
110
|
width: pt(image.widthPt),
|
|
98
111
|
height: pt(image.heightPt),
|
|
112
|
+
...image.rotationDeg !== void 0 ? { rotation60k: Math.round(-image.rotationDeg * 6e4) } : {},
|
|
113
|
+
...image.crop ? { crop: image.crop } : {},
|
|
99
114
|
paragraphProperties: {},
|
|
100
115
|
...alt ? { altText: alt } : {}
|
|
101
116
|
}
|
|
@@ -172,7 +187,7 @@ function dedupeLosses(losses) {
|
|
|
172
187
|
* a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
|
|
173
188
|
* its forty-nine paths one under another spilled it onto a second page.
|
|
174
189
|
*/
|
|
175
|
-
function shapeBlock(v, frame, zOrder) {
|
|
190
|
+
function shapeBlock(v, frame, zOrder, behind = false) {
|
|
176
191
|
const w = v.maxX - v.minX;
|
|
177
192
|
const h = v.maxY - v.minY;
|
|
178
193
|
const fx = (x) => x - v.minX;
|
|
@@ -222,6 +237,7 @@ function shapeBlock(v, frame, zOrder) {
|
|
|
222
237
|
} : void 0;
|
|
223
238
|
const float = frame !== void 0 ? {
|
|
224
239
|
wrap: "none",
|
|
240
|
+
...behind || v.darkens === true ? { behind: true } : {},
|
|
225
241
|
...zOrder !== void 0 ? { zOrder } : {},
|
|
226
242
|
posH: {
|
|
227
243
|
relativeFrom: "page",
|
|
@@ -287,6 +303,84 @@ function sectionFromPdfPages(pages) {
|
|
|
287
303
|
};
|
|
288
304
|
}
|
|
289
305
|
/**
|
|
306
|
+
* The margins the SOURCE used, measured off where its words actually sit.
|
|
307
|
+
*
|
|
308
|
+
* A PDF states none — text is placed anywhere on the MediaBox — so the reader
|
|
309
|
+
* used to leave them at zero rather than invent an inch. But the words
|
|
310
|
+
* themselves say where the margin was: the leftmost glyph on the page is the
|
|
311
|
+
* left margin, and reflowing inside it keeps the measure the author set instead
|
|
312
|
+
* of running the text from edge to edge.
|
|
313
|
+
*
|
|
314
|
+
* Measured on the MEDIAN page rather than the extreme one, so a single full-
|
|
315
|
+
* bleed rule or a page number in the corner does not collapse the margin for
|
|
316
|
+
* the whole document, and clamped so a strange page cannot leave no text area
|
|
317
|
+
* at all.
|
|
318
|
+
*
|
|
319
|
+
* @param section The section the page box gave, or `undefined`.
|
|
320
|
+
* @param shown Each page as it is shown, for its own width and height.
|
|
321
|
+
* @param pageRuns Each page's runs, already placed on the shown page.
|
|
322
|
+
* @returns The section with measured margins, or `section` when nothing is
|
|
323
|
+
* measurable.
|
|
324
|
+
*/
|
|
325
|
+
/** What the measure gives back, so the widest line still fits when re-set. */
|
|
326
|
+
var SLACK = .01;
|
|
327
|
+
/** How far a face's ascender stands above its baseline, as a fraction of the size. */
|
|
328
|
+
var ASCENDER = .8;
|
|
329
|
+
/** And its descender below — the two together are a little over one em. */
|
|
330
|
+
var DESCENDER = .22;
|
|
331
|
+
function withMeasuredMargins(section, shown, pageRuns) {
|
|
332
|
+
if (!section?.pageSize) return section;
|
|
333
|
+
const width = section.pageSize.width;
|
|
334
|
+
const height = section.pageSize.height;
|
|
335
|
+
const lefts = [];
|
|
336
|
+
const rights = [];
|
|
337
|
+
const tops = [];
|
|
338
|
+
const bottoms = [];
|
|
339
|
+
pageRuns.forEach((runs, i) => {
|
|
340
|
+
const page = shown[i];
|
|
341
|
+
if (!page || runs.length === 0) return;
|
|
342
|
+
let minX = Infinity;
|
|
343
|
+
let maxX = -Infinity;
|
|
344
|
+
let minY = Infinity;
|
|
345
|
+
let maxY = -Infinity;
|
|
346
|
+
let topSize = 0;
|
|
347
|
+
let bottomSize = 0;
|
|
348
|
+
for (const r of runs) {
|
|
349
|
+
if (!Number.isFinite(r.x) || !Number.isFinite(r.y)) continue;
|
|
350
|
+
minX = Math.min(minX, r.x);
|
|
351
|
+
maxX = Math.max(maxX, r.endX);
|
|
352
|
+
if (r.y < minY) {
|
|
353
|
+
minY = r.y;
|
|
354
|
+
bottomSize = r.fontSizePt;
|
|
355
|
+
}
|
|
356
|
+
if (r.y > maxY) {
|
|
357
|
+
maxY = r.y;
|
|
358
|
+
topSize = r.fontSizePt;
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
if (!Number.isFinite(minX) || !Number.isFinite(minY)) return;
|
|
362
|
+
lefts.push(minX);
|
|
363
|
+
rights.push(page.width - maxX);
|
|
364
|
+
tops.push(page.height - maxY - topSize * ASCENDER);
|
|
365
|
+
bottoms.push(minY - bottomSize * DESCENDER);
|
|
366
|
+
});
|
|
367
|
+
if (lefts.length === 0) return section;
|
|
368
|
+
const median = (xs) => {
|
|
369
|
+
const s = [...xs].sort((a, b) => a - b);
|
|
370
|
+
return s[Math.floor(s.length / 2)] ?? 0;
|
|
371
|
+
};
|
|
372
|
+
const clamp = (v, span) => pt(Math.max(0, Math.min(v, span / 3)));
|
|
373
|
+
return {
|
|
374
|
+
...section,
|
|
375
|
+
margins: {
|
|
376
|
+
left: clamp(median(lefts), width),
|
|
377
|
+
right: clamp(median(rights) - width * SLACK, width),
|
|
378
|
+
top: clamp(median(tops), height),
|
|
379
|
+
bottom: clamp(median(bottoms), height)
|
|
380
|
+
}
|
|
381
|
+
};
|
|
382
|
+
}
|
|
383
|
+
/**
|
|
290
384
|
* Assemble the final {@link FlowDoc} for a reconstruction: the body elements
|
|
291
385
|
* with their styles resolved against the empty style sheet, the lifted-image
|
|
292
386
|
* resource store, and the optional page {@link SectionProperties}. Shared by
|
|
@@ -305,4 +399,4 @@ function buildFlowDoc(body, resources = new ResourceStore(), section, embeddedFo
|
|
|
305
399
|
};
|
|
306
400
|
}
|
|
307
401
|
//#endregion
|
|
308
|
-
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock };
|
|
402
|
+
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock, withMeasuredMargins };
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
import { parseTtf } from "../core/font/ttf-parser.js";
|
|
2
2
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
3
|
-
import { embeddedFontName } from "./embedded-fonts.js";
|
|
4
3
|
import { parseToUnicodeCMap } from "./cmap.js";
|
|
5
4
|
import { textForGlyphName } from "./glyph-names.js";
|
|
5
|
+
import { standardFace, standardWidth } from "./standard-widths.js";
|
|
6
|
+
import { embeddedFontName } from "./embedded-fonts.js";
|
|
6
7
|
//#region src/pdf-reader/font.ts
|
|
7
8
|
/**
|
|
8
9
|
* Build a {@link ContentFont} (the interpreter's decode + advance hooks) from a
|
|
@@ -33,17 +34,52 @@ function buildContentFont(file, fontDict) {
|
|
|
33
34
|
const style = faceStyle(file, fontDict, isType0);
|
|
34
35
|
const name = embeddedFontName(file, fontDict);
|
|
35
36
|
const type3 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type3" ? type3Face(file, fontDict) : void 0;
|
|
36
|
-
const
|
|
37
|
+
const speechless = isType0 && unicode.size === 0;
|
|
38
|
+
const decodeOne = (code) => unicode.get(code) ?? (bytesPerCode === 1 ? latin1(code) : speechless ? "�" : "");
|
|
39
|
+
const simple = simpleWidths(file, fontDict, decodeOne);
|
|
37
40
|
const width = isType0 ? cidWidths(file, fontDict) : type3 ? (code) => simple(code) * type3.matrix[0] * 1e3 : simple;
|
|
38
41
|
return {
|
|
39
42
|
bytesPerCode,
|
|
40
43
|
...type3 ? { type3 } : {},
|
|
41
44
|
...name !== void 0 ? { name } : {},
|
|
42
|
-
decode: (codes) => codes.map((c) =>
|
|
45
|
+
decode: (codes) => codes.map((c) => readable(decodeOne(c))).join(""),
|
|
43
46
|
width,
|
|
44
47
|
...style
|
|
45
48
|
};
|
|
46
49
|
}
|
|
50
|
+
/**
|
|
51
|
+
* The character a code stands for when nothing in the font says.
|
|
52
|
+
*
|
|
53
|
+
* Latin-1 is the only guess there is, and it is a good one for a text font —
|
|
54
|
+
* but a C0 control is not a glyph. §9.4.3 shows GLYPHS, and a code falling back
|
|
55
|
+
* to one has landed there by accident: issue11549_reduced.pdf's one line came
|
|
56
|
+
* back as U+0007 through U+0011 and was drawn as six empty boxes over a page
|
|
57
|
+
* that shows nothing at all. A `/ToUnicode` that STATES a control is stating
|
|
58
|
+
* something and is left alone.
|
|
59
|
+
*/
|
|
60
|
+
function latin1(code) {
|
|
61
|
+
return code < 32 || code === 127 ? "�" : String.fromCharCode(code);
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* What a `/ToUnicode` gives, less what Unicode says is not a character.
|
|
65
|
+
*
|
|
66
|
+
* `U+FFFE` and `U+FFFF` are noncharacters and the `U+FDD0`–`U+FDEF` block with
|
|
67
|
+
* them; a lone surrogate is half of a pair that never came. A producer that
|
|
68
|
+
* maps its glyphs to any of these has said "no text here" in the only way the
|
|
69
|
+
* format lets it, and carrying them on writes bytes no reader can show —
|
|
70
|
+
* arial_unicode_ab_cidfont.pdf maps its four Arabic letters to `U+FFFF` and the
|
|
71
|
+
* page came back holding four of them. They become `U+FFFD`, which the
|
|
72
|
+
* reconstruction counts and reports rather than passing along.
|
|
73
|
+
*/
|
|
74
|
+
function readable(text) {
|
|
75
|
+
let out = "";
|
|
76
|
+
for (const ch of text) {
|
|
77
|
+
const cp = ch.codePointAt(0) ?? 0;
|
|
78
|
+
const noncharacter = (cp & 65534) === 65534 || cp >= 64976 && cp <= 65007 || cp === 0 || cp >= SURROGATE_FIRST && cp <= SURROGATE_LAST;
|
|
79
|
+
out += noncharacter ? "�" : ch;
|
|
80
|
+
}
|
|
81
|
+
return out;
|
|
82
|
+
}
|
|
47
83
|
/** The last Unicode code point in the BMP, and the surrogate block inside it. */
|
|
48
84
|
var BMP_END = 65535;
|
|
49
85
|
var SURROGATE_FIRST = 55296;
|
|
@@ -166,7 +202,11 @@ function namedGlyphs(file, fontDict) {
|
|
|
166
202
|
const text = textForGlyphName(name);
|
|
167
203
|
if (text !== void 0) out.set(code, text);
|
|
168
204
|
}
|
|
169
|
-
|
|
205
|
+
if (out.size > 0) return out;
|
|
206
|
+
if (asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) !== "Type3") return void 0;
|
|
207
|
+
const unreadable = /* @__PURE__ */ new Map();
|
|
208
|
+
for (const code of names.keys()) unreadable.set(code, "�");
|
|
209
|
+
return unreadable;
|
|
170
210
|
}
|
|
171
211
|
/** §9.6.6.1 `/Encoding` `/Differences` — code → glyph name, as the array runs. */
|
|
172
212
|
function differences(file, fontDict) {
|
|
@@ -186,6 +226,8 @@ function differences(file, fontDict) {
|
|
|
186
226
|
/** §9.8.2 `/Flags` — bit 7 is Italic, bit 19 ForceBold (bits numbered from 1). */
|
|
187
227
|
var FLAG_ITALIC = 64;
|
|
188
228
|
var FLAG_FORCE_BOLD = 1 << 18;
|
|
229
|
+
/** §9.8.1 `/FontWeight` — the lightest a face may state; below it is no weight. */
|
|
230
|
+
var LIGHTEST_WEIGHT = 100;
|
|
189
231
|
/** §9.8.1 `/FontWeight` — 400 is normal, 700 bold; 600 is where "bold" begins. */
|
|
190
232
|
var BOLD_WEIGHT = 600;
|
|
191
233
|
/**
|
|
@@ -207,23 +249,38 @@ var BOLD_WEIGHT = 600;
|
|
|
207
249
|
function faceStyle(file, fontDict, isType0) {
|
|
208
250
|
const owner = isType0 ? descendantFont(file, fontDict) : fontDict;
|
|
209
251
|
const descriptor = file.resolve(owner.get("FontDescriptor") ?? PDF_NULL);
|
|
252
|
+
const named = styleFromName(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
|
|
210
253
|
if (descriptor instanceof Map) {
|
|
211
254
|
const flags = asNumber(file.resolve(descriptor.get("Flags") ?? PDF_NULL), 0);
|
|
212
|
-
const
|
|
255
|
+
const weightVal = file.resolve(descriptor.get("FontWeight") ?? PDF_NULL);
|
|
213
256
|
const slant = asNumber(file.resolve(descriptor.get("ItalicAngle") ?? PDF_NULL), 0);
|
|
214
|
-
const bold =
|
|
215
|
-
const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0;
|
|
257
|
+
const bold = typeof weightVal === "number" && weightVal >= LIGHTEST_WEIGHT ? asNumber(weightVal, 0) >= BOLD_WEIGHT : (flags & FLAG_FORCE_BOLD) !== 0 || named.bold;
|
|
258
|
+
const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0 || named.italic;
|
|
216
259
|
return {
|
|
217
|
-
...bold ? { bold } : {},
|
|
218
|
-
...italic ? { italic } : {}
|
|
260
|
+
...bold ? { bold: true } : {},
|
|
261
|
+
...italic ? { italic: true } : {}
|
|
219
262
|
};
|
|
220
263
|
}
|
|
221
|
-
const name = asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)).replace(/^[A-Z]{6}\+/u, "");
|
|
222
|
-
const bold = /bold|black|heavy/iu.test(name);
|
|
223
|
-
const italic = /italic|oblique/iu.test(name);
|
|
224
264
|
return {
|
|
225
|
-
...bold ? { bold } : {},
|
|
226
|
-
...italic ? { italic } : {}
|
|
265
|
+
...named.bold ? { bold: true } : {},
|
|
266
|
+
...named.italic ? { italic: true } : {}
|
|
267
|
+
};
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* §9.6.2.2 — the style a font's NAME states, by the PostScript convention:
|
|
271
|
+
* `Family-Style`, or `Family,Style` as Word writes it.
|
|
272
|
+
*
|
|
273
|
+
* The separator is what makes this safe. A family whose name merely CONTAINS
|
|
274
|
+
* the word — "New Basrah Bold", "Damascus Bold", both real faces in
|
|
275
|
+
* ArabicCIDTrueType.pdf — is not a bold cut of anything, and reading it as one
|
|
276
|
+
* set two lines heavy that no reader sets heavy. `Times-Bold` is.
|
|
277
|
+
*/
|
|
278
|
+
function styleFromName(baseFont) {
|
|
279
|
+
const name = baseFont.replace(/^[A-Z]{6}\+/u, "");
|
|
280
|
+
const style = /[-,]([A-Za-z]+)$/u.exec(name)?.[1] ?? "";
|
|
281
|
+
return {
|
|
282
|
+
bold: /bold|black|heavy|semib|demi/iu.test(style),
|
|
283
|
+
italic: /italic|oblique/iu.test(style)
|
|
227
284
|
};
|
|
228
285
|
}
|
|
229
286
|
/** §9.7.4 — a `/Type0` font's one descendant CIDFont, which owns the descriptor. */
|
|
@@ -232,15 +289,19 @@ function descendantFont(file, fontDict) {
|
|
|
232
289
|
const first = Array.isArray(descFonts) ? file.resolve(descFonts[0] ?? PDF_NULL) : PDF_NULL;
|
|
233
290
|
return first instanceof Map ? first : /* @__PURE__ */ new Map();
|
|
234
291
|
}
|
|
235
|
-
function simpleWidths(file, fontDict) {
|
|
292
|
+
function simpleWidths(file, fontDict, decodeOne) {
|
|
236
293
|
const first = asNumber(file.resolve(fontDict.get("FirstChar") ?? PDF_NULL), 0);
|
|
237
294
|
const widthsVal = file.resolve(fontDict.get("Widths") ?? PDF_NULL);
|
|
238
295
|
const widths = Array.isArray(widthsVal) ? widthsVal : [];
|
|
239
296
|
const descriptor = file.resolve(fontDict.get("FontDescriptor") ?? PDF_NULL);
|
|
240
297
|
const missing = descriptor instanceof Map ? asNumber(file.resolve(descriptor.get("MissingWidth") ?? PDF_NULL), 0) : 0;
|
|
298
|
+
const face = standardFace(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
|
|
241
299
|
return (code) => {
|
|
242
300
|
const w = widths[code - first];
|
|
243
|
-
|
|
301
|
+
if (typeof w === "number") return w;
|
|
302
|
+
const built = face === void 0 ? void 0 : standardWidth(face, code, decodeOne(code));
|
|
303
|
+
if (built !== void 0) return built;
|
|
304
|
+
return missing > 0 ? missing : 500;
|
|
244
305
|
};
|
|
245
306
|
}
|
|
246
307
|
function cidWidths(file, fontDict) {
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { PdfValue } from '../pdf/objects.js';
|
|
2
|
+
import { PdfFile } from './document.js';
|
|
3
|
+
/** §7.10 — m numbers in, n numbers out. */
|
|
4
|
+
export type PdfFunction = (inputs: ReadonlyArray<number>) => Array<number>;
|
|
5
|
+
/**
|
|
6
|
+
* Read a `/Function` entry into something callable (§7.10).
|
|
7
|
+
*
|
|
8
|
+
* The entry may also be an ARRAY of functions, each giving one output, which is
|
|
9
|
+
* what a `/DeviceN` with a per-colorant transform states; that comes back as
|
|
10
|
+
* one function returning all of them in order.
|
|
11
|
+
*
|
|
12
|
+
* @param file The owning file.
|
|
13
|
+
* @param value The `/Function` (or `/TintTransform`) entry, unresolved.
|
|
14
|
+
* @returns The function, or `undefined` for one this cannot run.
|
|
15
|
+
*/
|
|
16
|
+
export declare function readFunction(file: PdfFile, value: PdfValue | undefined): PdfFunction | undefined;
|