reamkit 1.25.1 → 1.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +12 -5
  2. package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
  3. package/dist/esm/pdf-reader/annot-draw.js +374 -0
  4. package/dist/esm/pdf-reader/annots.d.ts +3 -1
  5. package/dist/esm/pdf-reader/annots.js +18 -4
  6. package/dist/esm/pdf-reader/ccitt.d.ts +18 -0
  7. package/dist/esm/pdf-reader/ccitt.js +70 -2
  8. package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
  9. package/dist/esm/pdf-reader/cie-color.js +112 -0
  10. package/dist/esm/pdf-reader/content.d.ts +47 -17
  11. package/dist/esm/pdf-reader/content.js +175 -14
  12. package/dist/esm/pdf-reader/display.d.ts +1 -1
  13. package/dist/esm/pdf-reader/display.js +21 -6
  14. package/dist/esm/pdf-reader/document.d.ts +6 -0
  15. package/dist/esm/pdf-reader/document.js +29 -1
  16. package/dist/esm/pdf-reader/embedded-fonts.d.ts +12 -3
  17. package/dist/esm/pdf-reader/embedded-fonts.js +22 -3
  18. package/dist/esm/pdf-reader/flow-build.d.ts +12 -4
  19. package/dist/esm/pdf-reader/flow-build.js +102 -8
  20. package/dist/esm/pdf-reader/font.js +97 -16
  21. package/dist/esm/pdf-reader/function.d.ts +16 -0
  22. package/dist/esm/pdf-reader/function.js +414 -0
  23. package/dist/esm/pdf-reader/glyph-names.d.ts +6 -0
  24. package/dist/esm/pdf-reader/glyph-names.js +408 -0
  25. package/dist/esm/pdf-reader/image-decode.d.ts +8 -4
  26. package/dist/esm/pdf-reader/image-decode.js +224 -12
  27. package/dist/esm/pdf-reader/images.d.ts +12 -0
  28. package/dist/esm/pdf-reader/images.js +109 -9
  29. package/dist/esm/pdf-reader/jbig2.d.ts +109 -0
  30. package/dist/esm/pdf-reader/jbig2.js +2606 -0
  31. package/dist/esm/pdf-reader/layout.d.ts +32 -0
  32. package/dist/esm/pdf-reader/layout.js +189 -40
  33. package/dist/esm/pdf-reader/lexer.d.ts +2 -0
  34. package/dist/esm/pdf-reader/lexer.js +4 -0
  35. package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
  36. package/dist/esm/pdf-reader/optional-content.js +93 -0
  37. package/dist/esm/pdf-reader/reader.d.ts +5 -2
  38. package/dist/esm/pdf-reader/reader.js +80 -7
  39. package/dist/esm/pdf-reader/shading.d.ts +71 -6
  40. package/dist/esm/pdf-reader/shading.js +185 -15
  41. package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
  42. package/dist/esm/pdf-reader/standard-metrics.js +18 -0
  43. package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
  44. package/dist/esm/pdf-reader/standard-widths.js +62 -0
  45. package/dist/esm/pdf-reader/tagged.js +204 -32
  46. package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
  47. package/dist/esm/pdf-reader/text-rules.js +112 -0
  48. package/dist/esm/pdf-reader/text.js +78 -4
  49. package/dist/esm/pdf-reader/vector.d.ts +5 -0
  50. package/dist/esm/pdf-reader/vector.js +39 -25
  51. package/dist/esm/word/docx-writer.js +213 -23
  52. package/package.json +1 -1
@@ -1,5 +1,6 @@
1
1
  import { PdfDict } from '../pdf/objects.js';
2
2
  import { PdfFile, PdfPage } from './document.js';
3
+ import { Loss } from '../core/ir/index.js';
3
4
  import { FontRegistry } from '../core/font/index.js';
4
5
  /**
5
6
  * Every embedded TrueType program the pages use, by the name a run will ask for
@@ -15,11 +16,19 @@ import { FontRegistry } from '../core/font/index.js';
15
16
  * (`/FontFile`) are different formats that the layout's parser does not take;
16
17
  * those keep their substitute.
17
18
  *
18
- * @param file The owning file.
19
- * @param pages The pages whose fonts are wanted.
19
+ * And only one whose OUTLINES this pipeline can carry: a program with no
20
+ * `glyf`/`loca` cannot be subset, and a face offered here is one the writer
21
+ * will be asked to embed. bug1186827.pdf ships an OpenType/CFF program under
22
+ * `/FontFile2`, which parses like any other and then killed the whole
23
+ * conversion — "Subsetting requires a TrueType font with glyf+loca tables" —
24
+ * rather than losing one face.
25
+ *
26
+ * @param file The owning file.
27
+ * @param pages The pages whose fonts are wanted.
28
+ * @param losses Appended to for a face the page embeds and this cannot carry.
20
29
  * @returns Name → a one-face registry holding that program.
21
30
  */
22
- export declare function collectEmbeddedFonts(file: PdfFile, pages: ReadonlyArray<PdfPage>): Map<string, FontRegistry>;
31
+ export declare function collectEmbeddedFonts(file: PdfFile, pages: ReadonlyArray<PdfPage>, losses?: Array<Loss>): Map<string, FontRegistry>;
23
32
  /**
24
33
  * The name a run set in `fontDict` will ask for: its `/BaseFont` without the
25
34
  * six-capital subset prefix (§9.6.4), lowercased.
@@ -1,5 +1,7 @@
1
+ import { parseTtf } from "../core/font/ttf-parser.js";
1
2
  import { FontRegistry } from "../core/font/font-registry.js";
2
3
  import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
4
+ import { FEATURES } from "../core/ir/features.js";
3
5
  //#region src/pdf-reader/embedded-fonts.ts
4
6
  var MAX_FORM_DEPTH = 8;
5
7
  /**
@@ -16,11 +18,19 @@ var MAX_FORM_DEPTH = 8;
16
18
  * (`/FontFile`) are different formats that the layout's parser does not take;
17
19
  * those keep their substitute.
18
20
  *
19
- * @param file The owning file.
20
- * @param pages The pages whose fonts are wanted.
21
+ * And only one whose OUTLINES this pipeline can carry: a program with no
22
+ * `glyf`/`loca` cannot be subset, and a face offered here is one the writer
23
+ * will be asked to embed. bug1186827.pdf ships an OpenType/CFF program under
24
+ * `/FontFile2`, which parses like any other and then killed the whole
25
+ * conversion — "Subsetting requires a TrueType font with glyf+loca tables" —
26
+ * rather than losing one face.
27
+ *
28
+ * @param file The owning file.
29
+ * @param pages The pages whose fonts are wanted.
30
+ * @param losses Appended to for a face the page embeds and this cannot carry.
21
31
  * @returns Name → a one-face registry holding that program.
22
32
  */
23
- function collectEmbeddedFonts(file, pages) {
33
+ function collectEmbeddedFonts(file, pages, losses) {
24
34
  const out = /* @__PURE__ */ new Map();
25
35
  const seen = /* @__PURE__ */ new Set();
26
36
  const visiting = /* @__PURE__ */ new Set();
@@ -32,6 +42,15 @@ function collectEmbeddedFonts(file, pages) {
32
42
  const program = fontProgram(file, fontDict);
33
43
  if (!program) return;
34
44
  try {
45
+ const parsed = parseTtf(program);
46
+ if (!parsed.tables.has("glyf") || !parsed.tables.has("loca")) {
47
+ losses?.push({
48
+ severity: "degraded",
49
+ feature: FEATURES.text,
50
+ detail: `embedded font ${name} has no TrueType outlines (a CFF program under /FontFile2); its text is re-set in a substitute face`
51
+ });
52
+ return;
53
+ }
35
54
  out.set(name, FontRegistry.fromBytes({ regular: program }));
36
55
  } catch {}
37
56
  };
@@ -1,10 +1,12 @@
1
- import { BodyElement, SectionProperties, TextOutline } from '../core/document-model/index.js';
1
+ import { BodyElement, ParagraphProperties, SectionProperties, TextOutline } from '../core/document-model/index.js';
2
2
  import { FlowDoc } from '../core/ir/flow.js';
3
3
  import { FontRegistry } from '../core/font/index.js';
4
4
  import { Loss, ResourceStore } from '../core/ir/index.js';
5
5
  import { PdfImage } from './images.js';
6
6
  import { PdfPage } from './document.js';
7
7
  import { PdfVector } from './vector.js';
8
+ import { TextMarkup } from './annot-draw.js';
9
+ import { TextRun } from './content.js';
8
10
  /**
9
11
  * A reconstruction's document plus the losses incurred reading it (e.g. an
10
12
  * undecodable image colour space) — surfaced through the reader's `LossReport`.
@@ -49,6 +51,8 @@ export interface TextSpan {
49
51
  readonly bold?: boolean;
50
52
  /** §9.8.1 — the face was a slanted one. */
51
53
  readonly italic?: boolean;
54
+ /** §12.5.6.10 — a text-markup annotation marks these words. */
55
+ readonly markup?: TextMarkup;
52
56
  }
53
57
  /**
54
58
  * Build a paragraph {@link BodyElement} from positioned {@link TextSpan}s,
@@ -56,13 +60,13 @@ export interface TextSpan {
56
60
  * survives as its own run) and squashing whitespace. With no hrefs this
57
61
  * collapses to a single run — the same shape {@link paragraphBlock} produces.
58
62
  */
59
- export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number): BodyElement;
63
+ export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number, placement?: Pick<ParagraphProperties, 'alignment' | 'spacingBefore'>): BodyElement;
60
64
  /**
61
65
  * Store a {@link PdfImage}'s bytes (content-addressed dedup) and build the image
62
66
  * {@link BodyElement} that references them, sized in points from the placement
63
67
  * CTM. `alt` becomes the block's alt text when given.
64
68
  */
65
- export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string, frame?: PageFrame, zOrder?: number): BodyElement;
69
+ export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string, frame?: PageFrame, zOrder?: number, behind?: boolean): BodyElement;
66
70
  /**
67
71
  * A line of text as an anchored box, standing where the page set it.
68
72
  *
@@ -100,7 +104,7 @@ export declare function dedupeLosses(losses: ReadonlyArray<Loss>): Array<Loss>;
100
104
  * a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
101
105
  * its forty-nine paths one under another spilled it onto a second page.
102
106
  */
103
- export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: number): BodyElement;
107
+ export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: number, behind?: boolean): BodyElement;
104
108
  /**
105
109
  * Derive the {@link SectionProperties} geometry from the source pages so a
106
110
  * reconstructed PDF re-renders at its real page size and orientation rather than
@@ -113,6 +117,10 @@ export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: num
113
117
  * `A4`. Returns `undefined` when there is no usable first-page box.
114
118
  */
115
119
  export declare function sectionFromPdfPages(pages: ReadonlyArray<PdfPage>): SectionProperties | undefined;
120
+ export declare function withMeasuredMargins(section: SectionProperties | undefined, shown: ReadonlyArray<{
121
+ width: number;
122
+ height: number;
123
+ }>, pageRuns: ReadonlyArray<ReadonlyArray<TextRun>>): SectionProperties | undefined;
116
124
  /**
117
125
  * Assemble the final {@link FlowDoc} for a reconstruction: the body elements
118
126
  * with their styles resolved against the empty style sheet, the lifted-image
@@ -27,11 +27,11 @@ function paragraphBlock(text, outlineLevel) {
27
27
  * survives as its own run) and squashing whitespace. With no hrefs this
28
28
  * collapses to a single run — the same shape {@link paragraphBlock} produces.
29
29
  */
30
- function paragraphFromRuns(spans, outlineLevel) {
30
+ function paragraphFromRuns(spans, outlineLevel, placement) {
31
31
  const merged = [];
32
32
  for (const s of spans) {
33
33
  const last = merged[merged.length - 1];
34
- if (last && last.href === s.href && last.sizePt === s.sizePt && last.colorHex === s.colorHex && last.fontName === s.fontName && last.outline?.colorHex === s.outline?.colorHex && last.outline?.widthPt === s.outline?.widthPt && last.bold === s.bold && last.italic === s.italic) last.text += s.text;
34
+ if (last && last.href === s.href && last.sizePt === s.sizePt && last.colorHex === s.colorHex && last.fontName === s.fontName && last.outline?.colorHex === s.outline?.colorHex && last.outline?.widthPt === s.outline?.widthPt && last.bold === s.bold && last.italic === s.italic && sameMarkup(last.markup, s.markup)) last.text += s.text;
35
35
  else merged.push({
36
36
  text: s.text,
37
37
  ...s.href !== void 0 ? { href: s.href } : {},
@@ -40,7 +40,8 @@ function paragraphFromRuns(spans, outlineLevel) {
40
40
  ...s.fontName !== void 0 ? { fontName: s.fontName } : {},
41
41
  ...s.outline !== void 0 ? { outline: s.outline } : {},
42
42
  ...s.bold !== void 0 ? { bold: s.bold } : {},
43
- ...s.italic !== void 0 ? { italic: s.italic } : {}
43
+ ...s.italic !== void 0 ? { italic: s.italic } : {},
44
+ ...s.markup !== void 0 ? { markup: s.markup } : {}
44
45
  });
45
46
  }
46
47
  const runs = merged.map((m) => ({
@@ -54,7 +55,10 @@ function paragraphFromRuns(spans, outlineLevel) {
54
55
  return {
55
56
  kind: "paragraph",
56
57
  paragraph: {
57
- properties: outlineLevel !== void 0 ? { outlineLevel } : {},
58
+ properties: {
59
+ ...outlineLevel !== void 0 ? { outlineLevel } : {},
60
+ ...placement
61
+ },
58
62
  runs: runs.filter((r) => r.text.length > 0).map((r) => ({
59
63
  text: r.text,
60
64
  properties: {
@@ -63,22 +67,31 @@ function paragraphFromRuns(spans, outlineLevel) {
63
67
  ...r.fontName !== void 0 ? { fontFamily: { ascii: r.fontName } } : {},
64
68
  ...r.outline !== void 0 ? { textOutline: r.outline } : {},
65
69
  ...r.bold ? { bold: true } : {},
66
- ...r.italic ? { italic: true } : {}
70
+ ...r.italic ? { italic: true } : {},
71
+ ...r.markup?.highlightHex !== void 0 ? { shadingColorHex: r.markup.highlightHex } : {},
72
+ ...r.markup?.underline !== void 0 ? { underline: r.markup.underline } : {},
73
+ ...r.markup?.underlineHex !== void 0 ? { underlineColorHex: r.markup.underlineHex } : {},
74
+ ...r.markup?.strike === true ? { strike: true } : {}
67
75
  },
68
76
  ...r.href ? { href: r.href } : {}
69
77
  }))
70
78
  }
71
79
  };
72
80
  }
81
+ /** Whether two runs are marked the same way, so they may join into one. */
82
+ function sameMarkup(a, b) {
83
+ return a?.highlightHex === b?.highlightHex && a?.underline === b?.underline && a?.underlineHex === b?.underlineHex && a?.strike === b?.strike;
84
+ }
73
85
  /**
74
86
  * Store a {@link PdfImage}'s bytes (content-addressed dedup) and build the image
75
87
  * {@link BodyElement} that references them, sized in points from the placement
76
88
  * CTM. `alt` becomes the block's alt text when given.
77
89
  */
78
- function imageBlock(image, resources, alt, frame, zOrder) {
90
+ function imageBlock(image, resources, alt, frame, zOrder, behind = false) {
79
91
  const resource = resources.put(image.bytes);
80
92
  const float = frame !== void 0 ? {
81
93
  wrap: "none",
94
+ ...behind ? { behind: true } : {},
82
95
  ...zOrder !== void 0 ? { zOrder } : {},
83
96
  posH: {
84
97
  relativeFrom: "page",
@@ -96,6 +109,8 @@ function imageBlock(image, resources, alt, frame, zOrder) {
96
109
  resource,
97
110
  width: pt(image.widthPt),
98
111
  height: pt(image.heightPt),
112
+ ...image.rotationDeg !== void 0 ? { rotation60k: Math.round(-image.rotationDeg * 6e4) } : {},
113
+ ...image.crop ? { crop: image.crop } : {},
99
114
  paragraphProperties: {},
100
115
  ...alt ? { altText: alt } : {}
101
116
  }
@@ -172,7 +187,7 @@ function dedupeLosses(losses) {
172
187
  * a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
173
188
  * its forty-nine paths one under another spilled it onto a second page.
174
189
  */
175
- function shapeBlock(v, frame, zOrder) {
190
+ function shapeBlock(v, frame, zOrder, behind = false) {
176
191
  const w = v.maxX - v.minX;
177
192
  const h = v.maxY - v.minY;
178
193
  const fx = (x) => x - v.minX;
@@ -222,6 +237,7 @@ function shapeBlock(v, frame, zOrder) {
222
237
  } : void 0;
223
238
  const float = frame !== void 0 ? {
224
239
  wrap: "none",
240
+ ...behind || v.darkens === true ? { behind: true } : {},
225
241
  ...zOrder !== void 0 ? { zOrder } : {},
226
242
  posH: {
227
243
  relativeFrom: "page",
@@ -287,6 +303,84 @@ function sectionFromPdfPages(pages) {
287
303
  };
288
304
  }
289
305
  /**
306
+ * The margins the SOURCE used, measured off where its words actually sit.
307
+ *
308
+ * A PDF states none — text is placed anywhere on the MediaBox — so the reader
309
+ * used to leave them at zero rather than invent an inch. But the words
310
+ * themselves say where the margin was: the leftmost glyph on the page is the
311
+ * left margin, and reflowing inside it keeps the measure the author set instead
312
+ * of running the text from edge to edge.
313
+ *
314
+ * Measured on the MEDIAN page rather than the extreme one, so a single full-
315
+ * bleed rule or a page number in the corner does not collapse the margin for
316
+ * the whole document, and clamped so a strange page cannot leave no text area
317
+ * at all.
318
+ *
319
+ * @param section The section the page box gave, or `undefined`.
320
+ * @param shown Each page as it is shown, for its own width and height.
321
+ * @param pageRuns Each page's runs, already placed on the shown page.
322
+ * @returns The section with measured margins, or `section` when nothing is
323
+ * measurable.
324
+ */
325
+ /** What the measure gives back, so the widest line still fits when re-set. */
326
+ var SLACK = .01;
327
+ /** How far a face's ascender stands above its baseline, as a fraction of the size. */
328
+ var ASCENDER = .8;
329
+ /** And its descender below — the two together are a little over one em. */
330
+ var DESCENDER = .22;
331
+ function withMeasuredMargins(section, shown, pageRuns) {
332
+ if (!section?.pageSize) return section;
333
+ const width = section.pageSize.width;
334
+ const height = section.pageSize.height;
335
+ const lefts = [];
336
+ const rights = [];
337
+ const tops = [];
338
+ const bottoms = [];
339
+ pageRuns.forEach((runs, i) => {
340
+ const page = shown[i];
341
+ if (!page || runs.length === 0) return;
342
+ let minX = Infinity;
343
+ let maxX = -Infinity;
344
+ let minY = Infinity;
345
+ let maxY = -Infinity;
346
+ let topSize = 0;
347
+ let bottomSize = 0;
348
+ for (const r of runs) {
349
+ if (!Number.isFinite(r.x) || !Number.isFinite(r.y)) continue;
350
+ minX = Math.min(minX, r.x);
351
+ maxX = Math.max(maxX, r.endX);
352
+ if (r.y < minY) {
353
+ minY = r.y;
354
+ bottomSize = r.fontSizePt;
355
+ }
356
+ if (r.y > maxY) {
357
+ maxY = r.y;
358
+ topSize = r.fontSizePt;
359
+ }
360
+ }
361
+ if (!Number.isFinite(minX) || !Number.isFinite(minY)) return;
362
+ lefts.push(minX);
363
+ rights.push(page.width - maxX);
364
+ tops.push(page.height - maxY - topSize * ASCENDER);
365
+ bottoms.push(minY - bottomSize * DESCENDER);
366
+ });
367
+ if (lefts.length === 0) return section;
368
+ const median = (xs) => {
369
+ const s = [...xs].sort((a, b) => a - b);
370
+ return s[Math.floor(s.length / 2)] ?? 0;
371
+ };
372
+ const clamp = (v, span) => pt(Math.max(0, Math.min(v, span / 3)));
373
+ return {
374
+ ...section,
375
+ margins: {
376
+ left: clamp(median(lefts), width),
377
+ right: clamp(median(rights) - width * SLACK, width),
378
+ top: clamp(median(tops), height),
379
+ bottom: clamp(median(bottoms), height)
380
+ }
381
+ };
382
+ }
383
+ /**
290
384
  * Assemble the final {@link FlowDoc} for a reconstruction: the body elements
291
385
  * with their styles resolved against the empty style sheet, the lifted-image
292
386
  * resource store, and the optional page {@link SectionProperties}. Shared by
@@ -305,4 +399,4 @@ function buildFlowDoc(body, resources = new ResourceStore(), section, embeddedFo
305
399
  };
306
400
  }
307
401
  //#endregion
308
- export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock };
402
+ export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock, withMeasuredMargins };
@@ -1,7 +1,9 @@
1
1
  import { parseTtf } from "../core/font/ttf-parser.js";
2
2
  import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
3
- import { embeddedFontName } from "./embedded-fonts.js";
4
3
  import { parseToUnicodeCMap } from "./cmap.js";
4
+ import { textForGlyphName } from "./glyph-names.js";
5
+ import { standardFace, standardWidth } from "./standard-widths.js";
6
+ import { embeddedFontName } from "./embedded-fonts.js";
5
7
  //#region src/pdf-reader/font.ts
6
8
  /**
7
9
  * Build a {@link ContentFont} (the interpreter's decode + advance hooks) from a
@@ -25,22 +27,59 @@ function buildContentFont(file, fontDict) {
25
27
  toUnicode = parsed.map;
26
28
  if (isType0) codeBytes = parsed.codeBytes;
27
29
  }
28
- const unicode = (isType0 && toUnicode.size === 0 ? embeddedCmap(file, fontDict) : void 0) ?? toUnicode;
30
+ const fromProgram = isType0 && toUnicode.size === 0 ? embeddedCmap(file, fontDict) : void 0;
31
+ const fromNames = !isType0 ? namedGlyphs(file, fontDict) : void 0;
32
+ const unicode = fromProgram ?? (toUnicode.size > 0 ? toUnicode : fromNames ?? toUnicode);
29
33
  const bytesPerCode = codeBytes;
30
34
  const style = faceStyle(file, fontDict, isType0);
31
35
  const name = embeddedFontName(file, fontDict);
32
36
  const type3 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type3" ? type3Face(file, fontDict) : void 0;
33
- const simple = simpleWidths(file, fontDict);
37
+ const speechless = isType0 && unicode.size === 0;
38
+ const decodeOne = (code) => unicode.get(code) ?? (bytesPerCode === 1 ? latin1(code) : speechless ? "�" : "");
39
+ const simple = simpleWidths(file, fontDict, decodeOne);
34
40
  const width = isType0 ? cidWidths(file, fontDict) : type3 ? (code) => simple(code) * type3.matrix[0] * 1e3 : simple;
35
41
  return {
36
42
  bytesPerCode,
37
43
  ...type3 ? { type3 } : {},
38
44
  ...name !== void 0 ? { name } : {},
39
- decode: (codes) => codes.map((c) => unicode.get(c) ?? (bytesPerCode === 1 ? String.fromCharCode(c) : "")).join(""),
45
+ decode: (codes) => codes.map((c) => readable(decodeOne(c))).join(""),
40
46
  width,
41
47
  ...style
42
48
  };
43
49
  }
50
+ /**
51
+ * The character a code stands for when nothing in the font says.
52
+ *
53
+ * Latin-1 is the only guess there is, and it is a good one for a text font —
54
+ * but a C0 control is not a glyph. §9.4.3 shows GLYPHS, and a code falling back
55
+ * to one has landed there by accident: issue11549_reduced.pdf's one line came
56
+ * back as U+0007 through U+0011 and was drawn as six empty boxes over a page
57
+ * that shows nothing at all. A `/ToUnicode` that STATES a control is stating
58
+ * something and is left alone.
59
+ */
60
+ function latin1(code) {
61
+ return code < 32 || code === 127 ? "�" : String.fromCharCode(code);
62
+ }
63
+ /**
64
+ * What a `/ToUnicode` gives, less what Unicode says is not a character.
65
+ *
66
+ * `U+FFFE` and `U+FFFF` are noncharacters and the `U+FDD0`–`U+FDEF` block with
67
+ * them; a lone surrogate is half of a pair that never came. A producer that
68
+ * maps its glyphs to any of these has said "no text here" in the only way the
69
+ * format lets it, and carrying them on writes bytes no reader can show —
70
+ * arial_unicode_ab_cidfont.pdf maps its four Arabic letters to `U+FFFF` and the
71
+ * page came back holding four of them. They become `U+FFFD`, which the
72
+ * reconstruction counts and reports rather than passing along.
73
+ */
74
+ function readable(text) {
75
+ let out = "";
76
+ for (const ch of text) {
77
+ const cp = ch.codePointAt(0) ?? 0;
78
+ const noncharacter = (cp & 65534) === 65534 || cp >= 64976 && cp <= 65007 || cp === 0 || cp >= SURROGATE_FIRST && cp <= SURROGATE_LAST;
79
+ out += noncharacter ? "�" : ch;
80
+ }
81
+ return out;
82
+ }
44
83
  /** The last Unicode code point in the BMP, and the surrogate block inside it. */
45
84
  var BMP_END = 65535;
46
85
  var SURROGATE_FIRST = 55296;
@@ -148,6 +187,27 @@ function fontMatrix(file, fontDict) {
148
187
  n[5]
149
188
  ];
150
189
  }
190
+ /**
191
+ * §9.6.6.1 — code → text for a simple font, out of the glyph names its
192
+ * `/Encoding /Differences` gives.
193
+ *
194
+ * Returns `undefined` where the font names nothing, so the caller keeps its
195
+ * Latin-1 reading rather than replacing it with an empty map.
196
+ */
197
+ function namedGlyphs(file, fontDict) {
198
+ const names = differences(file, fontDict);
199
+ if (names.size === 0) return void 0;
200
+ const out = /* @__PURE__ */ new Map();
201
+ for (const [code, name] of names) {
202
+ const text = textForGlyphName(name);
203
+ if (text !== void 0) out.set(code, text);
204
+ }
205
+ if (out.size > 0) return out;
206
+ if (asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) !== "Type3") return void 0;
207
+ const unreadable = /* @__PURE__ */ new Map();
208
+ for (const code of names.keys()) unreadable.set(code, "�");
209
+ return unreadable;
210
+ }
151
211
  /** §9.6.6.1 `/Encoding` `/Differences` — code → glyph name, as the array runs. */
152
212
  function differences(file, fontDict) {
153
213
  const out = /* @__PURE__ */ new Map();
@@ -166,6 +226,8 @@ function differences(file, fontDict) {
166
226
  /** §9.8.2 `/Flags` — bit 7 is Italic, bit 19 ForceBold (bits numbered from 1). */
167
227
  var FLAG_ITALIC = 64;
168
228
  var FLAG_FORCE_BOLD = 1 << 18;
229
+ /** §9.8.1 `/FontWeight` — the lightest a face may state; below it is no weight. */
230
+ var LIGHTEST_WEIGHT = 100;
169
231
  /** §9.8.1 `/FontWeight` — 400 is normal, 700 bold; 600 is where "bold" begins. */
170
232
  var BOLD_WEIGHT = 600;
171
233
  /**
@@ -187,23 +249,38 @@ var BOLD_WEIGHT = 600;
187
249
  function faceStyle(file, fontDict, isType0) {
188
250
  const owner = isType0 ? descendantFont(file, fontDict) : fontDict;
189
251
  const descriptor = file.resolve(owner.get("FontDescriptor") ?? PDF_NULL);
252
+ const named = styleFromName(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
190
253
  if (descriptor instanceof Map) {
191
254
  const flags = asNumber(file.resolve(descriptor.get("Flags") ?? PDF_NULL), 0);
192
- const weight = asNumber(file.resolve(descriptor.get("FontWeight") ?? PDF_NULL), 0);
255
+ const weightVal = file.resolve(descriptor.get("FontWeight") ?? PDF_NULL);
193
256
  const slant = asNumber(file.resolve(descriptor.get("ItalicAngle") ?? PDF_NULL), 0);
194
- const bold = weight >= BOLD_WEIGHT || (flags & FLAG_FORCE_BOLD) !== 0;
195
- const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0;
257
+ const bold = typeof weightVal === "number" && weightVal >= LIGHTEST_WEIGHT ? asNumber(weightVal, 0) >= BOLD_WEIGHT : (flags & FLAG_FORCE_BOLD) !== 0 || named.bold;
258
+ const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0 || named.italic;
196
259
  return {
197
- ...bold ? { bold } : {},
198
- ...italic ? { italic } : {}
260
+ ...bold ? { bold: true } : {},
261
+ ...italic ? { italic: true } : {}
199
262
  };
200
263
  }
201
- const name = asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)).replace(/^[A-Z]{6}\+/u, "");
202
- const bold = /bold|black|heavy/iu.test(name);
203
- const italic = /italic|oblique/iu.test(name);
204
264
  return {
205
- ...bold ? { bold } : {},
206
- ...italic ? { italic } : {}
265
+ ...named.bold ? { bold: true } : {},
266
+ ...named.italic ? { italic: true } : {}
267
+ };
268
+ }
269
+ /**
270
+ * §9.6.2.2 — the style a font's NAME states, by the PostScript convention:
271
+ * `Family-Style`, or `Family,Style` as Word writes it.
272
+ *
273
+ * The separator is what makes this safe. A family whose name merely CONTAINS
274
+ * the word — "New Basrah Bold", "Damascus Bold", both real faces in
275
+ * ArabicCIDTrueType.pdf — is not a bold cut of anything, and reading it as one
276
+ * set two lines heavy that no reader sets heavy. `Times-Bold` is.
277
+ */
278
+ function styleFromName(baseFont) {
279
+ const name = baseFont.replace(/^[A-Z]{6}\+/u, "");
280
+ const style = /[-,]([A-Za-z]+)$/u.exec(name)?.[1] ?? "";
281
+ return {
282
+ bold: /bold|black|heavy|semib|demi/iu.test(style),
283
+ italic: /italic|oblique/iu.test(style)
207
284
  };
208
285
  }
209
286
  /** §9.7.4 — a `/Type0` font's one descendant CIDFont, which owns the descriptor. */
@@ -212,15 +289,19 @@ function descendantFont(file, fontDict) {
212
289
  const first = Array.isArray(descFonts) ? file.resolve(descFonts[0] ?? PDF_NULL) : PDF_NULL;
213
290
  return first instanceof Map ? first : /* @__PURE__ */ new Map();
214
291
  }
215
- function simpleWidths(file, fontDict) {
292
+ function simpleWidths(file, fontDict, decodeOne) {
216
293
  const first = asNumber(file.resolve(fontDict.get("FirstChar") ?? PDF_NULL), 0);
217
294
  const widthsVal = file.resolve(fontDict.get("Widths") ?? PDF_NULL);
218
295
  const widths = Array.isArray(widthsVal) ? widthsVal : [];
219
296
  const descriptor = file.resolve(fontDict.get("FontDescriptor") ?? PDF_NULL);
220
297
  const missing = descriptor instanceof Map ? asNumber(file.resolve(descriptor.get("MissingWidth") ?? PDF_NULL), 0) : 0;
298
+ const face = standardFace(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
221
299
  return (code) => {
222
300
  const w = widths[code - first];
223
- return typeof w === "number" ? w : missing > 0 ? missing : 500;
301
+ if (typeof w === "number") return w;
302
+ const built = face === void 0 ? void 0 : standardWidth(face, code, decodeOne(code));
303
+ if (built !== void 0) return built;
304
+ return missing > 0 ? missing : 500;
224
305
  };
225
306
  }
226
307
  function cidWidths(file, fontDict) {
@@ -0,0 +1,16 @@
1
+ import { PdfValue } from '../pdf/objects.js';
2
+ import { PdfFile } from './document.js';
3
+ /** §7.10 — m numbers in, n numbers out. */
4
+ export type PdfFunction = (inputs: ReadonlyArray<number>) => Array<number>;
5
+ /**
6
+ * Read a `/Function` entry into something callable (§7.10).
7
+ *
8
+ * The entry may also be an ARRAY of functions, each giving one output, which is
9
+ * what a `/DeviceN` with a per-colorant transform states; that comes back as
10
+ * one function returning all of them in order.
11
+ *
12
+ * @param file The owning file.
13
+ * @param value The `/Function` (or `/TintTransform`) entry, unresolved.
14
+ * @returns The function, or `undefined` for one this cannot run.
15
+ */
16
+ export declare function readFunction(file: PdfFile, value: PdfValue | undefined): PdfFunction | undefined;