reamkit 1.24.0 → 1.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/README.md +22 -11
  2. package/dist/esm/core/converter/facade.d.ts +3 -3
  3. package/dist/esm/core/converter/facade.js +12 -0
  4. package/dist/esm/core/converter/ream.d.ts +48 -4
  5. package/dist/esm/core/converter/ream.js +25 -4
  6. package/dist/esm/core/document-model/index.d.ts +1 -1
  7. package/dist/esm/core/document-model/types.d.ts +17 -0
  8. package/dist/esm/core/outline.d.ts +17 -0
  9. package/dist/esm/core/outline.js +30 -0
  10. package/dist/esm/core/style-cascade/resolver.js +1 -0
  11. package/dist/esm/core/style-cascade/types.d.ts +3 -1
  12. package/dist/esm/excel/sheet-to-flow.d.ts +10 -0
  13. package/dist/esm/excel/sheet-to-flow.js +14 -1
  14. package/dist/esm/html/html-writer.js +3 -2
  15. package/dist/esm/index.d.ts +3 -0
  16. package/dist/esm/index.js +2 -1
  17. package/dist/esm/layout/page-doc.js +1 -1
  18. package/dist/esm/layout/styled-layout.js +41 -15
  19. package/dist/esm/markdown/markdown-writer.d.ts +41 -0
  20. package/dist/esm/markdown/markdown-writer.js +733 -0
  21. package/dist/esm/pdf/styled-page-emitter.js +20 -1
  22. package/dist/esm/pdf-reader/annots.d.ts +24 -0
  23. package/dist/esm/pdf-reader/annots.js +126 -0
  24. package/dist/esm/pdf-reader/content.d.ts +131 -5
  25. package/dist/esm/pdf-reader/content.js +169 -12
  26. package/dist/esm/pdf-reader/display.d.ts +56 -0
  27. package/dist/esm/pdf-reader/display.js +162 -0
  28. package/dist/esm/pdf-reader/document.d.ts +36 -1
  29. package/dist/esm/pdf-reader/document.js +92 -25
  30. package/dist/esm/pdf-reader/embedded-fonts.d.ts +31 -0
  31. package/dist/esm/pdf-reader/embedded-fonts.js +94 -0
  32. package/dist/esm/pdf-reader/flow-build.d.ts +61 -6
  33. package/dist/esm/pdf-reader/flow-build.js +128 -22
  34. package/dist/esm/pdf-reader/font.js +185 -4
  35. package/dist/esm/pdf-reader/image-decode.js +55 -4
  36. package/dist/esm/pdf-reader/images.d.ts +6 -0
  37. package/dist/esm/pdf-reader/images.js +25 -5
  38. package/dist/esm/pdf-reader/jpeg.d.ts +18 -0
  39. package/dist/esm/pdf-reader/jpeg.js +419 -0
  40. package/dist/esm/pdf-reader/layout.d.ts +1 -1
  41. package/dist/esm/pdf-reader/layout.js +221 -32
  42. package/dist/esm/pdf-reader/pattern-tint.d.ts +17 -0
  43. package/dist/esm/pdf-reader/pattern-tint.js +181 -0
  44. package/dist/esm/pdf-reader/reader.d.ts +9 -1
  45. package/dist/esm/pdf-reader/reader.js +22 -6
  46. package/dist/esm/pdf-reader/shading.d.ts +14 -0
  47. package/dist/esm/pdf-reader/shading.js +27 -1
  48. package/dist/esm/pdf-reader/tagged.js +156 -17
  49. package/dist/esm/pdf-reader/text.d.ts +13 -1
  50. package/dist/esm/pdf-reader/text.js +70 -3
  51. package/dist/esm/pdf-reader/vector.d.ts +25 -1
  52. package/dist/esm/pdf-reader/vector.js +168 -12
  53. package/dist/esm/pptx/slide-parser.js +5 -0
  54. package/dist/esm/word/docx-writer.js +11 -1
  55. package/dist/esm/word/drawing-parser.js +7 -1
  56. package/package.json +1 -1
@@ -2,6 +2,7 @@ import { pt } from "../core/ir/units.js";
2
2
  import { ResourceStore } from "../core/ir/resources.js";
3
3
  import { EMPTY_STYLE_SHEET, resolveBodyStyles } from "../core/style-cascade/resolver.js";
4
4
  import "../core/style-cascade/index.js";
5
+ import { displayOf } from "./display.js";
5
6
  //#region src/pdf-reader/flow-build.ts
6
7
  /**
7
8
  * Build a paragraph {@link BodyElement} from a single plain-text string,
@@ -30,16 +31,21 @@ function paragraphFromRuns(spans, outlineLevel) {
30
31
  const merged = [];
31
32
  for (const s of spans) {
32
33
  const last = merged[merged.length - 1];
33
- if (last && last.href === s.href) last.text += s.text;
34
- else if (s.href !== void 0) merged.push({
34
+ if (last && last.href === s.href && last.sizePt === s.sizePt && last.colorHex === s.colorHex && last.fontName === s.fontName && last.outline?.colorHex === s.outline?.colorHex && last.outline?.widthPt === s.outline?.widthPt && last.bold === s.bold && last.italic === s.italic) last.text += s.text;
35
+ else merged.push({
35
36
  text: s.text,
36
- href: s.href
37
+ ...s.href !== void 0 ? { href: s.href } : {},
38
+ ...s.sizePt !== void 0 ? { sizePt: s.sizePt } : {},
39
+ ...s.colorHex !== void 0 ? { colorHex: s.colorHex } : {},
40
+ ...s.fontName !== void 0 ? { fontName: s.fontName } : {},
41
+ ...s.outline !== void 0 ? { outline: s.outline } : {},
42
+ ...s.bold !== void 0 ? { bold: s.bold } : {},
43
+ ...s.italic !== void 0 ? { italic: s.italic } : {}
37
44
  });
38
- else merged.push({ text: s.text });
39
45
  }
40
46
  const runs = merged.map((m) => ({
41
- text: m.text.replace(/\s+/g, " "),
42
- href: m.href
47
+ ...m,
48
+ text: m.text.replace(/\s+/g, " ")
43
49
  })).filter((m) => m.text.length > 0);
44
50
  if (runs.length > 0) {
45
51
  runs[0].text = runs[0].text.replace(/^ /, "");
@@ -51,7 +57,14 @@ function paragraphFromRuns(spans, outlineLevel) {
51
57
  properties: outlineLevel !== void 0 ? { outlineLevel } : {},
52
58
  runs: runs.filter((r) => r.text.length > 0).map((r) => ({
53
59
  text: r.text,
54
- properties: {},
60
+ properties: {
61
+ ...r.sizePt !== void 0 ? { fontSizePt: pt(r.sizePt) } : {},
62
+ ...r.colorHex !== void 0 ? { colorHex: r.colorHex } : {},
63
+ ...r.fontName !== void 0 ? { fontFamily: { ascii: r.fontName } } : {},
64
+ ...r.outline !== void 0 ? { textOutline: r.outline } : {},
65
+ ...r.bold ? { bold: true } : {},
66
+ ...r.italic ? { italic: true } : {}
67
+ },
55
68
  ...r.href ? { href: r.href } : {}
56
69
  }))
57
70
  }
@@ -62,11 +75,25 @@ function paragraphFromRuns(spans, outlineLevel) {
62
75
  * {@link BodyElement} that references them, sized in points from the placement
63
76
  * CTM. `alt` becomes the block's alt text when given.
64
77
  */
65
- function imageBlock(image, resources, alt) {
78
+ function imageBlock(image, resources, alt, frame, zOrder) {
79
+ const resource = resources.put(image.bytes);
80
+ const float = frame !== void 0 ? {
81
+ wrap: "none",
82
+ ...zOrder !== void 0 ? { zOrder } : {},
83
+ posH: {
84
+ relativeFrom: "page",
85
+ offsetPt: pt(image.x - frame.left)
86
+ },
87
+ posV: {
88
+ relativeFrom: "page",
89
+ offsetPt: pt(Math.max(0, frame.top - image.y - image.heightPt))
90
+ }
91
+ } : void 0;
66
92
  return {
67
93
  kind: "image",
68
94
  image: {
69
- resource: resources.put(image.bytes),
95
+ ...float ? { float } : {},
96
+ resource,
70
97
  width: pt(image.widthPt),
71
98
  height: pt(image.heightPt),
72
99
  paragraphProperties: {},
@@ -74,6 +101,60 @@ function imageBlock(image, resources, alt) {
74
101
  }
75
102
  };
76
103
  }
104
+ /**
105
+ * A line of text as an anchored box, standing where the page set it.
106
+ *
107
+ * The flowed reconstruction reads a document OUT of a page: paragraphs in
108
+ * reading order, re-flowable, free to land wherever the next medium puts them.
109
+ * A form is not that document. 160F-2019.pdf is a grid of ruled boxes with a
110
+ * label in each, and a label means nothing an inch from the box it labels — the
111
+ * artwork is placed absolutely, so text that flows beside it lines up with none
112
+ * of it.
113
+ *
114
+ * @param spans The line's runs.
115
+ * @param box Its page-space rectangle (y-up, as PDF measures).
116
+ * @param frame The page's own corner, to measure the box off.
117
+ * @param zOrder Its place in the page's painting order.
118
+ * @param rotation60k §20.1.7.6 — how far the box turns about its own centre,
119
+ * for a baseline the page did not set flat.
120
+ * @returns A shape carrying the text, anchored where the glyphs were.
121
+ */
122
+ function positionedText(spans, box, frame, zOrder, rotation60k) {
123
+ const paragraph = paragraphFromRuns(spans);
124
+ return {
125
+ kind: "shape",
126
+ shape: {
127
+ float: {
128
+ wrap: "none",
129
+ zOrder,
130
+ posH: {
131
+ relativeFrom: "page",
132
+ offsetPt: pt(box.x - frame.left)
133
+ },
134
+ posV: {
135
+ relativeFrom: "page",
136
+ offsetPt: pt(Math.max(0, frame.top - box.y - box.height))
137
+ }
138
+ },
139
+ width: pt(Math.max(1, box.width)),
140
+ height: pt(Math.max(1, box.height)),
141
+ ...rotation60k !== void 0 ? { transform: { rotation60k } } : {},
142
+ geometry: {
143
+ kind: "preset",
144
+ preset: "rect"
145
+ },
146
+ fill: { kind: "none" },
147
+ text: {
148
+ content: [paragraph],
149
+ insetLeft: pt(0),
150
+ insetTop: pt(0),
151
+ insetRight: pt(0),
152
+ insetBottom: pt(0)
153
+ },
154
+ paragraphProperties: {}
155
+ }
156
+ };
157
+ }
77
158
  /** Collapse losses sharing a `detail` message (the same colour space dropped on many pages). */
78
159
  function dedupeLosses(losses) {
79
160
  const byDetail = /* @__PURE__ */ new Map();
@@ -84,10 +165,14 @@ function dedupeLosses(losses) {
84
165
  * Turn a lifted {@link PdfVector} path (filled EP10 / stroked EP11) into a
85
166
  * custom-geometry shape {@link BodyElement}. Page-space points (y-up) become
86
167
  * path-space (bbox-relative, y-down); the shape is sized from the bounding box
87
- * (plus the stroke thickness) and placed in flow order by the caller. A fill
88
- * becomes a solid fill, a stroke becomes the outline.
168
+ * (plus the stroke thickness). A fill becomes a solid fill, a stroke the outline.
169
+ *
170
+ * Given the page's frame the shape is ANCHORED where the page drew it, behind
171
+ * the text, rather than taking a place of its own in the flow. A drawing is not
172
+ * a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
173
+ * its forty-nine paths one under another spilled it onto a second page.
89
174
  */
90
- function shapeBlock(v) {
175
+ function shapeBlock(v, frame, zOrder) {
91
176
  const w = v.maxX - v.minX;
92
177
  const h = v.maxY - v.minY;
93
178
  const fx = (x) => x - v.minX;
@@ -116,22 +201,41 @@ function shapeBlock(v) {
116
201
  case "close": return { cmd: "close" };
117
202
  }
118
203
  });
119
- const thick = v.strokeHex !== void 0 ? Math.max(v.lineWidth ?? .75, .5) : 0;
204
+ const HAIRLINE_PT = .1;
205
+ const stated = v.lineWidth ?? .75;
206
+ const pen = v.strokeHex !== void 0 ? stated > 0 ? stated : HAIRLINE_PT : 0;
207
+ const thick = v.strokeHex !== void 0 ? Math.max(pen, .5) : 0;
208
+ const alpha = v.alpha !== void 0 ? { alpha: v.alpha } : {};
120
209
  const fill = v.gradient !== void 0 ? {
121
210
  kind: "gradient",
122
- gradient: v.gradient
211
+ gradient: v.gradient,
212
+ ...alpha
123
213
  } : v.fillHex !== void 0 ? {
124
214
  kind: "solid",
125
- colorHex: v.fillHex
215
+ colorHex: v.fillHex,
216
+ ...alpha
126
217
  } : { kind: "none" };
127
218
  const line = v.strokeHex !== void 0 ? {
128
- width: pt(thick),
219
+ width: pt(pen),
129
220
  colorHex: v.strokeHex,
130
221
  fill: "solid"
131
222
  } : void 0;
223
+ const float = frame !== void 0 ? {
224
+ wrap: "none",
225
+ ...zOrder !== void 0 ? { zOrder } : {},
226
+ posH: {
227
+ relativeFrom: "page",
228
+ offsetPt: pt(v.minX - frame.left)
229
+ },
230
+ posV: {
231
+ relativeFrom: "page",
232
+ offsetPt: pt(Math.max(0, frame.top - v.maxY))
233
+ }
234
+ } : void 0;
132
235
  return {
133
236
  kind: "shape",
134
237
  shape: {
238
+ ...float ? { float } : {},
135
239
  width: pt(Math.max(w, thick)),
136
240
  height: pt(Math.max(h, thick)),
137
241
  geometry: {
@@ -160,10 +264,11 @@ function shapeBlock(v) {
160
264
  * `A4`. Returns `undefined` when there is no usable first-page box.
161
265
  */
162
266
  function sectionFromPdfPages(pages) {
163
- const box = pages[0]?.mediaBox;
164
- if (!box) return void 0;
165
- const width = Math.abs(box[2] - box[0]);
166
- const height = Math.abs(box[3] - box[1]);
267
+ const first = pages[0];
268
+ if (!first) return void 0;
269
+ const shown = displayOf(first);
270
+ const width = shown.width;
271
+ const height = shown.height;
167
272
  if (!(width > 0 && height > 0)) return void 0;
168
273
  return {
169
274
  pageSize: {
@@ -188,15 +293,16 @@ function sectionFromPdfPages(pages) {
188
293
  * both reconstruction paths (the tagged fast-path EP3 and the heuristic layout
189
294
  * path EP4).
190
295
  */
191
- function buildFlowDoc(body, resources = new ResourceStore(), section) {
296
+ function buildFlowDoc(body, resources = new ResourceStore(), section, embeddedFonts) {
192
297
  return {
193
298
  kind: "flow",
194
299
  body: resolveBodyStyles([...body], EMPTY_STYLE_SHEET),
195
300
  sections: [],
196
301
  ...section ? { section } : {},
302
+ ...embeddedFonts && embeddedFonts.size > 0 ? { embeddedFonts } : {},
197
303
  styles: EMPTY_STYLE_SHEET,
198
304
  resources
199
305
  };
200
306
  }
201
307
  //#endregion
202
- export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock };
308
+ export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock };
@@ -1,4 +1,6 @@
1
+ import { parseTtf } from "../core/font/ttf-parser.js";
1
2
  import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
3
+ import { embeddedFontName } from "./embedded-fonts.js";
2
4
  import { parseToUnicodeCMap } from "./cmap.js";
3
5
  //#region src/pdf-reader/font.ts
4
6
  /**
@@ -21,16 +23,195 @@ function buildContentFont(file, fontDict) {
21
23
  if (tu instanceof PdfStream) {
22
24
  const parsed = parseToUnicodeCMap(file.streamData(tu));
23
25
  toUnicode = parsed.map;
24
- codeBytes = parsed.codeBytes;
26
+ if (isType0) codeBytes = parsed.codeBytes;
25
27
  }
26
- const width = isType0 ? cidWidths(file, fontDict) : simpleWidths(file, fontDict);
28
+ const unicode = (isType0 && toUnicode.size === 0 ? embeddedCmap(file, fontDict) : void 0) ?? toUnicode;
27
29
  const bytesPerCode = codeBytes;
30
+ const style = faceStyle(file, fontDict, isType0);
31
+ const name = embeddedFontName(file, fontDict);
32
+ const type3 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type3" ? type3Face(file, fontDict) : void 0;
33
+ const simple = simpleWidths(file, fontDict);
34
+ const width = isType0 ? cidWidths(file, fontDict) : type3 ? (code) => simple(code) * type3.matrix[0] * 1e3 : simple;
28
35
  return {
29
36
  bytesPerCode,
30
- decode: (codes) => codes.map((c) => toUnicode.get(c) ?? (bytesPerCode === 1 ? String.fromCharCode(c) : "")).join(""),
31
- width
37
+ ...type3 ? { type3 } : {},
38
+ ...name !== void 0 ? { name } : {},
39
+ decode: (codes) => codes.map((c) => unicode.get(c) ?? (bytesPerCode === 1 ? String.fromCharCode(c) : "")).join(""),
40
+ width,
41
+ ...style
32
42
  };
33
43
  }
44
+ /** The last Unicode code point in the BMP, and the surrogate block inside it. */
45
+ var BMP_END = 65535;
46
+ var SURROGATE_FIRST = 55296;
47
+ var SURROGATE_LAST = 57343;
48
+ /**
49
+ * §9.10.2 — code → Unicode read out of an embedded TrueType program's `cmap`,
50
+ * for a composite font that states no `/ToUnicode`.
51
+ *
52
+ * A `cmap` maps the other way, code point → glyph, so it is walked once and
53
+ * turned round. With `Identity-H` and no `/CIDToGIDMap` a code IS a glyph
54
+ * index, which is the case this exists for; a `/CIDToGIDMap` stream is read
55
+ * where one is present.
56
+ *
57
+ * Only TrueType (`/FontFile2`) is read. A CFF program (`/FontFile3`) carries
58
+ * its own charset and is a separate reading; a font with neither says nothing
59
+ * about its glyphs and nothing is invented.
60
+ */
61
+ function embeddedCmap(file, fontDict) {
62
+ const cidFont = descendantFont(file, fontDict);
63
+ const descriptor = file.resolve(cidFont.get("FontDescriptor") ?? PDF_NULL);
64
+ if (!(descriptor instanceof Map)) return void 0;
65
+ const program = file.resolve(descriptor.get("FontFile2") ?? PDF_NULL);
66
+ if (!(program instanceof PdfStream)) return void 0;
67
+ let glyphOf;
68
+ try {
69
+ glyphOf = parseTtf(file.streamData(program)).glyphForCodepoint;
70
+ } catch {
71
+ return;
72
+ }
73
+ const byGlyph = /* @__PURE__ */ new Map();
74
+ for (let cp = 32; cp <= BMP_END; cp++) {
75
+ if (cp >= SURROGATE_FIRST && cp <= SURROGATE_LAST) continue;
76
+ let gid = 0;
77
+ try {
78
+ gid = glyphOf(cp);
79
+ } catch {
80
+ continue;
81
+ }
82
+ if (gid > 0 && !byGlyph.has(gid)) byGlyph.set(gid, String.fromCodePoint(cp));
83
+ }
84
+ if (byGlyph.size === 0) return void 0;
85
+ const cidToGid = readCidToGid(file, cidFont);
86
+ if (!cidToGid) return byGlyph;
87
+ const out = /* @__PURE__ */ new Map();
88
+ cidToGid.forEach((gid, cid) => {
89
+ const text = byGlyph.get(gid);
90
+ if (text !== void 0) out.set(cid, text);
91
+ });
92
+ return out.size > 0 ? out : void 0;
93
+ }
94
+ /** §9.7.4.2 `/CIDToGIDMap` — a stream of two-byte glyph indices, CID by CID. */
95
+ function readCidToGid(file, cidFont) {
96
+ const map = file.resolve(cidFont.get("CIDToGIDMap") ?? PDF_NULL);
97
+ if (!(map instanceof PdfStream)) return void 0;
98
+ const bytes = file.streamData(map);
99
+ const out = [];
100
+ for (let i = 0; i + 1 < bytes.length; i += 2) out.push(bytes[i] << 8 | bytes[i + 1]);
101
+ return out;
102
+ }
103
+ /**
104
+ * §9.6.5 — a Type 3 font's glyphs are content streams, not outlines: what the
105
+ * face draws is whatever each procedure paints, in the resources the font
106
+ * states. `/Encoding` `/Differences` names the procedure a code selects and
107
+ * `/CharProcs` holds it.
108
+ *
109
+ * ContentStreamCycleType3insideType3.pdf is a page of them — a stroked square
110
+ * and a stroked triangle, with a second Type 3 font shown from inside the
111
+ * square — and with the procedures unread the page came back as two letters of
112
+ * substituted type an eighth of an inch tall.
113
+ */
114
+ function type3Face(file, fontDict) {
115
+ const procs = file.resolve(fontDict.get("CharProcs") ?? PDF_NULL);
116
+ if (!(procs instanceof Map)) return void 0;
117
+ const names = differences(file, fontDict);
118
+ const resourcesVal = file.resolve(fontDict.get("Resources") ?? PDF_NULL);
119
+ return {
120
+ matrix: fontMatrix(file, fontDict),
121
+ resources: resourcesVal instanceof Map ? resourcesVal : void 0,
122
+ proc: (code) => {
123
+ const glyph = names.get(code);
124
+ if (glyph === void 0) return void 0;
125
+ const stream = file.resolve(procs.get(glyph) ?? PDF_NULL);
126
+ return stream instanceof PdfStream ? stream : void 0;
127
+ }
128
+ };
129
+ }
130
+ /** §9.6.5 `/FontMatrix` — glyph space to text space; a thousandth by default. */
131
+ function fontMatrix(file, fontDict) {
132
+ const m = file.resolve(fontDict.get("FontMatrix") ?? PDF_NULL);
133
+ if (!Array.isArray(m) || m.length !== 6) return [
134
+ .001,
135
+ 0,
136
+ 0,
137
+ .001,
138
+ 0,
139
+ 0
140
+ ];
141
+ const n = m.map((v) => asNumber(file.resolve(v), 0));
142
+ return [
143
+ n[0],
144
+ n[1],
145
+ n[2],
146
+ n[3],
147
+ n[4],
148
+ n[5]
149
+ ];
150
+ }
151
+ /** §9.6.6.1 `/Encoding` `/Differences` — code → glyph name, as the array runs. */
152
+ function differences(file, fontDict) {
153
+ const out = /* @__PURE__ */ new Map();
154
+ const encoding = file.resolve(fontDict.get("Encoding") ?? PDF_NULL);
155
+ if (!(encoding instanceof Map)) return out;
156
+ const list = file.resolve(encoding.get("Differences") ?? PDF_NULL);
157
+ if (!Array.isArray(list)) return out;
158
+ let code = 0;
159
+ for (const entry of list) {
160
+ const value = file.resolve(entry);
161
+ if (typeof value === "number") code = value;
162
+ else if (value instanceof PdfName) out.set(code++, value.value);
163
+ }
164
+ return out;
165
+ }
166
+ /** §9.8.2 `/Flags` — bit 7 is Italic, bit 19 ForceBold (bits numbered from 1). */
167
+ var FLAG_ITALIC = 64;
168
+ var FLAG_FORCE_BOLD = 1 << 18;
169
+ /** §9.8.1 `/FontWeight` — 400 is normal, 700 bold; 600 is where "bold" begins. */
170
+ var BOLD_WEIGHT = 600;
171
+ /**
172
+ * §9.8.1 — whether the face a run is shown in is bold or slanted.
173
+ *
174
+ * A descriptor is the witness where there is one, and the ONLY witness: it
175
+ * states `/FontWeight`, `/ItalicAngle` and the `/Flags` bits, so one that gives
176
+ * neither a weight nor the ForceBold bit is saying the face is not bold.
177
+ * ArabicCIDTrueType.pdf shows why that matters — two of its four faces are
178
+ * called `NewBasrahBold` and `DamascusBold`, which is the family's own name and
179
+ * not a weight, and reading the name over the descriptor set two lines heavy
180
+ * that no reader sets heavy.
181
+ *
182
+ * The name is read only where no descriptor exists at all, which is the
183
+ * standard-14 case (§9.6.2.2): `Helvetica-BoldOblique` has nothing else to go
184
+ * on. The subset prefix (`ISVAYD+`) is dropped first — six arbitrary capitals
185
+ * may spell anything.
186
+ */
187
+ function faceStyle(file, fontDict, isType0) {
188
+ const owner = isType0 ? descendantFont(file, fontDict) : fontDict;
189
+ const descriptor = file.resolve(owner.get("FontDescriptor") ?? PDF_NULL);
190
+ if (descriptor instanceof Map) {
191
+ const flags = asNumber(file.resolve(descriptor.get("Flags") ?? PDF_NULL), 0);
192
+ const weight = asNumber(file.resolve(descriptor.get("FontWeight") ?? PDF_NULL), 0);
193
+ const slant = asNumber(file.resolve(descriptor.get("ItalicAngle") ?? PDF_NULL), 0);
194
+ const bold = weight >= BOLD_WEIGHT || (flags & FLAG_FORCE_BOLD) !== 0;
195
+ const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0;
196
+ return {
197
+ ...bold ? { bold } : {},
198
+ ...italic ? { italic } : {}
199
+ };
200
+ }
201
+ const name = asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)).replace(/^[A-Z]{6}\+/u, "");
202
+ const bold = /bold|black|heavy/iu.test(name);
203
+ const italic = /italic|oblique/iu.test(name);
204
+ return {
205
+ ...bold ? { bold } : {},
206
+ ...italic ? { italic } : {}
207
+ };
208
+ }
209
+ /** §9.7.4 — a `/Type0` font's one descendant CIDFont, which owns the descriptor. */
210
+ function descendantFont(file, fontDict) {
211
+ const descFonts = file.resolve(fontDict.get("DescendantFonts") ?? PDF_NULL);
212
+ const first = Array.isArray(descFonts) ? file.resolve(descFonts[0] ?? PDF_NULL) : PDF_NULL;
213
+ return first instanceof Map ? first : /* @__PURE__ */ new Map();
214
+ }
34
215
  function simpleWidths(file, fontDict) {
35
216
  const first = asNumber(file.resolve(fontDict.get("FirstChar") ?? PDF_NULL), 0);
36
217
  const widthsVal = file.resolve(fontDict.get("Widths") ?? PDF_NULL);
@@ -3,6 +3,7 @@ import { encodePng } from "../core/png-encode.js";
3
3
  import { lzwDecodeMsb } from "../core/lzw.js";
4
4
  import { reversePredictor } from "./predictor.js";
5
5
  import { decodeCcitt } from "./ccitt.js";
6
+ import { decodeJpeg } from "./jpeg.js";
6
7
  import { unzlibSync } from "fflate";
7
8
  //#region src/pdf-reader/image-decode.ts
8
9
  var MAX_PIXELS = 4e7;
@@ -30,10 +31,13 @@ function decodePdfImage(file, stream) {
30
31
  const filters = filterNames(file, d);
31
32
  const last = filters[filters.length - 1];
32
33
  if (last === "DCTDecode" || last === "DCT") {
33
- const degraded = hasSMask(file, d) ? "image transparency dropped (JPEG carries no alpha)" : void 0;
34
+ const jpegBytes = applyChainExceptLast(filters, stream.data);
35
+ const masked = hasSMask(file, d) ? maskedJpeg(file, d, jpegBytes, width, height) : void 0;
36
+ if (masked) return masked;
37
+ const degraded = hasSMask(file, d) ? "image transparency dropped (JPEG could not be decoded to attach its mask)" : void 0;
34
38
  return {
35
39
  ok: true,
36
- bytes: applyChainExceptLast(filters, stream.data),
40
+ bytes: jpegBytes,
37
41
  format: "jpeg",
38
42
  widthPx: width,
39
43
  heightPx: height,
@@ -276,14 +280,49 @@ function unpackSamples(raw, width, height, ncomp, bpc) {
276
280
  }
277
281
  return out;
278
282
  }
283
+ /**
284
+ * A JPEG with its `/SMask` folded in, as a PNG.
285
+ *
286
+ * A JPEG has no alpha channel, so the only way to honour the mask is to decode
287
+ * the picture, decode the mask, put them together and re-encode. Both are
288
+ * usually DCT themselves — 22060_A1_01_Plans.pdf stores each floor plan as a
289
+ * wash and its line work as a grey JPEG mask of the same 2480x2630 — so this
290
+ * costs a full baseline decode of two images, and is spent only where a mask
291
+ * actually exists.
292
+ *
293
+ * @returns The recomposed PNG, or `undefined` when either image is beyond the
294
+ * baseline decoder — the caller then carries the JPEG through as it is.
295
+ */
296
+ function maskedJpeg(file, d, jpegBytes, width, height) {
297
+ const image = decodeJpeg(jpegBytes);
298
+ if (!image) return void 0;
299
+ const alpha = decodeSMask(file, d, image.width, image.height);
300
+ if (!alpha) return void 0;
301
+ const { color: pngColor, samples } = combineAlpha(image.components === 1 ? {
302
+ color: "gray",
303
+ samples: image.samples
304
+ } : {
305
+ color: "rgb",
306
+ samples: image.samples
307
+ }, alpha);
308
+ return {
309
+ ok: true,
310
+ bytes: encodePng(image.width, image.height, pngColor, samples),
311
+ format: "png",
312
+ widthPx: width,
313
+ heightPx: height
314
+ };
315
+ }
279
316
  function decodeSMask(file, d, width, height) {
280
317
  const sm = file.resolve(d.get("SMask") ?? PDF_NULL);
281
318
  if (!(sm instanceof PdfStream)) return void 0;
282
319
  const sw = intOf(file.get(sm.dict, "Width"));
283
320
  const sh = intOf(file.get(sm.dict, "Height"));
284
321
  if (sw <= 0 || sh <= 0) return void 0;
285
- const decoded = decodeToSamples(file, sm, filterNames(file, sm.dict), sw, sh);
286
- if (typeof decoded === "string") return void 0;
322
+ const maskFilters = filterNames(file, sm.dict);
323
+ const maskLast = maskFilters[maskFilters.length - 1];
324
+ const decoded = maskLast === "DCTDecode" || maskLast === "DCT" ? jpegSamples(applyChainExceptLast(maskFilters, sm.data)) : decodeToSamples(file, sm, maskFilters, sw, sh);
325
+ if (decoded === void 0 || typeof decoded === "string") return void 0;
287
326
  const ch = decoded.color === "rgb" ? 3 : 1;
288
327
  const gray = new Uint8Array(sw * sh);
289
328
  for (let i = 0; i < sw * sh; i++) gray[i] = decoded.samples[i * ch];
@@ -298,6 +337,18 @@ function decodeSMask(file, d, width, height) {
298
337
  }
299
338
  return { data: out };
300
339
  }
340
+ /** A DCT-coded mask decoded to samples, or `undefined` past the baseline decoder. */
341
+ function jpegSamples(bytes) {
342
+ const decoded = decodeJpeg(bytes);
343
+ if (!decoded) return void 0;
344
+ return decoded.components === 1 ? {
345
+ color: "gray",
346
+ samples: decoded.samples
347
+ } : {
348
+ color: "rgb",
349
+ samples: decoded.samples
350
+ };
351
+ }
301
352
  function combineAlpha(color, alpha) {
302
353
  const px = alpha.data.length;
303
354
  if (color.color === "gray") {
@@ -17,6 +17,12 @@ export interface PdfImage {
17
17
  readonly y: number;
18
18
  /** Enclosing marked-content id, if the placement was inside a `/Figure`. */
19
19
  readonly mcid?: number;
20
+ /**
21
+ * §8.5.3 — where this was painted, as the chain of positions leading to it.
22
+ * The same key a lifted path carries, so the two can be ordered against each
23
+ * other: a picture drawn over a filled box has the larger key.
24
+ */
25
+ readonly orderKey: ReadonlyArray<number>;
20
26
  }
21
27
  /** The images lifted off one page plus any losses for images that could not be reconstructed. */
22
28
  export interface PageImages {
@@ -2,6 +2,7 @@ import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
2
2
  import { FEATURES } from "../core/ir/features.js";
3
3
  import { interpretContent, multiply } from "./content.js";
4
4
  import { decodePdfImage } from "./image-decode.js";
5
+ import { collectPageAppearances } from "./annots.js";
5
6
  //#region src/pdf-reader/images.ts
6
7
  var NO_FONTS = /* @__PURE__ */ new Map();
7
8
  var MAX_FORM_DEPTH = 12;
@@ -26,10 +27,23 @@ function collectPageImages(file, page) {
26
27
  detail
27
28
  });
28
29
  };
29
- const walk = (resources, content, baseCtm, depth, inheritedMcid) => {
30
+ const walk = (resources, content, baseCtm, depth, inheritedMcid, prefix) => {
30
31
  const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
31
32
  const xobjDict = xobjects instanceof Map ? xobjects : void 0;
32
- for (const placement of interpretContent(content, NO_FONTS, baseCtm).images) {
33
+ const result = interpretContent(content, NO_FONTS, baseCtm);
34
+ const patterns = resources ? file.get(resources, "Pattern") : PDF_NULL;
35
+ const patternDict = patterns instanceof Map ? patterns : void 0;
36
+ for (const vector of result.vectors) {
37
+ if (vector.patternName === void 0 || depth >= MAX_FORM_DEPTH) continue;
38
+ const stream = patternDict ? file.resolve(patternDict.get(vector.patternName) ?? PDF_NULL) : PDF_NULL;
39
+ if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
40
+ if (file.get(stream.dict, "PatternType") !== 1) continue;
41
+ visiting.add(stream);
42
+ const patternRes = file.get(stream.dict, "Resources");
43
+ walk(patternRes instanceof Map ? patternRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), baseCtm), depth + 1, inheritedMcid, [...prefix, vector.order]);
44
+ visiting.delete(stream);
45
+ }
46
+ for (const placement of result.images) {
33
47
  if (images.length >= MAX_IMAGES) return;
34
48
  const stream = xobjDict ? file.resolve(xobjDict.get(placement.name) ?? PDF_NULL) : PDF_NULL;
35
49
  if (!(stream instanceof PdfStream)) continue;
@@ -38,13 +52,16 @@ function collectPageImages(file, page) {
38
52
  if (subtype === "Image") {
39
53
  const decoded = decodePdfImage(file, stream);
40
54
  if (decoded.ok) {
41
- images.push(geometry(placement.ctm, decoded, mcid));
55
+ images.push({
56
+ ...geometry(placement.ctm, decoded, mcid),
57
+ orderKey: [...prefix, placement.order]
58
+ });
42
59
  if (decoded.degraded) addLoss("degraded", decoded.degraded);
43
60
  } else addLoss(decoded.severity, decoded.detail);
44
61
  } else if (subtype === "Form" && depth < MAX_FORM_DEPTH && !visiting.has(stream)) {
45
62
  visiting.add(stream);
46
63
  const formRes = file.get(stream.dict, "Resources");
47
- walk(formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, mcid);
64
+ walk(formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, mcid, [...prefix, placement.order]);
48
65
  visiting.delete(stream);
49
66
  }
50
67
  }
@@ -56,7 +73,10 @@ function collectPageImages(file, page) {
56
73
  1,
57
74
  0,
58
75
  0
59
- ], 0, void 0);
76
+ ], 0, void 0, []);
77
+ collectPageAppearances(file, page).forEach((appearance, index) => {
78
+ walk(appearance.resources ?? page.resources, file.streamData(appearance.stream), appearance.ctm, 1, void 0, [Number.MAX_SAFE_INTEGER, index]);
79
+ });
60
80
  return {
61
81
  images,
62
82
  losses: [...lossByDetail.values()]
@@ -0,0 +1,18 @@
1
+ /** A decoded JPEG: 8-bit interleaved samples, one or three components. */
2
+ export interface DecodedJpeg {
3
+ readonly width: number;
4
+ readonly height: number;
5
+ /** 1 = grayscale, 3 = RGB (already converted from YCbCr where it applies). */
6
+ readonly components: 1 | 3;
7
+ /** `width * height * components` bytes, row-major. */
8
+ readonly samples: Uint8Array;
9
+ }
10
+ /**
11
+ * Decode a baseline JPEG to interleaved 8-bit samples.
12
+ *
13
+ * @param bytes The JPEG stream, starting at its SOI marker.
14
+ * @returns The decoded image, or `undefined` when the stream is not a baseline
15
+ * JPEG this decoder handles (progressive, arithmetic, 12-bit, CMYK) or
16
+ * is malformed — the caller then carries the original bytes through.
17
+ */
18
+ export declare function decodeJpeg(bytes: Uint8Array): DecodedJpeg | undefined;