reamkit 1.26.0 → 1.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +9 -4
  2. package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
  3. package/dist/esm/pdf-reader/annot-draw.js +374 -0
  4. package/dist/esm/pdf-reader/annots.d.ts +3 -1
  5. package/dist/esm/pdf-reader/annots.js +18 -4
  6. package/dist/esm/pdf-reader/ccitt.d.ts +18 -0
  7. package/dist/esm/pdf-reader/ccitt.js +70 -2
  8. package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
  9. package/dist/esm/pdf-reader/cie-color.js +112 -0
  10. package/dist/esm/pdf-reader/content.d.ts +47 -3
  11. package/dist/esm/pdf-reader/content.js +108 -13
  12. package/dist/esm/pdf-reader/display.d.ts +1 -1
  13. package/dist/esm/pdf-reader/display.js +21 -6
  14. package/dist/esm/pdf-reader/document.d.ts +6 -0
  15. package/dist/esm/pdf-reader/document.js +29 -1
  16. package/dist/esm/pdf-reader/embedded-fonts.d.ts +12 -3
  17. package/dist/esm/pdf-reader/embedded-fonts.js +22 -3
  18. package/dist/esm/pdf-reader/flow-build.d.ts +12 -4
  19. package/dist/esm/pdf-reader/flow-build.js +102 -8
  20. package/dist/esm/pdf-reader/font.js +77 -16
  21. package/dist/esm/pdf-reader/function.d.ts +16 -0
  22. package/dist/esm/pdf-reader/function.js +414 -0
  23. package/dist/esm/pdf-reader/image-decode.d.ts +8 -4
  24. package/dist/esm/pdf-reader/image-decode.js +192 -11
  25. package/dist/esm/pdf-reader/images.d.ts +12 -0
  26. package/dist/esm/pdf-reader/images.js +109 -9
  27. package/dist/esm/pdf-reader/jbig2.d.ts +23 -0
  28. package/dist/esm/pdf-reader/jbig2.js +23 -14
  29. package/dist/esm/pdf-reader/layout.d.ts +32 -0
  30. package/dist/esm/pdf-reader/layout.js +189 -40
  31. package/dist/esm/pdf-reader/lexer.d.ts +2 -0
  32. package/dist/esm/pdf-reader/lexer.js +4 -0
  33. package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
  34. package/dist/esm/pdf-reader/optional-content.js +93 -0
  35. package/dist/esm/pdf-reader/reader.d.ts +5 -2
  36. package/dist/esm/pdf-reader/reader.js +80 -7
  37. package/dist/esm/pdf-reader/shading.d.ts +60 -8
  38. package/dist/esm/pdf-reader/shading.js +124 -20
  39. package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
  40. package/dist/esm/pdf-reader/standard-metrics.js +18 -0
  41. package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
  42. package/dist/esm/pdf-reader/standard-widths.js +62 -0
  43. package/dist/esm/pdf-reader/tagged.js +204 -32
  44. package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
  45. package/dist/esm/pdf-reader/text-rules.js +112 -0
  46. package/dist/esm/pdf-reader/text.js +78 -4
  47. package/dist/esm/pdf-reader/vector.d.ts +5 -0
  48. package/dist/esm/pdf-reader/vector.js +39 -26
  49. package/dist/esm/word/docx-writer.js +73 -11
  50. package/package.json +1 -1
@@ -1,11 +1,14 @@
1
1
  import { pt } from "../core/ir/units.js";
2
2
  import { ResourceStore } from "../core/ir/resources.js";
3
- import { displayOf, placeVectors } from "./display.js";
4
- import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock } from "./flow-build.js";
3
+ import { FEATURES } from "../core/ir/features.js";
5
4
  import { collectEmbeddedFonts } from "./embedded-fonts.js";
6
- import { collectPageImages } from "./images.js";
7
5
  import { extractPageText } from "./text.js";
8
6
  import { collectPageVectors } from "./vector.js";
7
+ import { displayOf, placeRuns, placeVectors } from "./display.js";
8
+ import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock, withMeasuredMargins } from "./flow-build.js";
9
+ import { collectPageImages } from "./images.js";
10
+ import { markDrawnRules } from "./text-rules.js";
11
+ import { endedParagraph } from "./layout.js";
9
12
  import { readStructTree } from "./struct-tree.js";
10
13
  //#region src/pdf-reader/tagged.ts
11
14
  var ASSUMED_CONTENT_WIDTH_PT = 468;
@@ -27,9 +30,23 @@ function reconstructTaggedPdf(file) {
27
30
  const root = readStructTree(file);
28
31
  if (!root) return void 0;
29
32
  const pages = file.pages();
30
- const pageRuns = pages.map((page) => {
33
+ const shown = pages.map((page) => displayOf(page));
34
+ const placedRuns = pages.map((page, i) => placeRuns(extractPageText(file, page), shown[i]));
35
+ const vectorLosses = [];
36
+ const pageVectors = pages.map((page, i) => {
37
+ const lifted = collectPageVectors(file, page, collectPageImages(file, page).images.map((img) => ({
38
+ minX: img.x,
39
+ minY: img.y,
40
+ maxX: img.x + img.widthPt,
41
+ maxY: img.y + img.heightPt
42
+ })));
43
+ vectorLosses.push(...lifted.losses);
44
+ return placeVectors(lifted.vectors, shown[i]);
45
+ });
46
+ const ruled = placedRuns.map((runs, i) => markDrawnRules(runs, pageVectors[i] ?? []));
47
+ const pageRuns = ruled.map(({ runs }) => {
31
48
  const byMcid = /* @__PURE__ */ new Map();
32
- for (const run of extractPageText(file, page)) {
49
+ for (const run of runs) {
33
50
  if (run.mcid === void 0) continue;
34
51
  const list = byMcid.get(run.mcid);
35
52
  if (list) list.push(run);
@@ -37,7 +54,12 @@ function reconstructTaggedPdf(file) {
37
54
  }
38
55
  return byMcid;
39
56
  });
40
- const runsOfMcid = (page, mcid) => pageRuns[page]?.get(mcid) ?? [];
57
+ const claimed = /* @__PURE__ */ new Set();
58
+ const runsOfMcid = (page, mcid) => {
59
+ const runs = pageRuns[page]?.get(mcid) ?? [];
60
+ for (const run of runs) claimed.add(run);
61
+ return runs;
62
+ };
41
63
  const resources = new ResourceStore();
42
64
  const pageImages = pages.map((page) => collectPageImages(file, page));
43
65
  const imageLosses = dedupeLosses(pageImages.flatMap((p) => p.losses));
@@ -58,23 +80,123 @@ function reconstructTaggedPdf(file) {
58
80
  const spans = [];
59
81
  node.mcids.forEach(({ page, mcid }, i) => {
60
82
  if (i > 0) spans.push({ text: " " });
61
- for (const run of runsOfMcid(page, mcid)) spans.push({
62
- text: run.text,
63
- sizePt: run.fontSizePt,
64
- ...run.colorHex !== "000000" ? { colorHex: run.colorHex } : {},
65
- ...run.fontName !== void 0 ? { fontName: run.fontName } : {},
66
- ...run.outlineHex !== void 0 ? { outline: {
67
- colorHex: run.outlineHex,
68
- widthPt: pt(run.outlineWidthPt ?? 1)
69
- } } : {},
70
- ...run.bold ? { bold: true } : {},
71
- ...run.italic ? { italic: true } : {},
72
- ...run.href !== void 0 ? { href: run.href } : {}
73
- });
83
+ for (const run of runsOfMcid(page, mcid)) spans.push(spanOf(run));
74
84
  });
75
85
  return spans;
76
86
  };
87
+ /** One run as the span that carries everything the page showed it with. */
88
+ const spanOf = (run) => ({
89
+ text: run.text.replaceAll("�", ""),
90
+ sizePt: run.fontSizePt,
91
+ ...run.colorHex !== "000000" ? { colorHex: run.colorHex } : {},
92
+ ...run.fontName !== void 0 ? { fontName: run.fontName } : {},
93
+ ...run.outlineHex !== void 0 ? { outline: {
94
+ colorHex: run.outlineHex,
95
+ widthPt: pt(run.outlineWidthPt ?? 1)
96
+ } } : {},
97
+ ...run.bold ? { bold: true } : {},
98
+ ...run.italic ? { italic: true } : {},
99
+ ...run.markup !== void 0 ? { markup: run.markup } : {},
100
+ ...run.href !== void 0 ? { href: run.href } : {}
101
+ });
77
102
  const collectText = (node) => squash([textOf(node), ...node.children.map(collectText)].join(" "));
103
+ const setting = /* @__PURE__ */ new Map();
104
+ /** The topmost and bottommost baseline under a node, and its largest face. */
105
+ function baselinesOf(node) {
106
+ let top;
107
+ let bottom;
108
+ let size = 0;
109
+ const visit = (n) => {
110
+ for (const { page, mcid } of n.mcids) for (const run of runsOfMcid(page, mcid)) {
111
+ if (top === void 0 || run.y > top) top = run.y;
112
+ if (bottom === void 0 || run.y < bottom) bottom = run.y;
113
+ if (run.fontSizePt > size) size = run.fontSizePt;
114
+ }
115
+ for (const child of n.children) visit(child);
116
+ };
117
+ visit(node);
118
+ return top !== void 0 && bottom !== void 0 && size > 0 ? {
119
+ top,
120
+ bottom,
121
+ size
122
+ } : void 0;
123
+ }
124
+ /**
125
+ * The element's own lines, gathered into the paragraphs they were SET as.
126
+ *
127
+ * A tree names elements, not lines, and a producer may put a whole page under
128
+ * one: annotation-underline.pdf marks every glyph on its page with the same
129
+ * id, so a heading, a blank line and a body line came back as one paragraph
130
+ * and re-wrapped into a single run-together line. The page still says where
131
+ * one ended — a line stopping well short of the measure stopped because its
132
+ * author stopped it — and that is the rule the heuristic reading already uses
133
+ * ({@link endedParagraph}). A properly tagged paragraph's inner lines all
134
+ * reach the measure, so nothing there changes.
135
+ */
136
+ function settingsOf(node) {
137
+ const spans = spansOf(node);
138
+ const lines = linesOf(node);
139
+ if (lines.length < 2) {
140
+ const set = baselinesOf(node);
141
+ return set ? [{
142
+ spans,
143
+ set
144
+ }] : [];
145
+ }
146
+ const measure = {
147
+ left: Math.min(...lines.map((l) => l.x)),
148
+ right: Math.max(...lines.map((l) => l.x + l.width))
149
+ };
150
+ const groups = [];
151
+ let prev;
152
+ for (const line of lines) {
153
+ const gap = prev === void 0 ? 0 : prev.y - line.y;
154
+ const opened = prev !== void 0 && gap > line.fontSize * 1.5;
155
+ if (groups.length === 0 || opened || prev && endedParagraph(prev, line, measure)) groups.push([]);
156
+ groups[groups.length - 1].push(line);
157
+ prev = line;
158
+ }
159
+ return groups.map((g) => ({
160
+ spans: g.flatMap((l, i) => i > 0 ? [{ text: " " }, ...l.spans] : [...l.spans]),
161
+ set: {
162
+ top: g[0].y,
163
+ bottom: g[g.length - 1].y,
164
+ size: Math.max(...g.map((l) => l.fontSize))
165
+ }
166
+ }));
167
+ }
168
+ /** The node's runs clustered onto the baselines they were shown on. */
169
+ function linesOf(node) {
170
+ const rows = [];
171
+ const visit = (n) => {
172
+ for (const { page, mcid } of n.mcids) for (const run of runsOfMcid(page, mcid)) {
173
+ if (run.angleDeg !== void 0) return;
174
+ const row = rows.find((r) => Math.abs(r.y - run.y) <= Math.max(r.size, run.fontSizePt) * .5);
175
+ if (row) {
176
+ row.runs.push(run);
177
+ row.size = Math.max(row.size, run.fontSizePt);
178
+ row.y = Math.min(row.y, run.y);
179
+ } else rows.push({
180
+ y: run.y,
181
+ size: run.fontSizePt,
182
+ runs: [run]
183
+ });
184
+ }
185
+ for (const child of n.children) visit(child);
186
+ };
187
+ visit(node);
188
+ return rows.sort((a, b) => b.y - a.y).map(({ y, runs }) => {
189
+ const ordered = [...runs].sort((a, b) => a.x - b.x);
190
+ const x = Math.min(...ordered.map((r) => r.x));
191
+ return {
192
+ y,
193
+ x,
194
+ width: Math.max(...ordered.map((r) => r.endX)) - x,
195
+ fontSize: Math.max(...ordered.map((r) => r.fontSizePt)),
196
+ spans: ordered.map((r) => spanOf(r))
197
+ };
198
+ }).filter((l) => l.spans.some((sp) => sp.text.trim().length > 0));
199
+ }
78
200
  function emit(node, out) {
79
201
  if (node.type === "Table") {
80
202
  const table = buildTable(node);
@@ -94,7 +216,11 @@ function reconstructTaggedPdf(file) {
94
216
  return;
95
217
  }
96
218
  if (node.children.length === 0) {
97
- if (textOf(node).length > 0) out.push(paragraphFromRuns(spansOf(node), headingLevel(node.type)));
219
+ if (textOf(node).length > 0) for (const part of settingsOf(node)) {
220
+ const el = paragraphFromRuns(part.spans, headingLevel(node.type));
221
+ setting.set(el, part.set);
222
+ out.push(el);
223
+ }
98
224
  return;
99
225
  }
100
226
  for (const child of node.children) emit(child, out);
@@ -164,21 +290,19 @@ function reconstructTaggedPdf(file) {
164
290
  }
165
291
  const body = [];
166
292
  emit(root, body);
293
+ spaceParagraphs(body, setting, shown[0]?.height ?? 0);
167
294
  let zOrder = -1e6;
168
- pages.forEach((page, index) => {
169
- const display = displayOf(page);
295
+ imageLosses.push(...vectorLosses);
296
+ pages.forEach((_page, index) => {
170
297
  const frame = {
171
298
  left: 0,
172
- top: display.height
299
+ top: shown[index].height
173
300
  };
174
- const lifted = collectPageVectors(file, page, (pageImages[index]?.images ?? []).map((img) => ({
175
- minX: img.x,
176
- minY: img.y,
177
- maxX: img.x + img.widthPt,
178
- maxY: img.y + img.heightPt
179
- })));
180
- imageLosses.push(...lifted.losses);
181
- for (const v of placeVectors(lifted.vectors, display)) body.push(shapeBlock(v, frame, zOrder++));
301
+ const taken = ruled[index]?.consumed;
302
+ for (const v of pageVectors[index] ?? []) {
303
+ if (taken?.has(v) === true) continue;
304
+ body.push(shapeBlock(v, frame, zOrder++, true));
305
+ }
182
306
  });
183
307
  const orphans = [];
184
308
  pageImages.forEach((p, page) => {
@@ -190,8 +314,16 @@ function reconstructTaggedPdf(file) {
190
314
  orphans.sort((a, b) => a.page - b.page || b.img.y - a.img.y);
191
315
  for (const { img } of orphans) body.push(imageBlock(img, resources));
192
316
  if (body.length === 0) return void 0;
317
+ if (placedRuns.some((page) => page.some((r) => r.text.includes("�")))) imageLosses.push({
318
+ severity: "dropped",
319
+ feature: FEATURES.text,
320
+ detail: "some glyphs map to no character — the font states no /ToUnicode and its program says nothing either, so that text is unrecoverable"
321
+ });
322
+ const onPage = placedRuns.flat().reduce((n, r) => n + r.text.length, 0);
323
+ const reached = [...claimed].reduce((n, r) => n + r.text.length, 0);
324
+ if (onPage > 0 && reached * 2 < onPage) return void 0;
193
325
  return {
194
- doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages), collectEmbeddedFonts(file, pages)),
326
+ doc: buildFlowDoc(body, resources, withMeasuredMargins(sectionFromPdfPages(pages), shown, placedRuns), collectEmbeddedFonts(file, pages, imageLosses)),
195
327
  losses: imageLosses
196
328
  };
197
329
  }
@@ -289,5 +421,45 @@ function headingLevel(type) {
289
421
  const m = /^H([1-6])$/.exec(type);
290
422
  return m ? Number(m[1]) - 1 : void 0;
291
423
  }
424
+ /**
425
+ * §17.3.1.33 `w:spacing` — the space the SOURCE left before each paragraph.
426
+ *
427
+ * A structure tree names the words and says nothing about how far apart they
428
+ * stood, so every tagged PDF came back at one flat leading: on
429
+ * annotation-polyline-polygon-without-appearance.pdf the two labels, set a
430
+ * third of a page apart above their own drawings, arrived as two lines
431
+ * touching. The page still says it — the gap between the last baseline of one
432
+ * paragraph and the first of the next, less the line it would have taken
433
+ * anyway. This is the rule the heuristic reading already uses, applied to the
434
+ * paragraphs the tree named.
435
+ *
436
+ * @param body The body elements, in order; amended in place.
437
+ * @param setting Where each paragraph was set, for those that were.
438
+ * @param pageHeight The shown page's height, which bounds any one gap.
439
+ */
440
+ function spaceParagraphs(body, setting, pageHeight) {
441
+ let prev;
442
+ body.forEach((el, i) => {
443
+ const here = setting.get(el);
444
+ if (!here) return;
445
+ const gap = prev !== void 0 ? prev.bottom - here.top : 0;
446
+ prev = here;
447
+ if (!(gap > 0)) return;
448
+ const opened = gap - here.size * 1.2;
449
+ if (!(opened > here.size * .3)) return;
450
+ if (el.kind !== "paragraph") return;
451
+ const most = pageHeight > 0 ? pageHeight / 3 : here.size * 3;
452
+ body[i] = {
453
+ ...el,
454
+ paragraph: {
455
+ ...el.paragraph,
456
+ properties: {
457
+ ...el.paragraph.properties,
458
+ spacingBefore: pt(Math.min(opened, most))
459
+ }
460
+ }
461
+ };
462
+ });
463
+ }
292
464
  //#endregion
293
465
  export { reconstructTaggedPdf };
@@ -0,0 +1,16 @@
1
+ import { TextRun } from './content.js';
2
+ import { PdfVector } from './vector.js';
3
+ /** The runs with their drawn rules read onto them, and the rules so read. */
4
+ export interface DrawnRules {
5
+ readonly runs: Array<TextRun>;
6
+ /** The vectors the runs took over: painting them again would double them. */
7
+ readonly consumed: ReadonlySet<PdfVector>;
8
+ }
9
+ /**
10
+ * Read the underlines and strikeouts a page DREW onto the runs they mark.
11
+ *
12
+ * @param runs The page's runs, placed on the shown page.
13
+ * @param vectors The page's lifted paths, placed the same way.
14
+ * @returns The runs, marked; and the paths that became those marks.
15
+ */
16
+ export declare function markDrawnRules(runs: ReadonlyArray<TextRun>, vectors: ReadonlyArray<PdfVector>): DrawnRules;
@@ -0,0 +1,112 @@
1
+ //#region src/pdf-reader/text-rules.ts
2
+ /** A rule no thicker than this is a line under words, not a box. */
3
+ var THICKEST_PT = 2;
4
+ /** …and no thinner than this: below it the page drew a seam, not a mark. */
5
+ var THINNEST_PT = .2;
6
+ /** …or than this fraction of the face it underlines, for a large one. */
7
+ var THICKEST_EM = .09;
8
+ /** How much of a run's advance a rule must cover to be that run's own. */
9
+ var COVERS = .6;
10
+ /**
11
+ * How far past the words it may run, in points, before it is a rule about
12
+ * something else.
13
+ *
14
+ * An underline begins and ends with the words it underlines; a table's cell
15
+ * border begins at the CELL, which is the text's own edge less the padding, and
16
+ * ends at the cell's other edge however short the text stops. TAMReview.pdf's
17
+ * tables are ruled 86.2 to 509.0 across a measure whose text stops well before
18
+ * it, and a proportional allowance let every one of them through.
19
+ */
20
+ var OVERHANG_PT = 2;
21
+ /** An underline sits within this fraction of the size below the baseline. */
22
+ var UNDER_EM = .35;
23
+ /** A strikeout crosses between these fractions of the size ABOVE it. */
24
+ var STRIKE_LOW = .15;
25
+ var STRIKE_HIGH = .5;
26
+ /**
27
+ * Read the underlines and strikeouts a page DREW onto the runs they mark.
28
+ *
29
+ * @param runs The page's runs, placed on the shown page.
30
+ * @param vectors The page's lifted paths, placed the same way.
31
+ * @returns The runs, marked; and the paths that became those marks.
32
+ */
33
+ function markDrawnRules(runs, vectors) {
34
+ const marks = /* @__PURE__ */ new Map();
35
+ const consumed = /* @__PURE__ */ new Set();
36
+ for (const v of vectors) {
37
+ if (!isRule(v)) continue;
38
+ const under = coveredRuns(runs, v, "under");
39
+ const through = under.length > 0 ? [] : coveredRuns(runs, v, "through");
40
+ const hit = under.length > 0 ? under : through;
41
+ if (hit.length === 0) continue;
42
+ const left = Math.min(...hit.map((r) => Math.min(r.x, r.endX)));
43
+ const right = Math.max(...hit.map((r) => Math.max(r.x, r.endX)));
44
+ if (!(right - left > 0)) continue;
45
+ if (left - v.minX > OVERHANG_PT || v.maxX - right > OVERHANG_PT) continue;
46
+ for (const run of hit) {
47
+ const had = marks.get(run) ?? {};
48
+ marks.set(run, under.length > 0 && v.fillHex !== void 0 ? {
49
+ ...had,
50
+ underline: v.fillHex
51
+ } : {
52
+ ...had,
53
+ strike: true
54
+ });
55
+ }
56
+ consumed.add(v);
57
+ }
58
+ if (marks.size === 0) return {
59
+ runs: [...runs],
60
+ consumed
61
+ };
62
+ return {
63
+ runs: runs.map((run) => {
64
+ const mark = marks.get(run);
65
+ if (!mark) return run;
66
+ return {
67
+ ...run,
68
+ markup: {
69
+ ...run.markup,
70
+ ...mark.underline !== void 0 && run.markup?.underline === void 0 ? {
71
+ underline: "single",
72
+ underlineHex: mark.underline
73
+ } : {},
74
+ ...mark.strike === true ? { strike: true } : {}
75
+ }
76
+ };
77
+ }),
78
+ consumed
79
+ };
80
+ }
81
+ /** A thin filled bar, which is the only shape an underline is drawn as. */
82
+ function isRule(v) {
83
+ if (v.fillHex === void 0 || v.strokeHex !== void 0 || v.gradient !== void 0) return false;
84
+ if (v.fillHex === "FFFFFF") return false;
85
+ const h = v.maxY - v.minY;
86
+ const w = v.maxX - v.minX;
87
+ return h >= THINNEST_PT && h <= THICKEST_PT && w > 2;
88
+ }
89
+ /**
90
+ * The runs a rule marks: on one baseline, covered along their advance, and at
91
+ * the offset from that baseline the mark is drawn at.
92
+ */
93
+ function coveredRuns(runs, v, where) {
94
+ const out = [];
95
+ const mid = (v.minY + v.maxY) / 2;
96
+ for (const run of runs) {
97
+ const size = run.fontSizePt;
98
+ if (!(size > 0) || run.angleDeg !== void 0) continue;
99
+ if (v.maxY - v.minY > Math.max(THICKEST_PT, size * THICKEST_EM)) continue;
100
+ const below = run.y - mid;
101
+ if (!(where === "under" ? below > 0 && below < size * UNDER_EM : below < -size * STRIKE_LOW && below > -size * STRIKE_HIGH)) continue;
102
+ const left = Math.min(run.x, run.endX);
103
+ const right = Math.max(run.x, run.endX);
104
+ const advance = right - left;
105
+ if (!(advance > 0)) continue;
106
+ if (Math.min(right, v.maxX) - Math.max(left, v.minX) < advance * COVERS) continue;
107
+ out.push(run);
108
+ }
109
+ return out;
110
+ }
111
+ //#endregion
112
+ export { markDrawnRules };
@@ -1,6 +1,8 @@
1
1
  import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
2
2
  import { IDENTITY, interpretContent, multiply } from "./content.js";
3
+ import { textMarkupOf } from "./annot-draw.js";
3
4
  import { collectPageAppearances } from "./annots.js";
5
+ import { hiddenProperties, hiddenXObject } from "./optional-content.js";
4
6
  import { buildContentFont } from "./font.js";
5
7
  import { patternTint } from "./pattern-tint.js";
6
8
  //#region src/pdf-reader/text.ts
@@ -21,17 +23,88 @@ function extractPageText(file, page) {
21
23
  collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs);
22
24
  for (const appearance of collectPageAppearances(file, page)) collectRuns(file, appearance.resources ?? page.resources, file.streamData(appearance.stream), appearance.ctm, 1, new Set([appearance.stream]), runs);
23
25
  const links = collectLinks(file, page);
24
- if (links.length === 0) return runs;
25
- return runs.map((run) => {
26
+ const marks = collectTextMarkup(file, page);
27
+ if (links.length === 0 && marks.length === 0) return runs;
28
+ return runs.flatMap((run) => {
26
29
  const link = links.find((l) => inRect(run.x, run.y, l.rect));
27
- return link ? {
30
+ const linked = link ? {
28
31
  ...run,
29
32
  href: link.href
30
33
  } : run;
34
+ const marked = marks.find((m) => m.quads.some((q) => touches(q, linked)));
35
+ if (!marked) return [linked];
36
+ const quad = marked.quads.find((q) => touches(q, linked));
37
+ return quad ? markedPieces(linked, quad, marked.mark) : [linked];
31
38
  });
32
39
  }
40
+ /**
41
+ * §12.5.6.10 — whether a marked quad reaches this run at all.
42
+ *
43
+ * The quad is a box round a run of text and the run is a baseline with an
44
+ * advance, so the test is the baseline falling inside the box's height while
45
+ * the two overlap horizontally by more than a hair.
46
+ */
47
+ function touches(q, run) {
48
+ if (run.y < q.y0 || run.y > q.y0 + q.h) return false;
49
+ const left = Math.min(run.x, run.endX);
50
+ const right = Math.max(run.x, run.endX);
51
+ return Math.min(right, q.x1) - Math.max(left, q.x0) > .5;
52
+ }
53
+ /**
54
+ * The run cut at the quad's edges, so only what the quad covers is marked.
55
+ *
56
+ * A quad round ONE WORD of a line must not claim the line: highlight_popup.pdf
57
+ * marks "PDF.js" in "Hello PDF.js World", which is one run, and the whole line
58
+ * came back highlighted. The run states its advance and its text and not where
59
+ * each glyph fell inside it, so the cut is made by proportion of the advance —
60
+ * a word off by a fraction of a letter where the face is proportional, against
61
+ * a line marked entirely wrongly.
62
+ */
63
+ function markedPieces(run, q, mark) {
64
+ const left = Math.min(run.x, run.endX);
65
+ const advance = Math.max(run.x, run.endX) - left;
66
+ const chars = [...run.text];
67
+ if (!(advance > 0) || chars.length === 0) return [{
68
+ ...run,
69
+ markup: mark
70
+ }];
71
+ const at = (x) => Math.max(0, Math.min(chars.length, Math.round((x - left) / advance * chars.length)));
72
+ const from = at(q.x0);
73
+ const to = at(q.x1);
74
+ if (from <= 0 && to >= chars.length) return [{
75
+ ...run,
76
+ markup: mark
77
+ }];
78
+ if (to <= from) return [run];
79
+ const per = advance / chars.length;
80
+ const piece = (a, b, marked) => ({
81
+ ...run,
82
+ text: chars.slice(a, b).join(""),
83
+ x: left + a * per,
84
+ endX: left + b * per,
85
+ ...marked ? { markup: mark } : {}
86
+ });
87
+ return [
88
+ ...from > 0 ? [piece(0, from, false)] : [],
89
+ piece(from, to, true),
90
+ ...to < chars.length ? [piece(to, chars.length, false)] : []
91
+ ];
92
+ }
93
+ /** Every text-markup annotation on the page, with the boxes it marks. */
94
+ function collectTextMarkup(file, page) {
95
+ const annots = file.get(page.dict, "Annots");
96
+ if (!Array.isArray(annots)) return [];
97
+ const out = [];
98
+ for (const entry of annots) {
99
+ const annot = file.resolve(entry);
100
+ if (!(annot instanceof Map)) continue;
101
+ const mark = textMarkupOf(file, annot);
102
+ if (mark) out.push(mark);
103
+ }
104
+ return out;
105
+ }
33
106
  function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
34
- const result = interpretContent(content, buildFonts(file, resources), baseCtm);
107
+ const result = interpretContent(content, buildFonts(file, resources), baseCtm, void 0, void 0, void 0, hiddenProperties(file, resources));
35
108
  out.push(...result.texts.map((r) => withPatternColour(file, resources, r, visiting)));
36
109
  if (depth >= MAX_FORM_DEPTH) return;
37
110
  for (const glyph of result.glyphs) {
@@ -46,6 +119,7 @@ function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
46
119
  for (const placement of result.images) {
47
120
  const stream = file.resolve(xobjects.get(placement.name) ?? PDF_NULL);
48
121
  if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
122
+ if (hiddenXObject(file, stream)) continue;
49
123
  const sub = file.get(stream.dict, "Subtype");
50
124
  if (!(sub instanceof PdfName) || sub.value !== "Form") continue;
51
125
  visiting.add(stream);
@@ -24,6 +24,11 @@ export interface PdfVector {
24
24
  readonly gradient?: ShapeGradient;
25
25
  /** §11.6.4.4 — how opaque the fill is, when the page asked for less than all. */
26
26
  readonly alpha?: number;
27
+ /**
28
+ * §11.3.5 — the fill only DARKENS what it covers, so the marks under it read
29
+ * through. A highlighter is this, and nothing on a page reads it as paint.
30
+ */
31
+ readonly darkens?: boolean;
27
32
  /** Present iff a qualifying stroke survived (EP11). */
28
33
  readonly strokeHex?: string;
29
34
  /** Stroke width in page-space points (EP11). */
@@ -1,30 +1,32 @@
1
1
  import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
2
2
  import { FEATURES } from "../core/ir/features.js";
3
3
  import { IDENTITY, interpretContent, multiply } from "./content.js";
4
+ import { buildAlphaMap, buildColorSpaceMap, buildShadingMap } from "./shading.js";
4
5
  import { collectPageAppearances } from "./annots.js";
6
+ import { hiddenProperties, hiddenXObject } from "./optional-content.js";
5
7
  import { buildFonts } from "./text.js";
6
- import { buildAlphaMap, buildColorSpaceMap, buildShadingMap } from "./shading.js";
7
8
  //#region src/pdf-reader/vector.ts
8
- /**
9
- * Every vector the page paints, its FORM XOBJECTS included (§8.8).
10
- *
11
- * A `Do` of a form is a call: its content stream draws in the caller's space
12
- * through the form's own `/Matrix`. Interpreting the page stream alone reads
13
- * only what the page drew directly, and a document that puts its artwork in
14
- * forms — as every CAD and drawing producer does — comes back with none of it.
15
- * 22060_A1_01_Plans.pdf holds eleven, and its floor plans were simply absent.
16
- *
17
- * The same walk `collectPageImages` already makes, with the same depth and
18
- * cycle guards, collecting paths instead of pictures.
19
- */
20
- function paintedVectors(file, page, shadings, alphas, spaces) {
9
+ function paintedVectors(file, page) {
21
10
  const out = [];
22
11
  const visiting = /* @__PURE__ */ new Set();
12
+ const stateCache = /* @__PURE__ */ new Map();
13
+ const mapsOf = (resources) => {
14
+ const had = stateCache.get(resources);
15
+ if (had) return had;
16
+ const made = {
17
+ shadings: buildShadingMap(file, resources),
18
+ alphas: buildAlphaMap(file, resources),
19
+ spaces: buildColorSpaceMap(file, resources)
20
+ };
21
+ stateCache.set(resources, made);
22
+ return made;
23
+ };
23
24
  const walk = (resources, content, baseCtm, depth, prefix) => {
24
25
  if (out.length >= MAX_VECTORS) return;
25
26
  const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
26
27
  const xobjDict = xobjects instanceof Map ? xobjects : void 0;
27
- const result = interpretContent(content, buildFonts(file, resources), baseCtm, shadings, alphas, spaces);
28
+ const maps = mapsOf(resources);
29
+ const result = interpretContent(content, buildFonts(file, resources), baseCtm, maps.shadings, maps.alphas, maps.spaces, hiddenProperties(file, resources));
28
30
  const events = [
29
31
  ...result.vectors.map((vector) => ({
30
32
  order: vector.order,
@@ -60,6 +62,7 @@ function paintedVectors(file, page, shadings, alphas, spaces) {
60
62
  if (depth >= MAX_FORM_DEPTH) continue;
61
63
  const stream = xobjDict ? file.resolve(xobjDict.get(placement.name) ?? PDF_NULL) : PDF_NULL;
62
64
  if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
65
+ if (hiddenXObject(file, stream)) continue;
63
66
  const subtype = file.get(stream.dict, "Subtype");
64
67
  if (!(subtype instanceof PdfName) || subtype.value !== "Form") continue;
65
68
  visiting.add(stream);
@@ -116,14 +119,22 @@ var MAX_FORM_DEPTH = 12;
116
119
  * clutter. Clips and the bare `sh` operator are not captured (a documented loss).
117
120
  */
118
121
  function collectPageVectors(file, page, occupied = []) {
119
- const [px0, py0, px1, py1] = page.mediaBox;
122
+ const [px0, py0, px1, py1] = page.cropBox;
120
123
  const pageArea = Math.max(1, Math.abs((px1 - px0) * (py1 - py0)));
121
- const shadings = buildShadingMap(file, page);
122
- const alphas = buildAlphaMap(file, page);
123
- const spaces = buildColorSpaceMap(file, page);
124
124
  const out = [];
125
125
  const painted = [...occupied];
126
- const raws = paintedVectors(file, page, shadings, alphas, spaces);
126
+ const raws = paintedVectors(file, page);
127
+ const losses = [];
128
+ const asked = /* @__PURE__ */ new Set();
129
+ for (const raw of raws) {
130
+ if (raw.masked === true) asked.add("PDF soft mask (/SMask in the graphics state) is not applied; the shape is drawn at full opacity throughout");
131
+ if (raw.blend !== void 0) asked.add(`PDF blend mode /${raw.blend} is not performed; the shape is drawn over what it was to blend with`);
132
+ }
133
+ for (const detail of asked) losses.push({
134
+ severity: "degraded",
135
+ feature: FEATURES.images,
136
+ detail
137
+ });
127
138
  for (const raw of raws) {
128
139
  if (out.length >= MAX_VECTORS) break;
129
140
  const v = clipped(raw);
@@ -139,7 +150,7 @@ function collectPageVectors(file, page, occupied = []) {
139
150
  const short = Math.min(w, h);
140
151
  const isBox = short >= MIN_SIDE && area >= MIN_AREA;
141
152
  const isRule = long >= MIN_RULE_LEN && short > 0;
142
- const filled = (v.gradient !== void 0 || solidFill) && (isBox || isRule) && area <= .85 * pageArea;
153
+ const filled = (v.gradient !== void 0 || solidFill) && (isBox || isRule);
143
154
  const stroked = v.strokeHex !== void 0 && (v.strokeHex !== "FFFFFF" || painted.some((box) => overlaps(box, b))) && Math.max(w, h) >= MIN_STROKE_LEN && area <= .85 * pageArea;
144
155
  if (!filled && !stroked) continue;
145
156
  painted.push(b);
@@ -148,6 +159,7 @@ function collectPageVectors(file, page, occupied = []) {
148
159
  segs: v.segs,
149
160
  ...filled ? v.gradient ? { gradient: v.gradient } : v.fillHex !== void 0 ? { fillHex: v.fillHex } : {} : {},
150
161
  ...filled && v.alpha !== void 0 ? { alpha: v.alpha } : {},
162
+ ...filled && v.darkens === true ? { darkens: true } : {},
151
163
  ...stroked ? {
152
164
  strokeHex: v.strokeHex,
153
165
  ...v.lineWidth !== void 0 ? { lineWidth: v.lineWidth } : {}
@@ -156,13 +168,14 @@ function collectPageVectors(file, page, occupied = []) {
156
168
  ...v.mcid !== void 0 ? { mcid: v.mcid } : {}
157
169
  });
158
170
  }
171
+ if (out.length >= MAX_VECTORS || raws.length >= MAX_VECTORS) losses.push({
172
+ severity: "dropped",
173
+ feature: FEATURES.shapes,
174
+ detail: `page carries more than ${String(MAX_VECTORS)} painted paths; the rest were not read`
175
+ });
159
176
  return {
160
177
  vectors: out,
161
- losses: out.length >= MAX_VECTORS || raws.length >= MAX_VECTORS ? [{
162
- severity: "dropped",
163
- feature: FEATURES.shapes,
164
- detail: `page carries more than ${String(MAX_VECTORS)} painted paths; the rest were not read`
165
- }] : []
178
+ losses
166
179
  };
167
180
  }
168
181
  /** Whether two boxes share any area at all. */