reamkit 1.29.0 → 1.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/README.md +60 -24
  2. package/dist/esm/core/converter/project.js +3 -1
  3. package/dist/esm/core/crypto/offcrypto.js +1 -1
  4. package/dist/esm/core/document-model/index.d.ts +1 -1
  5. package/dist/esm/core/document-model/types.d.ts +65 -0
  6. package/dist/esm/core/drawingml/shape-render.js +13 -1
  7. package/dist/esm/core/font/index.d.ts +2 -0
  8. package/dist/esm/core/font/ttf-build.d.ts +113 -0
  9. package/dist/esm/core/font/ttf-build.js +1224 -0
  10. package/dist/esm/core/font/ttf-subset.d.ts +19 -0
  11. package/dist/esm/core/font/ttf-subset.js +14 -1
  12. package/dist/esm/core/fonts/index.d.ts +1 -1
  13. package/dist/esm/core/fonts/provider.d.ts +17 -0
  14. package/dist/esm/core/fonts/provider.js +27 -2
  15. package/dist/esm/core/fonts/remote-fonts.d.ts +8 -0
  16. package/dist/esm/core/fonts/remote-fonts.js +108 -18
  17. package/dist/esm/core/ir/flow.d.ts +88 -1
  18. package/dist/esm/core/numbering/index.d.ts +1 -1
  19. package/dist/esm/core/numbering/state.d.ts +11 -1
  20. package/dist/esm/core/numbering/state.js +10 -1
  21. package/dist/esm/core/style-cascade/resolver.js +35 -4
  22. package/dist/esm/core/style-cascade/types.d.ts +19 -1
  23. package/dist/esm/core/style-cascade/types.js +3 -0
  24. package/dist/esm/index.d.ts +2 -2
  25. package/dist/esm/layout/page-doc.d.ts +10 -3
  26. package/dist/esm/layout/styled-layout.d.ts +48 -7
  27. package/dist/esm/layout/styled-layout.js +881 -119
  28. package/dist/esm/layout/turned-section.d.ts +38 -0
  29. package/dist/esm/layout/turned-section.js +193 -0
  30. package/dist/esm/pdf/styled-page-emitter.js +78 -2
  31. package/dist/esm/pdf-reader/annot-draw.js +93 -1
  32. package/dist/esm/pdf-reader/annots.d.ts +18 -0
  33. package/dist/esm/pdf-reader/annots.js +86 -9
  34. package/dist/esm/pdf-reader/cff-outline.d.ts +27 -0
  35. package/dist/esm/pdf-reader/cff-outline.js +169 -21
  36. package/dist/esm/pdf-reader/cmap.js +5 -2
  37. package/dist/esm/pdf-reader/content.d.ts +70 -4
  38. package/dist/esm/pdf-reader/content.js +172 -12
  39. package/dist/esm/pdf-reader/display.d.ts +36 -0
  40. package/dist/esm/pdf-reader/display.js +82 -1
  41. package/dist/esm/pdf-reader/document.js +5 -1
  42. package/dist/esm/pdf-reader/embedded-fonts.d.ts +25 -0
  43. package/dist/esm/pdf-reader/embedded-fonts.js +78 -9
  44. package/dist/esm/pdf-reader/encodings.d.ts +8 -0
  45. package/dist/esm/pdf-reader/encodings.js +25 -3
  46. package/dist/esm/pdf-reader/face-outlines.d.ts +78 -0
  47. package/dist/esm/pdf-reader/face-outlines.js +362 -0
  48. package/dist/esm/pdf-reader/figures.d.ts +52 -0
  49. package/dist/esm/pdf-reader/figures.js +433 -0
  50. package/dist/esm/pdf-reader/flow-build.d.ts +224 -7
  51. package/dist/esm/pdf-reader/flow-build.js +545 -39
  52. package/dist/esm/pdf-reader/font.d.ts +27 -1
  53. package/dist/esm/pdf-reader/font.js +632 -63
  54. package/dist/esm/pdf-reader/glyf-outline.d.ts +33 -0
  55. package/dist/esm/pdf-reader/glyf-outline.js +148 -3
  56. package/dist/esm/pdf-reader/glyph-names.js +154 -1
  57. package/dist/esm/pdf-reader/glyph-shapes.d.ts +18 -0
  58. package/dist/esm/pdf-reader/glyph-shapes.js +57 -0
  59. package/dist/esm/pdf-reader/image-decode.js +70 -4
  60. package/dist/esm/pdf-reader/images.d.ts +5 -0
  61. package/dist/esm/pdf-reader/images.js +4 -2
  62. package/dist/esm/pdf-reader/jbig2.d.ts +40 -1
  63. package/dist/esm/pdf-reader/jbig2.js +78 -16
  64. package/dist/esm/pdf-reader/jpeg.d.ts +6 -3
  65. package/dist/esm/pdf-reader/jpeg.js +21 -1
  66. package/dist/esm/pdf-reader/layout.d.ts +119 -2
  67. package/dist/esm/pdf-reader/layout.js +2219 -184
  68. package/dist/esm/pdf-reader/lexer.d.ts +10 -0
  69. package/dist/esm/pdf-reader/lexer.js +17 -0
  70. package/dist/esm/pdf-reader/page-numbers.d.ts +53 -0
  71. package/dist/esm/pdf-reader/page-numbers.js +167 -0
  72. package/dist/esm/pdf-reader/pattern-tint.d.ts +11 -1
  73. package/dist/esm/pdf-reader/pattern-tint.js +21 -3
  74. package/dist/esm/pdf-reader/regions.d.ts +25 -0
  75. package/dist/esm/pdf-reader/regions.js +167 -0
  76. package/dist/esm/pdf-reader/shading.d.ts +58 -2
  77. package/dist/esm/pdf-reader/shading.js +181 -11
  78. package/dist/esm/pdf-reader/struct-tree.js +112 -8
  79. package/dist/esm/pdf-reader/tagged.js +207 -28
  80. package/dist/esm/pdf-reader/text-rules.js +1 -1
  81. package/dist/esm/pdf-reader/text.d.ts +6 -3
  82. package/dist/esm/pdf-reader/text.js +106 -22
  83. package/dist/esm/pdf-reader/type1-outline.d.ts +11 -0
  84. package/dist/esm/pdf-reader/type1-outline.js +63 -8
  85. package/dist/esm/pdf-reader/vector.d.ts +8 -2
  86. package/dist/esm/pdf-reader/vector.js +105 -6
  87. package/dist/esm/word/doc/doc-reader.js +6 -2
  88. package/dist/esm/word/doc/doc-text.d.ts +6 -0
  89. package/dist/esm/word/doc/doc-text.js +19 -1
  90. package/dist/esm/word/document-parser.d.ts +2 -2
  91. package/dist/esm/word/document-parser.js +11 -1
  92. package/dist/esm/word/docx-reader.js +5 -3
  93. package/dist/esm/word/docx-writer.js +341 -54
  94. package/dist/esm/word/drawing-parser.d.ts +5 -3
  95. package/dist/esm/word/drawing-parser.js +56 -10
  96. package/dist/esm/word/font-embed.d.ts +30 -0
  97. package/dist/esm/word/font-embed.js +173 -0
  98. package/dist/esm/word/font-table.d.ts +10 -0
  99. package/dist/esm/word/font-table.js +13 -1
  100. package/dist/esm/word/index.js +1 -1
  101. package/dist/esm/word/numbering-parser.d.ts +3 -1
  102. package/dist/esm/word/numbering-parser.js +2 -1
  103. package/dist/esm/word/paragraph-properties.d.ts +7 -6
  104. package/dist/esm/word/paragraph-properties.js +16 -2
  105. package/dist/esm/word/run-properties.js +26 -0
  106. package/package.json +12 -5
@@ -2,13 +2,15 @@ import { pt } from "../core/ir/units.js";
2
2
  import { ResourceStore } from "../core/ir/resources.js";
3
3
  import { FEATURES } from "../core/ir/features.js";
4
4
  import { collectEmbeddedFonts } from "./embedded-fonts.js";
5
+ import { collectFaceFamilies } from "./font.js";
6
+ import { displayOf, placeRuns, placeVectors, textFrameOf, wordsTurnOf } from "./display.js";
7
+ import { ASCENDER, CARRIER_LINE_PT, buildFlowDoc, dedupeLosses, floatOntoSheet, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, sectionOnSheet, shapeBlock, spaceAfter, withMeasuredMargins } from "./flow-build.js";
8
+ import { faceOutlinesOf, kernedFaces, pageSpacing } from "./face-outlines.js";
5
9
  import { extractPageText } from "./text.js";
6
10
  import { collectPageVectors } from "./vector.js";
7
- import { displayOf, placeRuns, placeVectors } from "./display.js";
8
- import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock, withMeasuredMargins } from "./flow-build.js";
9
11
  import { collectPageImages } from "./images.js";
10
12
  import { markDrawnRules } from "./text-rules.js";
11
- import { endedParagraph } from "./layout.js";
13
+ import { FOOTER_PART, HEADER_PART, endedParagraph, footerBand, numberedFrom, numberingOf, pageTextEdges, runningFoot, stepsBetweenWords } from "./layout.js";
12
14
  import { readStructTree } from "./struct-tree.js";
13
15
  //#region src/pdf-reader/tagged.ts
14
16
  var ASSUMED_CONTENT_WIDTH_PT = 468;
@@ -30,8 +32,15 @@ function reconstructTaggedPdf(file) {
30
32
  const root = readStructTree(file);
31
33
  if (!root) return void 0;
32
34
  const pages = file.pages();
33
- const shown = pages.map((page) => displayOf(page));
34
- const placedRuns = pages.map((page, i) => placeRuns(extractPageText(file, page), shown[i]));
35
+ const sheets = pages.map((page) => displayOf(page));
36
+ const painted = /* @__PURE__ */ new Map();
37
+ const extracted = pages.map((page) => extractPageText(file, page, painted));
38
+ const spacing = pageSpacing(extracted);
39
+ const onSheets = extracted.map((runs, i) => placeRuns(runs, sheets[i]));
40
+ const worded = onSheets.filter((runs) => runs.some((r) => r.text.trim() !== ""));
41
+ const turned = worded.length > 0 && worded.every((runs) => wordsTurnOf(runs) === 270);
42
+ const shown = turned ? sheets.map((sheet) => textFrameOf(sheet)) : sheets;
43
+ const placedRuns = turned ? extracted.map((runs, i) => placeRuns(runs, shown[i])) : onSheets;
35
44
  const vectorLosses = [];
36
45
  const pageVectors = pages.map((page, i) => {
37
46
  const lifted = collectPageVectors(file, page, collectPageImages(file, page).images.map((img) => ({
@@ -78,10 +87,14 @@ function reconstructTaggedPdf(file) {
78
87
  const textOf = (node) => squash(node.mcids.map(({ page, mcid }) => runsOfMcid(page, mcid).map((r) => r.text).join("")).join(" "));
79
88
  const spansOf = (node) => {
80
89
  const spans = [];
81
- node.mcids.forEach(({ page, mcid }, i) => {
82
- if (i > 0) spans.push({ text: " " });
83
- for (const run of runsOfMcid(page, mcid)) spans.push(spanOf(run));
84
- });
90
+ let last;
91
+ for (const { page, mcid } of node.mcids) {
92
+ const runs = runsOfMcid(page, mcid);
93
+ const first = runs[0];
94
+ if (last !== void 0 && first !== void 0 && spacedApart(last, first)) spans.push(spaceAfter(spans[spans.length - 1]));
95
+ for (const run of runs) spans.push(spanOf(run));
96
+ last = runs[runs.length - 1] ?? last;
97
+ }
85
98
  return spans;
86
99
  };
87
100
  /** One run as the span that carries everything the page showed it with. */
@@ -101,6 +114,23 @@ function reconstructTaggedPdf(file) {
101
114
  });
102
115
  const collectText = (node) => squash([textOf(node), ...node.children.map(collectText)].join(" "));
103
116
  const setting = /* @__PURE__ */ new Map();
117
+ const pageOf = /* @__PURE__ */ new Map();
118
+ /** The highest baseline a page's own paragraphs stand on, page space (y up). */
119
+ const highestOf = (own) => {
120
+ const tops = own.flatMap((el) => {
121
+ const set = setting.get(el);
122
+ return set ? [set.top] : [];
123
+ });
124
+ return tops.length > 0 ? Math.max(...tops) : void 0;
125
+ };
126
+ /** The lowest baseline a page's own paragraphs reach, page space (y up). */
127
+ const lowestOf = (own) => {
128
+ const bottoms = own.flatMap((el) => {
129
+ const set = setting.get(el);
130
+ return set ? [set.bottom] : [];
131
+ });
132
+ return bottoms.length > 0 ? Math.min(...bottoms) : void 0;
133
+ };
104
134
  /** The topmost and bottommost baseline under a node, and its largest face. */
105
135
  function baselinesOf(node) {
106
136
  let top;
@@ -156,8 +186,10 @@ function reconstructTaggedPdf(file) {
156
186
  groups[groups.length - 1].push(line);
157
187
  prev = line;
158
188
  }
189
+ const ends = (spans) => /\s$/u.test(spans.at(-1)?.text ?? "");
190
+ const opens = (spans) => /^\s/u.test(spans[0]?.text ?? "");
159
191
  return groups.map((g) => ({
160
- spans: g.flatMap((l, i) => i > 0 ? [{ text: " " }, ...l.spans] : [...l.spans]),
192
+ spans: g.flatMap((l, i) => i > 0 && !ends(g[i - 1].spans) && !opens(l.spans) ? [spaceAfter(g[i - 1].spans.at(-1)), ...l.spans] : [...l.spans]),
161
193
  set: {
162
194
  top: g[0].y,
163
195
  bottom: g[g.length - 1].y,
@@ -198,33 +230,48 @@ function reconstructTaggedPdf(file) {
198
230
  }).filter((l) => l.spans.some((sp) => sp.text.trim().length > 0));
199
231
  }
200
232
  function emit(node, out) {
233
+ const on = (el) => {
234
+ const page = firstPageOf(node);
235
+ if (page !== void 0) pageOf.set(el, page);
236
+ return el;
237
+ };
201
238
  if (node.type === "Table") {
202
239
  const table = buildTable(node);
203
- if (table) out.push(table);
240
+ if (table) out.push(on(table));
204
241
  return;
205
242
  }
206
243
  if (node.type === "Figure") {
207
244
  for (const img of imagesForNode(node)) {
208
245
  emitted.add(img);
209
- out.push(imageBlock(img, resources, node.alt));
246
+ out.push(on(imageBlock(img, resources, node.alt)));
210
247
  }
211
248
  return;
212
249
  }
213
250
  if (node.type === "LI") {
214
251
  const text = collectText(node);
215
- if (text.length > 0) out.push(paragraphBlock(text, void 0));
252
+ if (text.length > 0) out.push(on(paragraphBlock(text, void 0)));
216
253
  return;
217
254
  }
218
255
  if (node.children.length === 0) {
219
256
  if (textOf(node).length > 0) for (const part of settingsOf(node)) {
220
257
  const el = paragraphFromRuns(part.spans, headingLevel(node.type));
221
258
  setting.set(el, part.set);
222
- out.push(el);
259
+ out.push(on(el));
223
260
  }
224
261
  return;
225
262
  }
226
263
  for (const child of node.children) emit(child, out);
227
264
  }
265
+ /** The first page any marked content under a node stands on. */
266
+ function firstPageOf(node) {
267
+ let first;
268
+ const visit = (n) => {
269
+ for (const { page } of n.mcids) if (first === void 0 || page < first) first = page;
270
+ for (const child of n.children) visit(child);
271
+ };
272
+ visit(node);
273
+ return first;
274
+ }
228
275
  function buildTable(tableNode) {
229
276
  const raw = [];
230
277
  const collectRows = (n) => {
@@ -288,9 +335,14 @@ function reconstructTaggedPdf(file) {
288
335
  cells
289
336
  };
290
337
  }
291
- const body = [];
292
- emit(root, body);
293
- spaceParagraphs(body, setting, shown[0]?.height ?? 0);
338
+ const named = [];
339
+ emit(root, named);
340
+ const byPage = pages.map(() => []);
341
+ let current = 0;
342
+ for (const el of named) {
343
+ current = Math.max(current, pageOf.get(el) ?? current);
344
+ byPage[current].push(el);
345
+ }
294
346
  let zOrder = -1e6;
295
347
  imageLosses.push(...vectorLosses);
296
348
  pages.forEach((_page, index) => {
@@ -299,31 +351,104 @@ function reconstructTaggedPdf(file) {
299
351
  top: shown[index].height
300
352
  };
301
353
  const taken = ruled[index]?.consumed;
354
+ const drawn = [];
302
355
  for (const v of pageVectors[index] ?? []) {
303
356
  if (taken?.has(v) === true) continue;
304
- body.push(shapeBlock(v, frame, zOrder++, true));
357
+ const shape = shapeBlock(v, frame, zOrder++, true);
358
+ drawn.push(turned ? floatOntoSheet(shape, shown[index].height) : shape);
305
359
  }
360
+ byPage[index].unshift(...drawn);
306
361
  });
307
- const orphans = [];
308
362
  pageImages.forEach((p, page) => {
309
- for (const img of p.images) if (!emitted.has(img)) orphans.push({
310
- page,
311
- img
312
- });
363
+ const left = p.images.filter((img) => !emitted.has(img)).sort((a, b) => b.y - a.y);
364
+ byPage[page].push(...left.map((img) => imageBlock(img, resources)));
365
+ });
366
+ if (byPage.every((own) => own.length === 0)) return void 0;
367
+ const loose = placedRuns.map((runs, i) => shown[i].sheet ? [] : runs.filter((r) => !claimed.has(r)));
368
+ const foot = runningFoot(loose, shown, "foot");
369
+ const head = runningFoot(loose, shown, "head");
370
+ const numberedBand = foot?.numbered === true ? foot : head?.numbered === true ? head : void 0;
371
+ const numbering = numberingOf(numberedBand);
372
+ const lifted = (r, i) => foot?.lift[i]?.has(r) === true || head?.lift[i]?.has(r) === true;
373
+ const measured = withMeasuredMargins(sectionFromPdfPages(pages, shown[0]), shown, placedRuns.map((runs, i) => runs.filter((r) => !lifted(r, i))), pageImages.map((p) => p.images), foot?.band);
374
+ const body = [];
375
+ const top = measured?.margins?.top ?? 0;
376
+ const bottom = measured?.margins?.bottom ?? 0;
377
+ const ends = byPage.map((own) => lowestOf(own));
378
+ const tops = byPage.map((own) => highestOf(own));
379
+ const sectionsFrom = (numbering?.runs ?? []).map((run) => run.from).filter((from) => from > 0);
380
+ const endsAt = [];
381
+ let sectionFrom = 0;
382
+ byPage.forEach((own, index) => {
383
+ if (sectionsFrom.includes(index)) {
384
+ endsAt.push({
385
+ at: body.length,
386
+ from: sectionFrom
387
+ });
388
+ sectionFrom = index;
389
+ }
390
+ const height = shown[index]?.height ?? 0;
391
+ spaceParagraphs(own, setting, height);
392
+ const ended = ends[index - 1];
393
+ const opens = index > 0 && (own.length === 0 || ended === void 0 || ended - bottom > (height - top - bottom) * SHORT_PAGE_SHARE);
394
+ const lead = own.findIndex((el) => !floats(el));
395
+ const set = lead >= 0 ? setting.get(own[lead]) : void 0;
396
+ const fresh = index === 0 || opens || sectionsFrom.includes(index);
397
+ const highest = tops[index];
398
+ const first = set !== void 0 && highest !== void 0 && set.top >= highest - set.size;
399
+ if (fresh && first && measured?.margins) {
400
+ const gap = height - top - (set.top + set.size * ASCENDER);
401
+ if (gap > 1) own[lead] = spacedBefore(own[lead], gap);
402
+ }
403
+ if (opens && !sectionsFrom.includes(index)) body.push(pageBreak(true));
404
+ else if (index === 0 && own.length === 0 && pages.length > 1) body.push(pageBreak(false));
405
+ body.push(...own);
313
406
  });
314
- orphans.sort((a, b) => a.page - b.page || b.img.y - a.img.y);
315
- for (const { img } of orphans) body.push(imageBlock(img, resources));
316
- if (body.length === 0) return void 0;
317
407
  if (placedRuns.some((page) => page.some((r) => r.text.includes("�")))) imageLosses.push({
318
408
  severity: "dropped",
319
409
  feature: FEATURES.text,
320
410
  detail: "some glyphs map to no character — the font states no /ToUnicode and its program says nothing either, so that text is unrecoverable"
321
411
  });
322
- const onPage = placedRuns.flat().reduce((n, r) => n + r.text.length, 0);
412
+ const onPage = placedRuns.flatMap((runs, i) => runs.filter((r) => !lifted(r, i))).reduce((n, r) => n + r.text.length, 0);
323
413
  const reached = [...claimed].reduce((n, r) => n + r.text.length, 0);
324
414
  if (onPage > 0 && reached * 2 < onPage) return void 0;
415
+ endsAt.push({
416
+ at: body.length,
417
+ from: sectionFrom
418
+ });
419
+ const stepped0 = stepsBetweenWords(placedRuns[0] ?? []);
420
+ const edges0 = pageTextEdges(placedRuns[0] ?? []);
421
+ const numeralOf = (of) => {
422
+ if (of === void 0 || of !== numberedBand || numbering === void 0) return void 0;
423
+ const first = of.lift.findIndex((set) => set.size > 0);
424
+ return first >= 0 ? numbering.numbers[first]?.text : void 0;
425
+ };
426
+ const footBand = foot ? footerBand(foot.band, stepped0, edges0, foot.numbered, numeralOf(foot)) : [];
427
+ const headBand = head ? footerBand(head.band, stepped0, edges0, head.numbered, numeralOf(head)) : [];
428
+ const sectionAt = (from) => {
429
+ const base = sectionOnSheet(measured, shown[0]);
430
+ return base ? {
431
+ ...base,
432
+ ...numberedFrom(numbering, from),
433
+ ...footBand.length > 0 ? { footers: [{
434
+ type: "default",
435
+ relationshipId: FOOTER_PART
436
+ }] } : {},
437
+ ...headBand.length > 0 ? { headers: [{
438
+ type: "default",
439
+ relationshipId: HEADER_PART
440
+ }] } : {}
441
+ } : base;
442
+ };
443
+ const sections = endsAt.length > 1 ? endsAt.flatMap((end) => {
444
+ const properties = sectionAt(end.from);
445
+ return properties ? [{
446
+ properties,
447
+ endIndex: end.at
448
+ }] : [];
449
+ }) : [];
325
450
  return {
326
- doc: buildFlowDoc(body, resources, withMeasuredMargins(sectionFromPdfPages(pages), shown, placedRuns, pageImages.map((p) => p.images)), collectEmbeddedFonts(file, pages, imageLosses)),
451
+ doc: buildFlowDoc(body, resources, sectionAt(0), collectEmbeddedFonts(file, pages, imageLosses), sections, footBand.length > 0 || headBand.length > 0 ? new Map([...footBand.length > 0 ? [[FOOTER_PART, footBand]] : [], ...headBand.length > 0 ? [[HEADER_PART, headBand]] : []]) : void 0, collectFaceFamilies(file, pages), faceOutlinesOf(painted, spacing), kernedFaces(spacing)),
327
452
  losses: imageLosses
328
453
  };
329
454
  }
@@ -414,9 +539,63 @@ function equalGrid(raw) {
414
539
  grid: Array.from({ length: numCols }, () => w)
415
540
  };
416
541
  }
542
+ /**
543
+ * Whether the page shows a space between one run and the next: the second
544
+ * starts another line, or stands clear of the first by more than a kern —
545
+ * and neither already carries the space.
546
+ */
547
+ function spacedApart(prev, next) {
548
+ if (/\s$/u.test(prev.text) || /^\s/u.test(next.text)) return false;
549
+ const size = Math.max(prev.fontSizePt, next.fontSizePt, 1);
550
+ if (Math.abs(prev.y - next.y) > size * .5) return true;
551
+ return next.x - prev.endX > size * MCID_SPACE_EM;
552
+ }
553
+ /** A gap between two stretches of marked content this wide, in ems, is a word space. */
554
+ var MCID_SPACE_EM = .15;
417
555
  function squash(text) {
418
556
  return text.replace(/\s+/g, " ").trim();
419
557
  }
558
+ /**
559
+ * How much of a page's text area the page may leave empty below its last line
560
+ * and still be FULL: a page that ends higher than this ended on purpose.
561
+ */
562
+ var SHORT_PAGE_SHARE = .25;
563
+ /** Whether an element FLOATS — a drawing anchored to its page, which takes no room. */
564
+ function floats(el) {
565
+ return el.kind === "image" && el.image.float !== void 0 || el.kind === "shape" && el.shape.float !== void 0;
566
+ }
567
+ /** A paragraph given the space the page left above it; anything else as it is. */
568
+ function spacedBefore(el, before) {
569
+ if (el.kind !== "paragraph") return el;
570
+ const properties = {
571
+ ...el.paragraph.properties,
572
+ spacingBefore: pt(before)
573
+ };
574
+ return {
575
+ ...el,
576
+ paragraph: {
577
+ ...el.paragraph,
578
+ properties
579
+ }
580
+ };
581
+ }
582
+ /**
583
+ * An empty paragraph that takes no room: the carrier of a page break, or of
584
+ * nothing at all on a blank first sheet the next page's break has to follow.
585
+ */
586
+ function pageBreak(breaks) {
587
+ return {
588
+ kind: "paragraph",
589
+ paragraph: {
590
+ properties: {
591
+ ...breaks ? { pageBreakBefore: true } : {},
592
+ spacingLine: CARRIER_LINE_PT,
593
+ spacingLineRule: "exact"
594
+ },
595
+ runs: []
596
+ }
597
+ };
598
+ }
420
599
  function headingLevel(type) {
421
600
  const m = /^H([1-6])$/.exec(type);
422
601
  return m ? Number(m[1]) - 1 : void 0;
@@ -95,7 +95,7 @@ var SAME_HEIGHT_PT = .5;
95
95
  * separate table rules is not.
96
96
  */
97
97
  function joinRules(vectors) {
98
- const rules = vectors.filter((v) => isRule(v));
98
+ const rules = vectors.filter((v) => v.glyph !== true && isRule(v));
99
99
  const byRow = /* @__PURE__ */ new Map();
100
100
  for (const v of rules) {
101
101
  const mid = (v.minY + v.maxY) / 2;
@@ -1,4 +1,5 @@
1
1
  import { ContentFont, TextRun } from './content.js';
2
+ import { ShownCodes } from './face-outlines.js';
2
3
  import { PdfDict } from '../pdf/objects.js';
3
4
  import { PdfFile, PdfPage } from './document.js';
4
5
  /**
@@ -8,11 +9,13 @@ import { PdfFile, PdfPage } from './document.js';
8
9
  * origin falls inside a `/Link` annotation's `/Rect` with that link's URI (EP8)
9
10
  * so hyperlinks survive.
10
11
  *
11
- * @param file The owning {@link PdfFile}.
12
- * @param page The page to extract.
12
+ * @param file The owning {@link PdfFile}.
13
+ * @param page The page to extract.
14
+ * @param painted Where to gather the codes each font painted, for a caller
15
+ * that embeds the faces (see `./face-outlines`).
13
16
  * @returns The page's runs, each carrying an `href` when it sits under a link.
14
17
  */
15
- export declare function extractPageText(file: PdfFile, page: PdfPage): Array<TextRun>;
18
+ export declare function extractPageText(file: PdfFile, page: PdfPage, painted?: ShownCodes): Array<TextRun>;
16
19
  /**
17
20
  * The `/Font` resources of one dictionary, built into interpreter fonts. Shared
18
21
  * with the path and picture walks, which need them for one thing only: a Type 3
@@ -2,10 +2,11 @@ import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
2
2
  import { buildColorSpaceMap, buildShadingMap } from "./shading.js";
3
3
  import { IDENTITY, interpretContent, multiply } from "./content.js";
4
4
  import { textMarkupOf } from "./annot-draw.js";
5
- import { collectPageAppearances } from "./annots.js";
6
- import { hiddenProperties, hiddenXObject } from "./optional-content.js";
5
+ import { appearanceContent, collectPageAppearances } from "./annots.js";
7
6
  import { buildContentFont } from "./font.js";
8
- import { patternTint } from "./pattern-tint.js";
7
+ import { hiddenProperties, hiddenXObject } from "./optional-content.js";
8
+ import { addShown } from "./face-outlines.js";
9
+ import { patternTint, tintedHex } from "./pattern-tint.js";
9
10
  //#region src/pdf-reader/text.ts
10
11
  var MAX_FORM_DEPTH = 8;
11
12
  /**
@@ -15,17 +16,24 @@ var MAX_FORM_DEPTH = 8;
15
16
  * origin falls inside a `/Link` annotation's `/Rect` with that link's URI (EP8)
16
17
  * so hyperlinks survive.
17
18
  *
18
- * @param file The owning {@link PdfFile}.
19
- * @param page The page to extract.
19
+ * @param file The owning {@link PdfFile}.
20
+ * @param page The page to extract.
21
+ * @param painted Where to gather the codes each font painted, for a caller
22
+ * that embeds the faces (see `./face-outlines`).
20
23
  * @returns The page's runs, each carrying an `href` when it sits under a link.
21
24
  */
22
- function extractPageText(file, page) {
25
+ function extractPageText(file, page, painted) {
23
26
  const runs = [];
24
- collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs);
25
- for (const appearance of collectPageAppearances(file, page)) collectRuns(file, appearance.resources ?? page.resources, file.streamData(appearance.stream), appearance.ctm, 1, new Set([appearance.stream]), runs);
27
+ collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs, painted);
28
+ const own = runs.length;
29
+ for (const appearance of collectPageAppearances(file, page)) collectRuns(file, appearance.resources ?? page.resources, appearanceContent(file, appearance), appearance.ctm, 1, new Set([appearance.stream]), runs, painted);
30
+ for (let i = own; i < runs.length; i++) runs[i] = {
31
+ ...runs[i],
32
+ annotation: true
33
+ };
26
34
  const links = collectLinks(file, page);
27
35
  const marks = collectTextMarkup(file, page);
28
- const shown = withoutRestrikes(runs);
36
+ const shown = withAccentsComposed(withoutRestrikes(runs));
29
37
  if (links.length === 0 && marks.length === 0) return shown;
30
38
  return shown.flatMap((run) => {
31
39
  const link = links.find((l) => inRect(run.x, run.y, l.rect));
@@ -163,6 +171,91 @@ function withoutRestrikes(runs) {
163
171
  }
164
172
  /** How near, in ems, a re-strike of the same text lands to the one it thickens. */
165
173
  var RESTRIKE_EM = .08;
174
+ /**
175
+ * §9.4.3 — an accent struck over a letter, composed with it.
176
+ *
177
+ * TeX sets an accented letter its font has no glyph for as two: the accent,
178
+ * and the letter drawn back under it — "ï" is a dieresis and a dotless i.
179
+ * Read as they come, comments.pdf's "naïve" came back "na¨ıve", the accent a
180
+ * character of the word. A run that is one spacing accent, with the pen taken
181
+ * back under it for the run after it, is the accent of that run's first letter
182
+ * — or, where the accent was struck after its letter, of the last letter of the
183
+ * run before it. The two are written as the one character Unicode composes them
184
+ * into, a dotless i or j taking back its dot's place under the accent.
185
+ *
186
+ * @param runs The page's runs, in painting order.
187
+ * @returns The runs with each such accent composed into its letter.
188
+ */
189
+ function withAccentsComposed(runs) {
190
+ const out = [];
191
+ for (let i = 0; i < runs.length; i++) {
192
+ const run = runs[i];
193
+ const mark = ACCENTS.get(run.text);
194
+ const next = runs[i + 1];
195
+ const prev = out[out.length - 1];
196
+ if (mark !== void 0 && next !== void 0 && struckOver(run, next, "first")) {
197
+ const [first = "", ...rest] = [...next.text];
198
+ out.push({
199
+ ...next,
200
+ text: accented(first, mark) + rest.join("")
201
+ });
202
+ i++;
203
+ continue;
204
+ }
205
+ if (mark !== void 0 && prev !== void 0 && struckOver(run, prev, "last")) {
206
+ const chars = [...prev.text];
207
+ const last = chars.pop() ?? "";
208
+ out[out.length - 1] = {
209
+ ...prev,
210
+ text: chars.join("") + accented(last, mark)
211
+ };
212
+ continue;
213
+ }
214
+ out.push(run);
215
+ }
216
+ return out;
217
+ }
218
+ /**
219
+ * Whether an accent stands over the first letter of a run (the pen taken back
220
+ * under it) or over the last (the pen taken back to strike it): on one
221
+ * baseline, give or take the lift an accent over a capital takes, and
222
+ * overlapping it by more than a hair.
223
+ */
224
+ function struckOver(accent, base, which) {
225
+ if (accent.angleDeg !== void 0 || base.angleDeg !== void 0) return false;
226
+ const size = base.fontSizePt || accent.fontSizePt || 10;
227
+ if (Math.abs(accent.y - base.y) > size * ACCENT_LIFT_EM) return false;
228
+ if (!/^\p{L}/u.test(which === "first" ? base.text : [...base.text].at(-1) ?? "")) return false;
229
+ const hair = size * ACCENT_OVERLAP_EM;
230
+ return which === "first" ? base.x < accent.endX - hair && base.x > accent.x - size : accent.x < base.endX - hair && accent.x > base.x;
231
+ }
232
+ /** A letter with an accent composed onto it: a dotless i or j takes back its dot's place. */
233
+ function accented(letter, mark) {
234
+ return ((letter === "ı" ? "i" : letter === "ȷ" ? "j" : letter) + mark).normalize("NFC");
235
+ }
236
+ /** How far over the line, in ems, TeX lifts an accent to stand over a capital. */
237
+ var ACCENT_LIFT_EM = .6;
238
+ /** How far, in ems, an accent must overlap its letter to stand over it. */
239
+ var ACCENT_OVERLAP_EM = .05;
240
+ /** The spacing accents TeX strikes over a letter, and the combining marks they are. */
241
+ var ACCENTS = new Map([
242
+ ["`", "̀"],
243
+ ["´", "́"],
244
+ ["ˆ", "̂"],
245
+ ["^", "̂"],
246
+ ["˜", "̃"],
247
+ ["~", "̃"],
248
+ ["¯", "̄"],
249
+ ["ˉ", "̄"],
250
+ ["˘", "̆"],
251
+ ["˙", "̇"],
252
+ ["¨", "̈"],
253
+ ["˚", "̊"],
254
+ ["˝", "̋"],
255
+ ["ˇ", "̌"],
256
+ ["¸", "̧"],
257
+ ["˛", "̨"]
258
+ ]);
166
259
  var spaceCache = /* @__PURE__ */ new WeakMap();
167
260
  /** The page's shading patterns, read once per resource dictionary. */
168
261
  var shadingCache = /* @__PURE__ */ new WeakMap();
@@ -182,14 +275,15 @@ function spacesOf(file, resources) {
182
275
  spaceCache.set(resources, made);
183
276
  return made;
184
277
  }
185
- function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
278
+ function collectRuns(file, resources, content, baseCtm, depth, visiting, out, shown) {
186
279
  const result = interpretContent(content, buildFonts(file, resources), baseCtm, shadingsOf(file, resources), void 0, spacesOf(file, resources), hiddenProperties(file, resources));
187
280
  out.push(...result.texts.map((r) => withPatternColour(file, resources, r, visiting)));
281
+ if (shown) addShown(shown, result.shown);
188
282
  if (depth >= MAX_FORM_DEPTH) return;
189
283
  for (const glyph of result.glyphs) {
190
284
  if (visiting.has(glyph.stream)) continue;
191
285
  visiting.add(glyph.stream);
192
- collectRuns(file, glyph.resources ?? resources, file.streamData(glyph.stream), glyph.ctm, depth + 1, visiting, out);
286
+ collectRuns(file, glyph.resources ?? resources, file.streamData(glyph.stream), glyph.ctm, depth + 1, visiting, out, shown);
193
287
  visiting.delete(glyph.stream);
194
288
  }
195
289
  if (!resources) return;
@@ -203,7 +297,7 @@ function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
203
297
  if (!(sub instanceof PdfName) || sub.value !== "Form") continue;
204
298
  visiting.add(stream);
205
299
  const formRes = file.get(stream.dict, "Resources");
206
- collectRuns(file, formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, visiting, out);
300
+ collectRuns(file, formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, visiting, out, shown);
207
301
  visiting.delete(stream);
208
302
  }
209
303
  }
@@ -244,16 +338,6 @@ function withPatternColour(file, resources, run, visiting) {
244
338
  visiting.delete(stream);
245
339
  }
246
340
  }
247
- /** A colour laid over white paper at `coverage` strength, as a 6-hex string. */
248
- function tintedHex(colorHex, coverage) {
249
- const k = Math.min(1, Math.max(0, coverage));
250
- if (k >= 1) return colorHex;
251
- const channel = (at) => {
252
- const c = Number.parseInt(colorHex.slice(at, at + 2), 16);
253
- return Math.round(255 - (255 - (Number.isFinite(c) ? c : 0)) * k).toString(16).toUpperCase().padStart(2, "0");
254
- };
255
- return `${channel(0)}${channel(2)}${channel(4)}`;
256
- }
257
341
  /**
258
342
  * The `/Font` resources of one dictionary, built into interpreter fonts. Shared
259
343
  * with the path and picture walks, which need them for one thing only: a Type 3
@@ -10,6 +10,17 @@ export interface Type1Font {
10
10
  readonly has: (name: string) => boolean;
11
11
  /** The program's own `/Encoding`: code → glyph name, where it states one. */
12
12
  readonly encoding?: ReadonlyMap<number, string>;
13
+ /**
14
+ * The embedding the program's licence allows — the OS/2 `fsType` Adobe
15
+ * writes into `FontInfo` as `/FSType`, where the program states one.
16
+ */
17
+ readonly fsType?: number;
18
+ /**
19
+ * §6.4 — how far the pen moves after a glyph, in thousandths of an em: the
20
+ * width its `hsbw` (or `sbw`) states. `undefined` where the program holds no
21
+ * such glyph or its charstring states no width first.
22
+ */
23
+ readonly advance: (name: string) => number | undefined;
13
24
  }
14
25
  /**
15
26
  * Read an embedded Type 1 program.