reamkit 1.30.0 → 1.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/README.md +60 -24
  2. package/dist/esm/core/converter/project.js +3 -1
  3. package/dist/esm/core/crypto/offcrypto.js +1 -1
  4. package/dist/esm/core/document-model/index.d.ts +1 -1
  5. package/dist/esm/core/document-model/types.d.ts +63 -0
  6. package/dist/esm/core/font/index.d.ts +2 -0
  7. package/dist/esm/core/font/ttf-build.d.ts +113 -0
  8. package/dist/esm/core/font/ttf-build.js +1224 -0
  9. package/dist/esm/core/font/ttf-subset.d.ts +19 -0
  10. package/dist/esm/core/font/ttf-subset.js +14 -1
  11. package/dist/esm/core/fonts/provider.d.ts +17 -0
  12. package/dist/esm/core/fonts/provider.js +27 -2
  13. package/dist/esm/core/fonts/remote-fonts.js +8 -1
  14. package/dist/esm/core/ir/flow.d.ts +71 -1
  15. package/dist/esm/core/numbering/index.d.ts +1 -1
  16. package/dist/esm/core/numbering/state.d.ts +11 -1
  17. package/dist/esm/core/numbering/state.js +10 -1
  18. package/dist/esm/core/style-cascade/resolver.js +35 -4
  19. package/dist/esm/core/style-cascade/types.d.ts +19 -1
  20. package/dist/esm/core/style-cascade/types.js +3 -0
  21. package/dist/esm/index.d.ts +2 -2
  22. package/dist/esm/layout/page-doc.d.ts +10 -3
  23. package/dist/esm/layout/styled-layout.d.ts +48 -7
  24. package/dist/esm/layout/styled-layout.js +881 -119
  25. package/dist/esm/layout/turned-section.d.ts +38 -0
  26. package/dist/esm/layout/turned-section.js +193 -0
  27. package/dist/esm/pdf/styled-page-emitter.js +78 -2
  28. package/dist/esm/pdf-reader/cff-outline.d.ts +27 -0
  29. package/dist/esm/pdf-reader/cff-outline.js +169 -21
  30. package/dist/esm/pdf-reader/content.d.ts +53 -0
  31. package/dist/esm/pdf-reader/content.js +9 -1
  32. package/dist/esm/pdf-reader/display.d.ts +36 -0
  33. package/dist/esm/pdf-reader/display.js +66 -1
  34. package/dist/esm/pdf-reader/document.js +5 -1
  35. package/dist/esm/pdf-reader/embedded-fonts.d.ts +2 -1
  36. package/dist/esm/pdf-reader/embedded-fonts.js +32 -2
  37. package/dist/esm/pdf-reader/encodings.d.ts +8 -0
  38. package/dist/esm/pdf-reader/encodings.js +25 -3
  39. package/dist/esm/pdf-reader/face-outlines.d.ts +78 -0
  40. package/dist/esm/pdf-reader/face-outlines.js +362 -0
  41. package/dist/esm/pdf-reader/figures.d.ts +52 -0
  42. package/dist/esm/pdf-reader/figures.js +433 -0
  43. package/dist/esm/pdf-reader/flow-build.d.ts +131 -5
  44. package/dist/esm/pdf-reader/flow-build.js +345 -26
  45. package/dist/esm/pdf-reader/font.js +321 -36
  46. package/dist/esm/pdf-reader/glyf-outline.d.ts +33 -0
  47. package/dist/esm/pdf-reader/glyf-outline.js +135 -1
  48. package/dist/esm/pdf-reader/glyph-names.js +154 -1
  49. package/dist/esm/pdf-reader/layout.d.ts +93 -0
  50. package/dist/esm/pdf-reader/layout.js +1703 -214
  51. package/dist/esm/pdf-reader/page-numbers.d.ts +53 -0
  52. package/dist/esm/pdf-reader/page-numbers.js +167 -0
  53. package/dist/esm/pdf-reader/tagged.js +182 -23
  54. package/dist/esm/pdf-reader/text.d.ts +6 -3
  55. package/dist/esm/pdf-reader/text.js +98 -9
  56. package/dist/esm/pdf-reader/type1-outline.d.ts +11 -0
  57. package/dist/esm/pdf-reader/type1-outline.js +63 -8
  58. package/dist/esm/pdf-reader/vector.js +71 -1
  59. package/dist/esm/word/doc/doc-reader.js +6 -2
  60. package/dist/esm/word/doc/doc-text.d.ts +6 -0
  61. package/dist/esm/word/doc/doc-text.js +19 -1
  62. package/dist/esm/word/document-parser.d.ts +2 -2
  63. package/dist/esm/word/document-parser.js +11 -1
  64. package/dist/esm/word/docx-reader.js +5 -3
  65. package/dist/esm/word/docx-writer.js +207 -21
  66. package/dist/esm/word/drawing-parser.d.ts +5 -3
  67. package/dist/esm/word/drawing-parser.js +49 -8
  68. package/dist/esm/word/font-embed.d.ts +30 -0
  69. package/dist/esm/word/font-embed.js +173 -0
  70. package/dist/esm/word/font-table.d.ts +10 -0
  71. package/dist/esm/word/font-table.js +13 -1
  72. package/dist/esm/word/index.js +1 -1
  73. package/dist/esm/word/numbering-parser.d.ts +3 -1
  74. package/dist/esm/word/numbering-parser.js +2 -1
  75. package/dist/esm/word/paragraph-properties.d.ts +7 -6
  76. package/dist/esm/word/paragraph-properties.js +14 -2
  77. package/dist/esm/word/run-properties.js +26 -0
  78. package/package.json +8 -3
@@ -0,0 +1,53 @@
1
+ import { NumberingFormat } from '../core/document-model/index.js';
2
+ /** A page's number as its band prints it. */
3
+ export interface PageNumber {
4
+ /** The numeral as the page shows it: "iv", "47". */
5
+ readonly text: string;
6
+ readonly value: number;
7
+ readonly format: NumberingFormat;
8
+ }
9
+ /** A stretch of pages counted in one sequence (§17.6.12). */
10
+ export interface NumberingRun {
11
+ /** The first page of the stretch. */
12
+ readonly from: number;
13
+ readonly format: NumberingFormat;
14
+ /** The number the stretch's first page carries. */
15
+ readonly start: number;
16
+ }
17
+ /** What the pages' bands say about their numbering. */
18
+ export interface PageNumbering {
19
+ /** Each page's number, where its band shows one. */
20
+ readonly numbers: ReadonlyArray<PageNumber | undefined>;
21
+ /** The sequences the pages are counted in, first page first. */
22
+ readonly runs: ReadonlyArray<NumberingRun>;
23
+ }
24
+ /**
25
+ * The numerals a band's text holds, in order.
26
+ *
27
+ * @param text The band's text on one page.
28
+ * @returns Its numerals; a roman one only where it is a numeral as written.
29
+ */
30
+ export declare function numeralsIn(text: string): Array<PageNumber>;
31
+ /**
32
+ * The page numbers a document's running band prints, and the sequences they
33
+ * run in.
34
+ *
35
+ * The number is the numeral that CHANGES from page to page: "Page 3 of 10" is
36
+ * counted by its first, "Chapter I — 47" by its last. A sequence goes on while
37
+ * each page's number is one more than the last in the same numerals, and
38
+ * starts again where it does not — i, ii, iii and then 1 are two sequences.
39
+ * A page whose band shows no number is counted on.
40
+ *
41
+ * @param bands Each page's band text, or undefined where the page has none.
42
+ * @returns The numbering, or undefined where the bands show no page number
43
+ * worth reading as one.
44
+ */
45
+ export declare function pageNumberingOf(bands: ReadonlyArray<string | undefined>): PageNumbering | undefined;
46
+ /**
47
+ * The sequence a page is counted in.
48
+ *
49
+ * @param runs The document's sequences, first page first.
50
+ * @param page The page.
51
+ * @returns The run that page belongs to.
52
+ */
53
+ export declare function runOf(runs: ReadonlyArray<NumberingRun>, page: number): NumberingRun | undefined;
@@ -0,0 +1,167 @@
1
+ //#region src/pdf-reader/page-numbers.ts
2
+ /** A numeral standing as a word of its own: digits, or roman in one case. */
3
+ var NUMERAL = /(?<=^|\s)(\d{1,4}|[ivxlcdm]{1,9}|[IVXLCDM]{1,9})(?=\s|$)/gu;
4
+ /**
5
+ * More sequences than this and the "numbers" are something else that changes
6
+ * from page to page — a date, a figure count — and the pages are left counted
7
+ * the ordinary way.
8
+ */
9
+ var MOST_RUNS = 4;
10
+ /**
11
+ * …unless the sequences run long: a book leaves its blank pages out and skips
12
+ * their numbers, and freeculture.pdf's count jumps two at five of its chapters
13
+ * — seven sequences over three hundred and thirty-five numbered pages, and
14
+ * read as none its footer printed no number at all. A date or a figure count
15
+ * starts a sequence of its own on nearly every page.
16
+ */
17
+ var LEAST_RUN_PAGES = 10;
18
+ /**
19
+ * The numerals a band's text holds, in order.
20
+ *
21
+ * @param text The band's text on one page.
22
+ * @returns Its numerals; a roman one only where it is a numeral as written.
23
+ */
24
+ function numeralsIn(text) {
25
+ const out = [];
26
+ for (const m of text.matchAll(NUMERAL)) {
27
+ const token = m[1];
28
+ if (/^\d+$/u.test(token)) {
29
+ out.push({
30
+ text: token,
31
+ value: Number(token),
32
+ format: "decimal"
33
+ });
34
+ continue;
35
+ }
36
+ const value = romanValue(token);
37
+ if (value === void 0) continue;
38
+ out.push({
39
+ text: token,
40
+ value,
41
+ format: token === token.toLowerCase() ? "lowerRoman" : "upperRoman"
42
+ });
43
+ }
44
+ return out;
45
+ }
46
+ /**
47
+ * A roman numeral's value, where the letters ARE one as a numeral is written
48
+ * — "iv", "xiv", "MCMXC" — and not merely letters a numeral uses: "ic", "vx"
49
+ * and "mix" are words or nothing.
50
+ */
51
+ function romanValue(token) {
52
+ const upper = token.toUpperCase();
53
+ const worth = {
54
+ I: 1,
55
+ V: 5,
56
+ X: 10,
57
+ L: 50,
58
+ C: 100,
59
+ D: 500,
60
+ M: 1e3
61
+ };
62
+ let value = 0;
63
+ for (let i = 0; i < upper.length; i++) {
64
+ const here = worth[upper[i]];
65
+ const next = worth[upper[i + 1] ?? ""] ?? 0;
66
+ value += here < next ? -here : here;
67
+ }
68
+ return value > 0 && value < 4e3 && toRoman(value) === upper ? value : void 0;
69
+ }
70
+ /** The canonical roman numeral for a value, which is how a valid one reads. */
71
+ function toRoman(value) {
72
+ const steps = [
73
+ [1e3, "M"],
74
+ [900, "CM"],
75
+ [500, "D"],
76
+ [400, "CD"],
77
+ [100, "C"],
78
+ [90, "XC"],
79
+ [50, "L"],
80
+ [40, "XL"],
81
+ [10, "X"],
82
+ [9, "IX"],
83
+ [5, "V"],
84
+ [4, "IV"],
85
+ [1, "I"]
86
+ ];
87
+ let out = "";
88
+ let left = value;
89
+ for (const [n, s] of steps) while (left >= n) {
90
+ out += s;
91
+ left -= n;
92
+ }
93
+ return out;
94
+ }
95
+ /**
96
+ * The page numbers a document's running band prints, and the sequences they
97
+ * run in.
98
+ *
99
+ * The number is the numeral that CHANGES from page to page: "Page 3 of 10" is
100
+ * counted by its first, "Chapter I — 47" by its last. A sequence goes on while
101
+ * each page's number is one more than the last in the same numerals, and
102
+ * starts again where it does not — i, ii, iii and then 1 are two sequences.
103
+ * A page whose band shows no number is counted on.
104
+ *
105
+ * @param bands Each page's band text, or undefined where the page has none.
106
+ * @returns The numbering, or undefined where the bands show no page number
107
+ * worth reading as one.
108
+ */
109
+ function pageNumberingOf(bands) {
110
+ const found = bands.map((text) => text === void 0 ? [] : numeralsIn(text));
111
+ const counted = found.filter((n) => n.length > 0);
112
+ if (counted.length < 2) return void 0;
113
+ const width = Math.min(...counted.map((n) => n.length));
114
+ let at = -1;
115
+ for (let k = 0; k < width && at < 0; k++) if (new Set(counted.map((n) => n[k].text)).size > 1) at = k;
116
+ if (at < 0) return void 0;
117
+ const numbers = found.map((n) => n[at]);
118
+ const runs = [];
119
+ numbers.forEach((number, page) => {
120
+ if (number === void 0) return;
121
+ const run = runs[runs.length - 1];
122
+ if (run !== void 0) {
123
+ const expected = run.start + (page - run.from);
124
+ if (number.format === run.format && number.value === expected) return;
125
+ runs.push({
126
+ from: page,
127
+ format: number.format,
128
+ start: number.value
129
+ });
130
+ return;
131
+ }
132
+ const start = number.value - page;
133
+ if (start >= 1) runs.push({
134
+ from: 0,
135
+ format: number.format,
136
+ start
137
+ });
138
+ else runs.push({
139
+ from: 0,
140
+ format: "decimal",
141
+ start: 1
142
+ }, {
143
+ from: page,
144
+ format: number.format,
145
+ start: number.value
146
+ });
147
+ });
148
+ if (runs.length === 0 || runs.length > Math.max(MOST_RUNS, counted.length / LEAST_RUN_PAGES)) return;
149
+ return {
150
+ numbers,
151
+ runs
152
+ };
153
+ }
154
+ /**
155
+ * The sequence a page is counted in.
156
+ *
157
+ * @param runs The document's sequences, first page first.
158
+ * @param page The page.
159
+ * @returns The run that page belongs to.
160
+ */
161
+ function runOf(runs, page) {
162
+ let found;
163
+ for (const run of runs) if (run.from <= page) found = run;
164
+ return found;
165
+ }
166
+ //#endregion
167
+ export { pageNumberingOf, runOf };
@@ -3,13 +3,14 @@ import { ResourceStore } from "../core/ir/resources.js";
3
3
  import { FEATURES } from "../core/ir/features.js";
4
4
  import { collectEmbeddedFonts } from "./embedded-fonts.js";
5
5
  import { collectFaceFamilies } from "./font.js";
6
+ import { displayOf, placeRuns, placeVectors, textFrameOf, wordsTurnOf } from "./display.js";
7
+ import { ASCENDER, CARRIER_LINE_PT, buildFlowDoc, dedupeLosses, floatOntoSheet, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, sectionOnSheet, shapeBlock, spaceAfter, withMeasuredMargins } from "./flow-build.js";
8
+ import { faceOutlinesOf, kernedFaces, pageSpacing } from "./face-outlines.js";
6
9
  import { extractPageText } from "./text.js";
7
10
  import { collectPageVectors } from "./vector.js";
8
- import { displayOf, placeRuns, placeVectors } from "./display.js";
9
- import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock, spaceAfter, withMeasuredMargins } from "./flow-build.js";
10
11
  import { collectPageImages } from "./images.js";
11
12
  import { markDrawnRules } from "./text-rules.js";
12
- import { endedParagraph } from "./layout.js";
13
+ import { FOOTER_PART, HEADER_PART, endedParagraph, footerBand, numberedFrom, numberingOf, pageTextEdges, runningFoot, stepsBetweenWords } from "./layout.js";
13
14
  import { readStructTree } from "./struct-tree.js";
14
15
  //#region src/pdf-reader/tagged.ts
15
16
  var ASSUMED_CONTENT_WIDTH_PT = 468;
@@ -31,8 +32,15 @@ function reconstructTaggedPdf(file) {
31
32
  const root = readStructTree(file);
32
33
  if (!root) return void 0;
33
34
  const pages = file.pages();
34
- const shown = pages.map((page) => displayOf(page));
35
- const placedRuns = pages.map((page, i) => placeRuns(extractPageText(file, page), shown[i]));
35
+ const sheets = pages.map((page) => displayOf(page));
36
+ const painted = /* @__PURE__ */ new Map();
37
+ const extracted = pages.map((page) => extractPageText(file, page, painted));
38
+ const spacing = pageSpacing(extracted);
39
+ const onSheets = extracted.map((runs, i) => placeRuns(runs, sheets[i]));
40
+ const worded = onSheets.filter((runs) => runs.some((r) => r.text.trim() !== ""));
41
+ const turned = worded.length > 0 && worded.every((runs) => wordsTurnOf(runs) === 270);
42
+ const shown = turned ? sheets.map((sheet) => textFrameOf(sheet)) : sheets;
43
+ const placedRuns = turned ? extracted.map((runs, i) => placeRuns(runs, shown[i])) : onSheets;
36
44
  const vectorLosses = [];
37
45
  const pageVectors = pages.map((page, i) => {
38
46
  const lifted = collectPageVectors(file, page, collectPageImages(file, page).images.map((img) => ({
@@ -106,6 +114,23 @@ function reconstructTaggedPdf(file) {
106
114
  });
107
115
  const collectText = (node) => squash([textOf(node), ...node.children.map(collectText)].join(" "));
108
116
  const setting = /* @__PURE__ */ new Map();
117
+ const pageOf = /* @__PURE__ */ new Map();
118
+ /** The highest baseline a page's own paragraphs stand on, page space (y up). */
119
+ const highestOf = (own) => {
120
+ const tops = own.flatMap((el) => {
121
+ const set = setting.get(el);
122
+ return set ? [set.top] : [];
123
+ });
124
+ return tops.length > 0 ? Math.max(...tops) : void 0;
125
+ };
126
+ /** The lowest baseline a page's own paragraphs reach, page space (y up). */
127
+ const lowestOf = (own) => {
128
+ const bottoms = own.flatMap((el) => {
129
+ const set = setting.get(el);
130
+ return set ? [set.bottom] : [];
131
+ });
132
+ return bottoms.length > 0 ? Math.min(...bottoms) : void 0;
133
+ };
109
134
  /** The topmost and bottommost baseline under a node, and its largest face. */
110
135
  function baselinesOf(node) {
111
136
  let top;
@@ -205,33 +230,48 @@ function reconstructTaggedPdf(file) {
205
230
  }).filter((l) => l.spans.some((sp) => sp.text.trim().length > 0));
206
231
  }
207
232
  function emit(node, out) {
233
+ const on = (el) => {
234
+ const page = firstPageOf(node);
235
+ if (page !== void 0) pageOf.set(el, page);
236
+ return el;
237
+ };
208
238
  if (node.type === "Table") {
209
239
  const table = buildTable(node);
210
- if (table) out.push(table);
240
+ if (table) out.push(on(table));
211
241
  return;
212
242
  }
213
243
  if (node.type === "Figure") {
214
244
  for (const img of imagesForNode(node)) {
215
245
  emitted.add(img);
216
- out.push(imageBlock(img, resources, node.alt));
246
+ out.push(on(imageBlock(img, resources, node.alt)));
217
247
  }
218
248
  return;
219
249
  }
220
250
  if (node.type === "LI") {
221
251
  const text = collectText(node);
222
- if (text.length > 0) out.push(paragraphBlock(text, void 0));
252
+ if (text.length > 0) out.push(on(paragraphBlock(text, void 0)));
223
253
  return;
224
254
  }
225
255
  if (node.children.length === 0) {
226
256
  if (textOf(node).length > 0) for (const part of settingsOf(node)) {
227
257
  const el = paragraphFromRuns(part.spans, headingLevel(node.type));
228
258
  setting.set(el, part.set);
229
- out.push(el);
259
+ out.push(on(el));
230
260
  }
231
261
  return;
232
262
  }
233
263
  for (const child of node.children) emit(child, out);
234
264
  }
265
+ /** The first page any marked content under a node stands on. */
266
+ function firstPageOf(node) {
267
+ let first;
268
+ const visit = (n) => {
269
+ for (const { page } of n.mcids) if (first === void 0 || page < first) first = page;
270
+ for (const child of n.children) visit(child);
271
+ };
272
+ visit(node);
273
+ return first;
274
+ }
235
275
  function buildTable(tableNode) {
236
276
  const raw = [];
237
277
  const collectRows = (n) => {
@@ -295,9 +335,14 @@ function reconstructTaggedPdf(file) {
295
335
  cells
296
336
  };
297
337
  }
298
- const body = [];
299
- emit(root, body);
300
- spaceParagraphs(body, setting, shown[0]?.height ?? 0);
338
+ const named = [];
339
+ emit(root, named);
340
+ const byPage = pages.map(() => []);
341
+ let current = 0;
342
+ for (const el of named) {
343
+ current = Math.max(current, pageOf.get(el) ?? current);
344
+ byPage[current].push(el);
345
+ }
301
346
  let zOrder = -1e6;
302
347
  imageLosses.push(...vectorLosses);
303
348
  pages.forEach((_page, index) => {
@@ -306,31 +351,104 @@ function reconstructTaggedPdf(file) {
306
351
  top: shown[index].height
307
352
  };
308
353
  const taken = ruled[index]?.consumed;
354
+ const drawn = [];
309
355
  for (const v of pageVectors[index] ?? []) {
310
356
  if (taken?.has(v) === true) continue;
311
- body.push(shapeBlock(v, frame, zOrder++, true));
357
+ const shape = shapeBlock(v, frame, zOrder++, true);
358
+ drawn.push(turned ? floatOntoSheet(shape, shown[index].height) : shape);
312
359
  }
360
+ byPage[index].unshift(...drawn);
313
361
  });
314
- const orphans = [];
315
362
  pageImages.forEach((p, page) => {
316
- for (const img of p.images) if (!emitted.has(img)) orphans.push({
317
- page,
318
- img
319
- });
363
+ const left = p.images.filter((img) => !emitted.has(img)).sort((a, b) => b.y - a.y);
364
+ byPage[page].push(...left.map((img) => imageBlock(img, resources)));
365
+ });
366
+ if (byPage.every((own) => own.length === 0)) return void 0;
367
+ const loose = placedRuns.map((runs, i) => shown[i].sheet ? [] : runs.filter((r) => !claimed.has(r)));
368
+ const foot = runningFoot(loose, shown, "foot");
369
+ const head = runningFoot(loose, shown, "head");
370
+ const numberedBand = foot?.numbered === true ? foot : head?.numbered === true ? head : void 0;
371
+ const numbering = numberingOf(numberedBand);
372
+ const lifted = (r, i) => foot?.lift[i]?.has(r) === true || head?.lift[i]?.has(r) === true;
373
+ const measured = withMeasuredMargins(sectionFromPdfPages(pages, shown[0]), shown, placedRuns.map((runs, i) => runs.filter((r) => !lifted(r, i))), pageImages.map((p) => p.images), foot?.band);
374
+ const body = [];
375
+ const top = measured?.margins?.top ?? 0;
376
+ const bottom = measured?.margins?.bottom ?? 0;
377
+ const ends = byPage.map((own) => lowestOf(own));
378
+ const tops = byPage.map((own) => highestOf(own));
379
+ const sectionsFrom = (numbering?.runs ?? []).map((run) => run.from).filter((from) => from > 0);
380
+ const endsAt = [];
381
+ let sectionFrom = 0;
382
+ byPage.forEach((own, index) => {
383
+ if (sectionsFrom.includes(index)) {
384
+ endsAt.push({
385
+ at: body.length,
386
+ from: sectionFrom
387
+ });
388
+ sectionFrom = index;
389
+ }
390
+ const height = shown[index]?.height ?? 0;
391
+ spaceParagraphs(own, setting, height);
392
+ const ended = ends[index - 1];
393
+ const opens = index > 0 && (own.length === 0 || ended === void 0 || ended - bottom > (height - top - bottom) * SHORT_PAGE_SHARE);
394
+ const lead = own.findIndex((el) => !floats(el));
395
+ const set = lead >= 0 ? setting.get(own[lead]) : void 0;
396
+ const fresh = index === 0 || opens || sectionsFrom.includes(index);
397
+ const highest = tops[index];
398
+ const first = set !== void 0 && highest !== void 0 && set.top >= highest - set.size;
399
+ if (fresh && first && measured?.margins) {
400
+ const gap = height - top - (set.top + set.size * ASCENDER);
401
+ if (gap > 1) own[lead] = spacedBefore(own[lead], gap);
402
+ }
403
+ if (opens && !sectionsFrom.includes(index)) body.push(pageBreak(true));
404
+ else if (index === 0 && own.length === 0 && pages.length > 1) body.push(pageBreak(false));
405
+ body.push(...own);
320
406
  });
321
- orphans.sort((a, b) => a.page - b.page || b.img.y - a.img.y);
322
- for (const { img } of orphans) body.push(imageBlock(img, resources));
323
- if (body.length === 0) return void 0;
324
407
  if (placedRuns.some((page) => page.some((r) => r.text.includes("�")))) imageLosses.push({
325
408
  severity: "dropped",
326
409
  feature: FEATURES.text,
327
410
  detail: "some glyphs map to no character — the font states no /ToUnicode and its program says nothing either, so that text is unrecoverable"
328
411
  });
329
- const onPage = placedRuns.flat().reduce((n, r) => n + r.text.length, 0);
412
+ const onPage = placedRuns.flatMap((runs, i) => runs.filter((r) => !lifted(r, i))).reduce((n, r) => n + r.text.length, 0);
330
413
  const reached = [...claimed].reduce((n, r) => n + r.text.length, 0);
331
414
  if (onPage > 0 && reached * 2 < onPage) return void 0;
415
+ endsAt.push({
416
+ at: body.length,
417
+ from: sectionFrom
418
+ });
419
+ const stepped0 = stepsBetweenWords(placedRuns[0] ?? []);
420
+ const edges0 = pageTextEdges(placedRuns[0] ?? []);
421
+ const numeralOf = (of) => {
422
+ if (of === void 0 || of !== numberedBand || numbering === void 0) return void 0;
423
+ const first = of.lift.findIndex((set) => set.size > 0);
424
+ return first >= 0 ? numbering.numbers[first]?.text : void 0;
425
+ };
426
+ const footBand = foot ? footerBand(foot.band, stepped0, edges0, foot.numbered, numeralOf(foot)) : [];
427
+ const headBand = head ? footerBand(head.band, stepped0, edges0, head.numbered, numeralOf(head)) : [];
428
+ const sectionAt = (from) => {
429
+ const base = sectionOnSheet(measured, shown[0]);
430
+ return base ? {
431
+ ...base,
432
+ ...numberedFrom(numbering, from),
433
+ ...footBand.length > 0 ? { footers: [{
434
+ type: "default",
435
+ relationshipId: FOOTER_PART
436
+ }] } : {},
437
+ ...headBand.length > 0 ? { headers: [{
438
+ type: "default",
439
+ relationshipId: HEADER_PART
440
+ }] } : {}
441
+ } : base;
442
+ };
443
+ const sections = endsAt.length > 1 ? endsAt.flatMap((end) => {
444
+ const properties = sectionAt(end.from);
445
+ return properties ? [{
446
+ properties,
447
+ endIndex: end.at
448
+ }] : [];
449
+ }) : [];
332
450
  return {
333
- doc: buildFlowDoc(body, resources, withMeasuredMargins(sectionFromPdfPages(pages), shown, placedRuns, pageImages.map((p) => p.images)), collectEmbeddedFonts(file, pages, imageLosses), [], void 0, collectFaceFamilies(file, pages)),
451
+ doc: buildFlowDoc(body, resources, sectionAt(0), collectEmbeddedFonts(file, pages, imageLosses), sections, footBand.length > 0 || headBand.length > 0 ? new Map([...footBand.length > 0 ? [[FOOTER_PART, footBand]] : [], ...headBand.length > 0 ? [[HEADER_PART, headBand]] : []]) : void 0, collectFaceFamilies(file, pages), faceOutlinesOf(painted, spacing), kernedFaces(spacing)),
334
452
  losses: imageLosses
335
453
  };
336
454
  }
@@ -437,6 +555,47 @@ var MCID_SPACE_EM = .15;
437
555
  function squash(text) {
438
556
  return text.replace(/\s+/g, " ").trim();
439
557
  }
558
+ /**
559
+ * How much of a page's text area the page may leave empty below its last line
560
+ * and still be FULL: a page that ends higher than this ended on purpose.
561
+ */
562
+ var SHORT_PAGE_SHARE = .25;
563
+ /** Whether an element FLOATS — a drawing anchored to its page, which takes no room. */
564
+ function floats(el) {
565
+ return el.kind === "image" && el.image.float !== void 0 || el.kind === "shape" && el.shape.float !== void 0;
566
+ }
567
+ /** A paragraph given the space the page left above it; anything else as it is. */
568
+ function spacedBefore(el, before) {
569
+ if (el.kind !== "paragraph") return el;
570
+ const properties = {
571
+ ...el.paragraph.properties,
572
+ spacingBefore: pt(before)
573
+ };
574
+ return {
575
+ ...el,
576
+ paragraph: {
577
+ ...el.paragraph,
578
+ properties
579
+ }
580
+ };
581
+ }
582
+ /**
583
+ * An empty paragraph that takes no room: the carrier of a page break, or of
584
+ * nothing at all on a blank first sheet the next page's break has to follow.
585
+ */
586
+ function pageBreak(breaks) {
587
+ return {
588
+ kind: "paragraph",
589
+ paragraph: {
590
+ properties: {
591
+ ...breaks ? { pageBreakBefore: true } : {},
592
+ spacingLine: CARRIER_LINE_PT,
593
+ spacingLineRule: "exact"
594
+ },
595
+ runs: []
596
+ }
597
+ };
598
+ }
440
599
  function headingLevel(type) {
441
600
  const m = /^H([1-6])$/.exec(type);
442
601
  return m ? Number(m[1]) - 1 : void 0;
@@ -1,4 +1,5 @@
1
1
  import { ContentFont, TextRun } from './content.js';
2
+ import { ShownCodes } from './face-outlines.js';
2
3
  import { PdfDict } from '../pdf/objects.js';
3
4
  import { PdfFile, PdfPage } from './document.js';
4
5
  /**
@@ -8,11 +9,13 @@ import { PdfFile, PdfPage } from './document.js';
8
9
  * origin falls inside a `/Link` annotation's `/Rect` with that link's URI (EP8)
9
10
  * so hyperlinks survive.
10
11
  *
11
- * @param file The owning {@link PdfFile}.
12
- * @param page The page to extract.
12
+ * @param file The owning {@link PdfFile}.
13
+ * @param page The page to extract.
14
+ * @param painted Where to gather the codes each font painted, for a caller
15
+ * that embeds the faces (see `./face-outlines`).
13
16
  * @returns The page's runs, each carrying an `href` when it sits under a link.
14
17
  */
15
- export declare function extractPageText(file: PdfFile, page: PdfPage): Array<TextRun>;
18
+ export declare function extractPageText(file: PdfFile, page: PdfPage, painted?: ShownCodes): Array<TextRun>;
16
19
  /**
17
20
  * The `/Font` resources of one dictionary, built into interpreter fonts. Shared
18
21
  * with the path and picture walks, which need them for one thing only: a Type 3