reamkit 1.25.1 → 1.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +12 -5
  2. package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
  3. package/dist/esm/pdf-reader/annot-draw.js +374 -0
  4. package/dist/esm/pdf-reader/annots.d.ts +3 -1
  5. package/dist/esm/pdf-reader/annots.js +18 -4
  6. package/dist/esm/pdf-reader/ccitt.d.ts +18 -0
  7. package/dist/esm/pdf-reader/ccitt.js +70 -2
  8. package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
  9. package/dist/esm/pdf-reader/cie-color.js +112 -0
  10. package/dist/esm/pdf-reader/content.d.ts +47 -17
  11. package/dist/esm/pdf-reader/content.js +175 -14
  12. package/dist/esm/pdf-reader/display.d.ts +1 -1
  13. package/dist/esm/pdf-reader/display.js +21 -6
  14. package/dist/esm/pdf-reader/document.d.ts +6 -0
  15. package/dist/esm/pdf-reader/document.js +29 -1
  16. package/dist/esm/pdf-reader/embedded-fonts.d.ts +12 -3
  17. package/dist/esm/pdf-reader/embedded-fonts.js +22 -3
  18. package/dist/esm/pdf-reader/flow-build.d.ts +12 -4
  19. package/dist/esm/pdf-reader/flow-build.js +102 -8
  20. package/dist/esm/pdf-reader/font.js +97 -16
  21. package/dist/esm/pdf-reader/function.d.ts +16 -0
  22. package/dist/esm/pdf-reader/function.js +414 -0
  23. package/dist/esm/pdf-reader/glyph-names.d.ts +6 -0
  24. package/dist/esm/pdf-reader/glyph-names.js +408 -0
  25. package/dist/esm/pdf-reader/image-decode.d.ts +8 -4
  26. package/dist/esm/pdf-reader/image-decode.js +224 -12
  27. package/dist/esm/pdf-reader/images.d.ts +12 -0
  28. package/dist/esm/pdf-reader/images.js +109 -9
  29. package/dist/esm/pdf-reader/jbig2.d.ts +109 -0
  30. package/dist/esm/pdf-reader/jbig2.js +2606 -0
  31. package/dist/esm/pdf-reader/layout.d.ts +32 -0
  32. package/dist/esm/pdf-reader/layout.js +189 -40
  33. package/dist/esm/pdf-reader/lexer.d.ts +2 -0
  34. package/dist/esm/pdf-reader/lexer.js +4 -0
  35. package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
  36. package/dist/esm/pdf-reader/optional-content.js +93 -0
  37. package/dist/esm/pdf-reader/reader.d.ts +5 -2
  38. package/dist/esm/pdf-reader/reader.js +80 -7
  39. package/dist/esm/pdf-reader/shading.d.ts +71 -6
  40. package/dist/esm/pdf-reader/shading.js +185 -15
  41. package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
  42. package/dist/esm/pdf-reader/standard-metrics.js +18 -0
  43. package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
  44. package/dist/esm/pdf-reader/standard-widths.js +62 -0
  45. package/dist/esm/pdf-reader/tagged.js +204 -32
  46. package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
  47. package/dist/esm/pdf-reader/text-rules.js +112 -0
  48. package/dist/esm/pdf-reader/text.js +78 -4
  49. package/dist/esm/pdf-reader/vector.d.ts +5 -0
  50. package/dist/esm/pdf-reader/vector.js +39 -25
  51. package/dist/esm/word/docx-writer.js +213 -23
  52. package/package.json +1 -1
@@ -1,5 +1,7 @@
1
1
  import { PdfFile } from './document.js';
2
2
  import { Reconstruction } from './flow-build.js';
3
+ /** §9.10.2 — a glyph the face maps to no character (see `./font`). */
4
+ export declare const UNMAPPED = "\uFFFD";
3
5
  /**
4
6
  * Heuristically reconstruct an untagged PDF into a {@link Reconstruction}
5
7
  * (E-PDF EP4). With no structure tree there is only positioned content, so
@@ -16,3 +18,33 @@ import { Reconstruction } from './flow-build.js';
16
18
  * @returns The reconstructed {@link FlowDoc} plus any read-time losses.
17
19
  */
18
20
  export declare function reconstructByLayout(file: PdfFile, mode?: 'flow' | 'positional'): Reconstruction;
21
+ /**
22
+ * Whether a line ENDED a paragraph, rather than wrapping into the next.
23
+ *
24
+ * Leading alone cannot tell the two apart: five labels stacked at 15pt with a
25
+ * 12pt face look exactly like five wrapped lines, and alphatrans.pdf's five are
26
+ * read as one paragraph and re-wrapped into two. But a wrapping engine pulls
27
+ * the next word UP — so a line that stops well short of the measure stopped
28
+ * because its author stopped it, and the line after it begins something new.
29
+ * The same rule separates two paragraphs set with no extra space between them,
30
+ * which used to run together for the same reason.
31
+ *
32
+ * Only where both lines start at the same edge. Where they do not, the block is
33
+ * placed rather than set — a centred title's every line is short of the measure
34
+ * and none of them ends anything.
35
+ *
36
+ * @param prev The line before: where it starts, how wide it is, its face.
37
+ * @param next The line after — only where it starts matters.
38
+ * @param column The measure both were set in, when it is known.
39
+ * @returns Whether the first line ended a paragraph.
40
+ */
41
+ export declare function endedParagraph(prev: {
42
+ x: number;
43
+ width: number;
44
+ fontSize: number;
45
+ }, next: {
46
+ x: number;
47
+ }, column: {
48
+ left: number;
49
+ right: number;
50
+ } | undefined): boolean;
@@ -1,14 +1,14 @@
1
1
  import { pt } from "../core/ir/units.js";
2
2
  import { ResourceStore } from "../core/ir/resources.js";
3
3
  import { FEATURES } from "../core/ir/features.js";
4
- import { displayOf, placeImages, placeRuns, placeVectors } from "./display.js";
5
- import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock } from "./flow-build.js";
6
- import { collectEmbeddedFonts } from "./embedded-fonts.js";
7
4
  import { isRightToLeft } from "./content.js";
8
- import { collectPageImages } from "./images.js";
5
+ import { collectEmbeddedFonts } from "./embedded-fonts.js";
9
6
  import { extractPageText } from "./text.js";
10
7
  import { collectPageVectors } from "./vector.js";
11
- //#region src/pdf-reader/layout.ts
8
+ import { displayOf, placeImages, placeRuns, placeVectors } from "./display.js";
9
+ import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, positionedText, sectionFromPdfPages, shapeBlock, withMeasuredMargins } from "./flow-build.js";
10
+ import { collectPageImages } from "./images.js";
11
+ import { markDrawnRules } from "./text-rules.js";
12
12
  /**
13
13
  * Heuristically reconstruct an untagged PDF into a {@link Reconstruction}
14
14
  * (E-PDF EP4). With no structure tree there is only positioned content, so
@@ -31,6 +31,11 @@ function reconstructByLayout(file, mode = "flow") {
31
31
  const medianFont = median(pageRuns.flat().map((r) => r.fontSizePt).filter((s) => s > 0)) || 12;
32
32
  const resources = new ResourceStore();
33
33
  const losses = [];
34
+ if (pageRuns.some((page) => page.some((r) => r.text.includes("�")))) losses.push({
35
+ severity: "dropped",
36
+ feature: FEATURES.text,
37
+ detail: "some glyphs map to no character — the font states no /ToUnicode and its program says nothing either, so that text is unrecoverable"
38
+ });
34
39
  if (pageRuns.some((page) => page.some((r) => r.fillPatternName !== void 0))) losses.push({
35
40
  severity: "degraded",
36
41
  feature: FEATURES.text,
@@ -42,11 +47,12 @@ function reconstructByLayout(file, mode = "flow") {
42
47
  const display = shown[i];
43
48
  const pageWidth = display.width;
44
49
  const gutter = detectGutter(runs, pageWidth);
50
+ const stepped = stepsBetweenWords(runs);
45
51
  const blocks = [];
46
52
  const addColumn = (allRuns, col) => {
47
53
  const colRuns = mode === "positional" ? allRuns.filter((r) => r.type3 !== true && r.invisible !== true) : allRuns;
48
54
  if (mode === "positional") {
49
- for (const [angle, runs] of byAngle(colRuns)) for (const line of groupIntoLines(rotate(runs, -angle), true)) {
55
+ for (const [angle, runs] of byAngle(colRuns)) for (const line of groupIntoLines(rotate(runs, -angle), true, stepped)) {
50
56
  if (line.text.length === 0) continue;
51
57
  const box = turnedBox(line, angle, pageWidth);
52
58
  placed.push({
@@ -58,18 +64,21 @@ function reconstructByLayout(file, mode = "flow") {
58
64
  }
59
65
  return;
60
66
  }
61
- const lines = groupIntoLines(colRuns).filter((l) => l.text.length > 0);
62
- for (const para of groupIntoParagraphs(lines)) blocks.push({
67
+ const lines = groupIntoLines(colRuns, false, stepped).filter((l) => l.text.length > 0);
68
+ const measure = lines.length > 0 ? {
69
+ left: Math.min(...lines.map((l) => l.x)),
70
+ right: Math.max(...lines.map((l) => l.x + l.width))
71
+ } : void 0;
72
+ for (const para of groupIntoParagraphs(lines, measure, display.height)) blocks.push({
63
73
  col,
64
74
  top: para.top,
65
- el: paragraphFromRuns(para.spans, headingLevel(para.fontSize, medianFont))
75
+ el: paragraphFromRuns(para.spans, headingLevel(para.fontSize, medianFont), {
76
+ ...para.alignment !== void 0 ? { alignment: para.alignment } : {},
77
+ ...para.spacingBefore !== void 0 ? { spacingBefore: pt(para.spacingBefore) } : {}
78
+ })
66
79
  });
67
80
  };
68
81
  const placed = [];
69
- if (gutter !== void 0) {
70
- addColumn(runs.filter((r) => r.x < gutter), 0);
71
- addColumn(runs.filter((r) => r.x >= gutter), 1);
72
- } else addColumn(runs, 0);
73
82
  const colOf = (centerX) => gutter !== void 0 && centerX >= gutter ? 1 : 0;
74
83
  const frame = {
75
84
  left: 0,
@@ -88,19 +97,26 @@ function reconstructByLayout(file, mode = "flow") {
88
97
  maxY: img.y + img.heightPt
89
98
  })));
90
99
  losses.push(...lifted.losses);
91
- const vectors = placeVectors(lifted.vectors, display);
100
+ const placedVectors = placeVectors(lifted.vectors, display);
101
+ const ruled = markDrawnRules(runs, placedVectors);
102
+ const vectors = placedVectors.filter((v) => !ruled.consumed.has(v));
103
+ if (gutter !== void 0) {
104
+ addColumn(ruled.runs.filter((r) => r.x < gutter), 0);
105
+ addColumn(ruled.runs.filter((r) => r.x >= gutter), 1);
106
+ } else addColumn(ruled.runs, 0);
107
+ const under = mode !== "positional";
92
108
  [
93
109
  ...imgs.images.map((img) => ({
94
110
  key: img.orderKey,
95
111
  col: colOf(img.x + img.widthPt / 2),
96
112
  top: img.y + img.heightPt,
97
- make: (z) => imageBlock(img, resources, void 0, frame, z)
113
+ make: (z) => imageBlock(img, resources, void 0, frame, z, under)
98
114
  })),
99
115
  ...vectors.map((v) => ({
100
116
  key: v.orderKey,
101
117
  col: colOf((v.minX + v.maxX) / 2),
102
118
  top: v.maxY,
103
- make: (z) => shapeBlock(v, frame, z)
119
+ make: (z) => shapeBlock(v, frame, z, under)
104
120
  })),
105
121
  ...placed
106
122
  ].sort((a, b) => compareOrder(a.key, b.key)).forEach((mark, z) => {
@@ -124,8 +140,9 @@ function reconstructByLayout(file, mode = "flow") {
124
140
  });
125
141
  for (const block of blocks) body.push(block.el);
126
142
  });
143
+ const section = sectionFromPdfPages(pages);
127
144
  return {
128
- doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages), collectEmbeddedFonts(file, pages)),
145
+ doc: buildFlowDoc(body, resources, mode === "positional" ? section : withMeasuredMargins(section, shown, pageRuns), collectEmbeddedFonts(file, pages, losses)),
129
146
  losses: dedupeLosses(losses)
130
147
  };
131
148
  }
@@ -241,7 +258,7 @@ function rotation60kOf(angleDeg) {
241
258
  * 0.05, and thirty at 0.05 em or more. A twentieth of an em sits in that gap.
242
259
  */
243
260
  var BASELINE_STEP_EM = .05;
244
- function groupIntoLines(runs, split = false) {
261
+ function groupIntoLines(runs, split = false, stepped = false) {
245
262
  const sorted = [...runs].sort((a, b) => b.y - a.y || a.x - b.x);
246
263
  const clusters = [];
247
264
  for (const run of sorted) {
@@ -259,7 +276,7 @@ function groupIntoLines(runs, split = false) {
259
276
  return clusters.flatMap((c) => {
260
277
  const ordered = c.runs.sort((a, b) => a.x - b.x);
261
278
  const fontSize = c.fontSize || 10;
262
- if (!split) return [lineOf(ordered, c.y, fontSize)];
279
+ if (!split) return [lineOf(ordered, c.y, fontSize, stepped)];
263
280
  const pieces = [[]];
264
281
  for (const run of ordered) {
265
282
  const prev = pieces[pieces.length - 1];
@@ -268,30 +285,96 @@ function groupIntoLines(runs, split = false) {
268
285
  if (last !== void 0 && (Math.abs(run.x - last.endX) > size * SPACE_GAP_EM || Math.abs(run.y - prev[0].y) > size * BASELINE_STEP_EM || isRightToLeft(run.text) || isRightToLeft(last.text))) pieces.push([]);
269
286
  pieces[pieces.length - 1].push(run);
270
287
  }
271
- return pieces.map((piece) => lineOf(piece, piece[0].y, Math.max(...piece.map((r) => r.fontSizePt || 0)) || fontSize));
288
+ return pieces.map((piece) => lineOf(piece, piece[0].y, Math.max(...piece.map((r) => r.fontSizePt || 0)) || fontSize, stepped));
272
289
  });
273
290
  }
291
+ /**
292
+ * Where a line's INK is, which is not how far its pen travelled.
293
+ *
294
+ * A run of spaces advances the pen and marks nothing. basicapi.pdf sets its
295
+ * page number as thirty-one spaces and "page 1 / 3" in ONE run, reaching 635pt
296
+ * across a 595pt sheet — and the measure taken off that line was wide enough
297
+ * that the centred title in the same column no longer looked centred.
298
+ *
299
+ * The blanks are deducted at the face's own space width, which the run carries
300
+ * (§9.4.4); where the face states none, at a quarter of the size.
301
+ */
302
+ function inkSpan(runs) {
303
+ const marked = runs.filter((r) => r.text.trim().length > 0);
304
+ if (marked.length === 0) {
305
+ const x = runs[0].x;
306
+ return {
307
+ x,
308
+ width: runs[runs.length - 1].endX - x
309
+ };
310
+ }
311
+ const first = marked[0];
312
+ const last = marked[marked.length - 1];
313
+ const space = (r) => r.spaceWidthPt !== void 0 && r.spaceWidthPt > 0 ? r.spaceWidthPt : (r.fontSizePt || 10) * .25;
314
+ const lead = (/^\s*/u.exec(first.text)?.[0].length ?? 0) * space(first);
315
+ const trail = (/\s*$/u.exec(last.text)?.[0].length ?? 0) * space(last);
316
+ const x = first.x + lead;
317
+ return {
318
+ x,
319
+ width: Math.max(0, last.endX - trail - x)
320
+ };
321
+ }
274
322
  /** One run of runs, left to right on a shared baseline, as a {@link Line}. */
275
- function lineOf(runs, y, fontSize) {
276
- const ordered = lineSpans(runs, fontSize);
323
+ function lineOf(runs, y, fontSize, stepped) {
324
+ const ordered = lineSpans(runs, fontSize, stepped);
277
325
  const spans = ordered.every((s) => s.text.trim() === "" || isRightToLeft(s.text)) ? [...ordered].reverse() : ordered;
278
- const x = runs[0].x;
326
+ const ink = inkSpan(runs);
279
327
  return {
280
- x,
281
- width: runs[runs.length - 1].endX - x,
328
+ x: ink.x,
329
+ width: ink.width,
282
330
  y,
283
331
  fontSize,
284
332
  text: spans.map((s) => s.text).join("").replace(/\s+/g, " ").trim(),
285
333
  spans
286
334
  };
287
335
  }
288
- function lineSpans(runs, fontSize) {
336
+ /**
337
+ * The gap between two runs that means a WORD SPACE stood there.
338
+ *
339
+ * A page that draws its own spaces has already said where its words divide, and
340
+ * a gap between two of its runs is a COLUMN or a placement — 160F-2019.pdf is
341
+ * ruled into fields a quarter-inch apart and a generous threshold keeps them
342
+ * apart. A page that draws none has said nothing, and every word boundary on it
343
+ * is a gap: bigboundingbox.pdf steps 0.226 em between words and never writes a
344
+ * space, so at a quarter em its every line ran together — "OrangeDemoInc.",
345
+ * "Whenpayingbycheck,pleasecompletethispaymentadvice".
346
+ *
347
+ * The two want different thresholds, and the page says which it is (see
348
+ * {@link stepsBetweenWords}). The tight one still clears the gaps a producer
349
+ * leaves INSIDE a word when it splits one for kerning, which measure eight
350
+ * hundredths of an em at their widest across this corpus.
351
+ */
352
+ function spaceGap(prev, fontSize, stepped) {
353
+ return (prev.fontSizePt || fontSize) * (stepped ? STEPPED_SPACE_EM : DRAWN_SPACE_EM);
354
+ }
355
+ /** A page that writes its own spaces: only a wide gap means anything more. */
356
+ var DRAWN_SPACE_EM = .25;
357
+ /** A page that writes none: the step between its words is all there is. */
358
+ var STEPPED_SPACE_EM = .12;
359
+ /**
360
+ * Whether this page STEPS between its words rather than writing spaces.
361
+ *
362
+ * Counted rather than guessed: bigboundingbox.pdf writes a space in one run in
363
+ * a hundred, TAMReview.pdf in a third of them, and no page does a little of
364
+ * both. A page with almost no text says nothing either way and keeps the
365
+ * cautious reading.
366
+ */
367
+ function stepsBetweenWords(runs) {
368
+ if (runs.length < 8) return false;
369
+ return runs.filter((r) => /\s/u.test(r.text)).length / runs.length < .05;
370
+ }
371
+ function lineSpans(runs, fontSize, stepped) {
289
372
  const spans = [];
290
- let prevEnd;
373
+ let prev;
291
374
  for (const run of runs) {
292
- if (prevEnd !== void 0 && run.x - prevEnd > fontSize * .25) spans.push({ text: " " });
375
+ if (prev !== void 0 && run.x - prev.endX > spaceGap(prev, fontSize, stepped)) spans.push({ text: " " });
293
376
  spans.push({
294
- text: run.text,
377
+ text: run.text.replaceAll("�", ""),
295
378
  sizePt: run.fontSizePt,
296
379
  ...run.colorHex !== "000000" ? { colorHex: run.colorHex } : {},
297
380
  ...run.fontName !== void 0 ? { fontName: run.fontName } : {},
@@ -301,26 +384,92 @@ function lineSpans(runs, fontSize) {
301
384
  } } : {},
302
385
  ...run.bold ? { bold: true } : {},
303
386
  ...run.italic ? { italic: true } : {},
387
+ ...run.markup !== void 0 ? { markup: run.markup } : {},
304
388
  ...run.href !== void 0 ? { href: run.href } : {}
305
389
  });
306
- prevEnd = run.endX;
390
+ prev = run;
307
391
  }
308
392
  return spans;
309
393
  }
310
- function groupIntoParagraphs(lines) {
394
+ function groupIntoParagraphs(lines, column, pageHeight = 0) {
311
395
  const groups = [];
312
- let prevY;
396
+ const gaps = [];
397
+ let prev;
313
398
  for (const line of lines) {
314
- const gap = prevY !== void 0 ? prevY - line.y : 0;
315
- if (groups.length === 0 || prevY !== void 0 && gap > line.fontSize * 1.5) groups.push([]);
399
+ const gap = prev !== void 0 ? prev.y - line.y : 0;
400
+ const opened = prev !== void 0 && gap > line.fontSize * 1.5;
401
+ if (groups.length === 0 || opened || prev !== void 0 && endedParagraph(prev, line, column)) {
402
+ groups.push([]);
403
+ gaps.push(prev === void 0 ? 0 : gap);
404
+ }
316
405
  groups[groups.length - 1].push(line);
317
- prevY = line.y;
406
+ prev = line;
318
407
  }
319
- return groups.map((g) => ({
320
- spans: g.flatMap((l, i) => i > 0 ? [{ text: " " }, ...l.spans] : [...l.spans]),
321
- fontSize: Math.max(...g.map((l) => l.fontSize)),
322
- top: g[0].y
408
+ return groups.map((g, i) => {
409
+ const first = g[0];
410
+ const fontSize = Math.max(...g.map((l) => l.fontSize));
411
+ const opened = (gaps[i] ?? 0) - fontSize * 1.2;
412
+ const most = pageHeight > 0 ? pageHeight / 3 : fontSize * 3;
413
+ const spacingBefore = opened > fontSize * .3 ? Math.min(opened, most) : void 0;
414
+ return {
415
+ spans: g.flatMap((l, k) => k > 0 ? [{ text: " " }, ...l.spans] : [...l.spans]),
416
+ fontSize,
417
+ top: first.y,
418
+ ...spacingBefore !== void 0 ? { spacingBefore } : {},
419
+ ...alignmentOf(g, column)
420
+ };
421
+ });
422
+ }
423
+ /**
424
+ * Whether a line ENDED a paragraph, rather than wrapping into the next.
425
+ *
426
+ * Leading alone cannot tell the two apart: five labels stacked at 15pt with a
427
+ * 12pt face look exactly like five wrapped lines, and alphatrans.pdf's five are
428
+ * read as one paragraph and re-wrapped into two. But a wrapping engine pulls
429
+ * the next word UP — so a line that stops well short of the measure stopped
430
+ * because its author stopped it, and the line after it begins something new.
431
+ * The same rule separates two paragraphs set with no extra space between them,
432
+ * which used to run together for the same reason.
433
+ *
434
+ * Only where both lines start at the same edge. Where they do not, the block is
435
+ * placed rather than set — a centred title's every line is short of the measure
436
+ * and none of them ends anything.
437
+ *
438
+ * @param prev The line before: where it starts, how wide it is, its face.
439
+ * @param next The line after — only where it starts matters.
440
+ * @param column The measure both were set in, when it is known.
441
+ * @returns Whether the first line ended a paragraph.
442
+ */
443
+ function endedParagraph(prev, next, column) {
444
+ if (!column) return false;
445
+ const width = column.right - column.left;
446
+ if (!(width > 0)) return false;
447
+ if (Math.abs(prev.x - next.x) > Math.max(prev.fontSize, 4)) return false;
448
+ return column.right - (prev.x + prev.width) > width * .25;
449
+ }
450
+ /**
451
+ * §17.3.1.13 — where a paragraph sits across its column, which is the only
452
+ * witness a PDF leaves of how it was set: every line is placed absolutely and
453
+ * nothing says "centred".
454
+ *
455
+ * A paragraph whose lines are inset by about as much on each side is centred; a
456
+ * one-line paragraph pushed to the right edge is right-aligned. Everything else
457
+ * is left alone — a justified paragraph and a ragged-right one look the same
458
+ * from here, and guessing between them would re-set the body of every document.
459
+ */
460
+ function alignmentOf(lines, column) {
461
+ if (!column || lines.length === 0) return {};
462
+ const width = column.right - column.left;
463
+ if (!(width > 0)) return {};
464
+ const insets = lines.map((l) => ({
465
+ lead: l.x - column.left,
466
+ trail: column.right - (l.x + l.width)
323
467
  }));
468
+ const meaningful = width * .1;
469
+ const even = width * .06;
470
+ if (insets.every((i) => Math.abs(i.lead - i.trail) <= even) && Math.max(...insets.map((i) => Math.min(i.lead, i.trail))) >= meaningful) return { alignment: "center" };
471
+ if (insets.every((i) => i.trail <= even) && Math.max(...insets.map((i) => i.lead)) >= meaningful) return { alignment: "right" };
472
+ return {};
324
473
  }
325
474
  function headingLevel(fontSize, medianFont) {
326
475
  if (fontSize >= medianFont * 1.5) return 0;
@@ -342,4 +491,4 @@ function compareOrder(a, b) {
342
491
  return a.length - b.length;
343
492
  }
344
493
  //#endregion
345
- export { reconstructByLayout };
494
+ export { endedParagraph, reconstructByLayout };
@@ -45,6 +45,8 @@ export declare class Lexer {
45
45
  constructor(buf: Uint8Array, pos?: number);
46
46
  /** The length of the underlying byte buffer. */
47
47
  get length(): number;
48
+ /** The bytes between two offsets, as a view onto the buffer. */
49
+ slice(from: number, to: number): Uint8Array;
48
50
  /** The byte at index `i`, or −1 when out of range. */
49
51
  byteAt(i: number): number;
50
52
  /** §7.2.3 — skip whitespace and `%`-to-end-of-line comments. */
@@ -34,6 +34,10 @@ var Lexer = class {
34
34
  get length() {
35
35
  return this.buf.length;
36
36
  }
37
+ /** The bytes between two offsets, as a view onto the buffer. */
38
+ slice(from, to) {
39
+ return this.buf.subarray(Math.max(0, from), Math.min(this.buf.length, Math.max(from, to)));
40
+ }
37
41
  /** The byte at index `i`, or −1 when out of range. */
38
42
  byteAt(i) {
39
43
  return i >= 0 && i < this.buf.length ? this.buf[i] : -1;
@@ -0,0 +1,36 @@
1
+ import { PdfDict, PdfStream, PdfValue } from '../pdf/objects.js';
2
+ import { PdfFile } from './document.js';
3
+ /**
4
+ * The optional-content groups the file's DEFAULT configuration turns off
5
+ * (§8.11.4.3).
6
+ *
7
+ * `/BaseState` says what an unlisted group does — `/ON` unless the file says
8
+ * `/OFF` — and the `/ON` and `/OFF` arrays name the exceptions, `/OFF` last.
9
+ *
10
+ * @param file The document.
11
+ * @returns The resolved OCG dictionaries that are hidden.
12
+ */
13
+ export declare function hiddenGroups(file: PdfFile): ReadonlySet<PdfValue>;
14
+ /**
15
+ * Whether an `/OC` entry names something the page does not show.
16
+ *
17
+ * The entry is either a group itself or an `/OCMD` — a membership dictionary
18
+ * naming several groups and a `/P` policy over them (§8.11.2.3). `AnyOn` is the
19
+ * default and the common case: the content shows if any of its groups does.
20
+ *
21
+ * @param file The document.
22
+ * @param oc The `/OC` value, unresolved.
23
+ * @returns `true` where the content it guards is hidden.
24
+ */
25
+ export declare function hiddenByOc(file: PdfFile, oc: PdfValue | undefined): boolean;
26
+ /**
27
+ * The `/Properties` names a page's content may name in `/OC … BDC` that are
28
+ * hidden (§8.11.3.2).
29
+ *
30
+ * @param file The document.
31
+ * @param resources The resource dictionary in force.
32
+ * @returns Name → hidden, for the names that ARE hidden.
33
+ */
34
+ export declare function hiddenProperties(file: PdfFile, resources: PdfDict | undefined): Set<string>;
35
+ /** Whether an XObject carries an `/OC` that hides it (§8.11.3.1). */
36
+ export declare function hiddenXObject(file: PdfFile, stream: PdfStream): boolean;
@@ -0,0 +1,93 @@
1
+ import { PDF_NULL, PdfName } from "../pdf/objects.js";
2
+ //#region src/pdf-reader/optional-content.ts
3
+ /** The document's own answer to "is this group shown?", worked out once. */
4
+ var cache = /* @__PURE__ */ new WeakMap();
5
+ /** An `/OCMD` naming more groups than this is not read: it is not a document. */
6
+ var MAX_GROUPS = 4096;
7
+ /**
8
+ * The optional-content groups the file's DEFAULT configuration turns off
9
+ * (§8.11.4.3).
10
+ *
11
+ * `/BaseState` says what an unlisted group does — `/ON` unless the file says
12
+ * `/OFF` — and the `/ON` and `/OFF` arrays name the exceptions, `/OFF` last.
13
+ *
14
+ * @param file The document.
15
+ * @returns The resolved OCG dictionaries that are hidden.
16
+ */
17
+ function hiddenGroups(file) {
18
+ const had = cache.get(file);
19
+ if (had) return had;
20
+ const hidden = /* @__PURE__ */ new Set();
21
+ const props = file.get(file.catalog, "OCProperties");
22
+ const config = props instanceof Map ? file.get(props, "D") : void 0;
23
+ if (config instanceof Map) {
24
+ const base = file.get(config, "BaseState");
25
+ if (base instanceof PdfName && base.value === "OFF") {
26
+ const all = file.get(props instanceof Map ? props : /* @__PURE__ */ new Map(), "OCGs");
27
+ for (const g of asArray(file, all)) hidden.add(g);
28
+ }
29
+ for (const g of asArray(file, file.get(config, "ON"))) hidden.delete(g);
30
+ for (const g of asArray(file, file.get(config, "OFF"))) hidden.add(g);
31
+ }
32
+ cache.set(file, hidden);
33
+ return hidden;
34
+ }
35
+ /**
36
+ * Whether an `/OC` entry names something the page does not show.
37
+ *
38
+ * The entry is either a group itself or an `/OCMD` — a membership dictionary
39
+ * naming several groups and a `/P` policy over them (§8.11.2.3). `AnyOn` is the
40
+ * default and the common case: the content shows if any of its groups does.
41
+ *
42
+ * @param file The document.
43
+ * @param oc The `/OC` value, unresolved.
44
+ * @returns `true` where the content it guards is hidden.
45
+ */
46
+ function hiddenByOc(file, oc) {
47
+ if (oc === void 0) return false;
48
+ const hidden = hiddenGroups(file);
49
+ const resolved = file.resolve(oc);
50
+ if (!(resolved instanceof Map)) return false;
51
+ const type = file.get(resolved, "Type");
52
+ if (!(type instanceof PdfName) || type.value !== "OCMD") return hidden.has(resolved);
53
+ const groups = asArray(file, file.get(resolved, "OCGs"));
54
+ const single = file.resolve(resolved.get("OCGs") ?? PDF_NULL);
55
+ const list = groups.length > 0 ? groups : single instanceof Map ? [single] : [];
56
+ if (list.length === 0) return false;
57
+ const on = list.filter((g) => !hidden.has(g)).length;
58
+ const policy = file.get(resolved, "P");
59
+ switch (policy instanceof PdfName ? policy.value : "AnyOn") {
60
+ case "AllOn": return on < list.length;
61
+ case "AnyOff": return on === list.length;
62
+ case "AllOff": return on > 0;
63
+ default: return on === 0;
64
+ }
65
+ }
66
+ /**
67
+ * The `/Properties` names a page's content may name in `/OC … BDC` that are
68
+ * hidden (§8.11.3.2).
69
+ *
70
+ * @param file The document.
71
+ * @param resources The resource dictionary in force.
72
+ * @returns Name → hidden, for the names that ARE hidden.
73
+ */
74
+ function hiddenProperties(file, resources) {
75
+ const out = /* @__PURE__ */ new Set();
76
+ if (!resources) return out;
77
+ const props = file.get(resources, "Properties");
78
+ if (!(props instanceof Map)) return out;
79
+ for (const [name, value] of props) if (hiddenByOc(file, value)) out.add(name);
80
+ return out;
81
+ }
82
+ /** Whether an XObject carries an `/OC` that hides it (§8.11.3.1). */
83
+ function hiddenXObject(file, stream) {
84
+ return hiddenByOc(file, stream.dict.get("OC"));
85
+ }
86
+ /** An array entry, resolved; a missing or malformed one comes back empty. */
87
+ function asArray(file, value) {
88
+ const r = value !== void 0 ? file.resolve(value) : void 0;
89
+ if (!Array.isArray(r)) return [];
90
+ return r.slice(0, MAX_GROUPS).map((v) => file.resolve(v));
91
+ }
92
+ //#endregion
93
+ export { hiddenProperties, hiddenXObject };
@@ -13,14 +13,17 @@ import { FlowDoc } from '../core/ir/flow.js';
13
13
  * opens permissions-only encryption.
14
14
  * @param filters Decoders for `/Filter` names this reader does not implement
15
15
  * (§7.4); see {@link StreamFilters}.
16
- * @param layout `'flow'` reads a re-flowable document out of the page —
16
+ * @param layout `'auto'` (the default) lets the FILE decide: a page that is
17
+ * mostly marks is reproduced, one that is mostly lines is re-set,
18
+ * and the reader records which it chose. `'flow'` reads a
19
+ * re-flowable document out of the page —
17
20
  * paragraphs and tables in reading order, from the structure
18
21
  * tree where there is one. `'positional'` keeps the page: every
19
22
  * line stands where its glyphs do, beside the artwork, which is
20
23
  * what a form or a drawing needs and what a paragraph cannot be.
21
24
  * @returns The reconstructed FlowDoc and its accumulated {@link Loss} report.
22
25
  */
23
- export declare function readPdf(bytes: Uint8Array, password?: string, layout?: 'flow' | 'positional', filters?: StreamFilters): ReadResult<FlowDoc>;
26
+ export declare function readPdf(bytes: Uint8Array, password?: string, layout?: 'flow' | 'positional' | 'auto', filters?: StreamFilters): ReadResult<FlowDoc>;
24
27
  /**
25
28
  * The `pdfReader` adapter: a {@link DocumentReader} that sniffs the `%PDF-`
26
29
  * header and parses the bytes into a {@link FlowDoc} (E-PDF EP5).