pdf-codec.js 5.0.1 → 5.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/interpret.js CHANGED
@@ -432,24 +432,38 @@ function runContentStream(bytes, resources, initialState, context, items, depth,
432
432
  const widthPer1000 = glyph?.widthPer1000 ?? FALLBACK_GLYPH_WIDTH_PER_1000;
433
433
  const byteLength = glyph?.byteLengthConsumed ?? 1;
434
434
  const isSingleByteSpace = byteLength === 1 && codes[offset] === 32;
435
- const tx = (widthPer1000 / 1e3 * gs.fontSizePt + gs.charSpace + (isSingleByteSpace ? gs.wordSpace : 0)) * gs.horizScale;
436
- text.tm = multiplyMatrices(translationMatrix(tx, 0), text.tm);
435
+ const spacing = gs.charSpace + (isSingleByteSpace ? gs.wordSpace : 0);
436
+ const vertical = glyph?.vertical;
437
+ text.tm = vertical === void 0 ? multiplyMatrices(translationMatrix((widthPer1000 / 1e3 * gs.fontSizePt + spacing) * gs.horizScale, 0), text.tm) : multiplyMatrices(translationMatrix(0, vertical.displacementPer1000 / 1e3 * gs.fontSizePt + spacing), text.tm);
437
438
  offset += byteLength;
438
439
  }
439
440
  };
441
+ const positionVectorOf = (codes) => {
442
+ const fontResourceName = gs.fontResourceName;
443
+ if (fontResourceName === void 0) return;
444
+ const vertical = context.fontMetrics.glyphAdvance(fontResourceName, resources, codes, 0)?.vertical;
445
+ if (vertical === void 0) return;
446
+ return {
447
+ x: vertical.positionXPer1000 / 1e3,
448
+ y: vertical.positionYPer1000 / 1e3
449
+ };
450
+ };
440
451
  const showTextArray = (elements) => {
441
452
  const fontResourceName = gs.fontResourceName;
442
453
  if (fontResourceName === void 0) return;
454
+ const vertical = context.fontMetrics.isVertical(fontResourceName, resources);
443
455
  const startMatrix = computeTrm(gs, text);
444
456
  const chunks = [];
445
457
  let totalLength = 0;
458
+ let firstGlyphPosition;
446
459
  for (const el of elements) if (el.kind === "string") {
460
+ firstGlyphPosition ??= vertical ? positionVectorOf(el.bytes) : void 0;
447
461
  chunks.push(el.bytes);
448
462
  totalLength += el.bytes.length;
449
463
  advanceThroughString(el.bytes);
450
464
  } else if (el.kind === "number") {
451
- const adjustment = -(el.value / 1e3) * gs.fontSizePt * gs.horizScale;
452
- text.tm = multiplyMatrices(translationMatrix(adjustment, 0), text.tm);
465
+ const adjustment = -(el.value / 1e3) * gs.fontSizePt;
466
+ text.tm = multiplyMatrices(vertical ? translationMatrix(0, adjustment) : translationMatrix(adjustment * gs.horizScale, 0), text.tm);
453
467
  }
454
468
  if (totalLength === 0) return;
455
469
  const combined = new Uint8Array(totalLength);
@@ -459,15 +473,17 @@ function runContentStream(bytes, resources, initialState, context, items, depth,
459
473
  at += chunk.length;
460
474
  }
461
475
  const endMatrix = computeTrm(gs, text);
476
+ const shift = firstGlyphPosition === void 0 ? void 0 : translationMatrix(-firstGlyphPosition.x, -firstGlyphPosition.y);
462
477
  pushItem({
463
478
  kind: "text",
464
479
  codes: combined,
465
480
  fontResourceName,
466
481
  resources,
467
- startMatrix,
468
- endMatrix,
482
+ startMatrix: shift === void 0 ? startMatrix : multiplyMatrices(shift, startMatrix),
483
+ endMatrix: shift === void 0 ? endMatrix : multiplyMatrices(shift, endMatrix),
469
484
  sizePt: gs.fontSizePt,
470
- color: gs.fillColor
485
+ color: gs.fillColor,
486
+ ...vertical ? { vertical: true } : {}
471
487
  });
472
488
  };
473
489
  const showText = (bytes) => {
package/dist/layout.cjs CHANGED
@@ -13,6 +13,7 @@ const LayoutTextSchema = zod.z.object({
13
13
  color: document_schema_js.ColorSchema,
14
14
  widthPt: zod.z.number().nonnegative().optional(),
15
15
  rotationDeg: zod.z.number().optional(),
16
+ writingMode: zod.z.literal("vertical").optional(),
16
17
  underline: zod.z.boolean().optional(),
17
18
  layer: zod.z.string().optional(),
18
19
  actualText: zod.z.string().optional(),
package/dist/layout.d.cts CHANGED
@@ -25,6 +25,7 @@ declare const LayoutTextSchema: z.ZodObject<{
25
25
  }, z.core.$strip>;
26
26
  widthPt: z.ZodOptional<z.ZodNumber>;
27
27
  rotationDeg: z.ZodOptional<z.ZodNumber>;
28
+ writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
28
29
  underline: z.ZodOptional<z.ZodBoolean>;
29
30
  layer: z.ZodOptional<z.ZodString>;
30
31
  actualText: z.ZodOptional<z.ZodString>;
@@ -243,6 +244,7 @@ declare const LayoutItemSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
243
244
  }, z.core.$strip>;
244
245
  widthPt: z.ZodOptional<z.ZodNumber>;
245
246
  rotationDeg: z.ZodOptional<z.ZodNumber>;
247
+ writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
246
248
  underline: z.ZodOptional<z.ZodBoolean>;
247
249
  layer: z.ZodOptional<z.ZodString>;
248
250
  actualText: z.ZodOptional<z.ZodString>;
@@ -475,6 +477,7 @@ declare const LayoutPageSchema: z.ZodObject<{
475
477
  }, z.core.$strip>;
476
478
  widthPt: z.ZodOptional<z.ZodNumber>;
477
479
  rotationDeg: z.ZodOptional<z.ZodNumber>;
480
+ writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
478
481
  underline: z.ZodOptional<z.ZodBoolean>;
479
482
  layer: z.ZodOptional<z.ZodString>;
480
483
  actualText: z.ZodOptional<z.ZodString>;
@@ -825,6 +828,7 @@ declare const LayoutDocumentSchema: z.ZodObject<{
825
828
  }, z.core.$strip>;
826
829
  widthPt: z.ZodOptional<z.ZodNumber>;
827
830
  rotationDeg: z.ZodOptional<z.ZodNumber>;
831
+ writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
828
832
  underline: z.ZodOptional<z.ZodBoolean>;
829
833
  layer: z.ZodOptional<z.ZodString>;
830
834
  actualText: z.ZodOptional<z.ZodString>;
package/dist/layout.d.ts CHANGED
@@ -25,6 +25,7 @@ declare const LayoutTextSchema: z.ZodObject<{
25
25
  }, z.core.$strip>;
26
26
  widthPt: z.ZodOptional<z.ZodNumber>;
27
27
  rotationDeg: z.ZodOptional<z.ZodNumber>;
28
+ writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
28
29
  underline: z.ZodOptional<z.ZodBoolean>;
29
30
  layer: z.ZodOptional<z.ZodString>;
30
31
  actualText: z.ZodOptional<z.ZodString>;
@@ -243,6 +244,7 @@ declare const LayoutItemSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
243
244
  }, z.core.$strip>;
244
245
  widthPt: z.ZodOptional<z.ZodNumber>;
245
246
  rotationDeg: z.ZodOptional<z.ZodNumber>;
247
+ writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
246
248
  underline: z.ZodOptional<z.ZodBoolean>;
247
249
  layer: z.ZodOptional<z.ZodString>;
248
250
  actualText: z.ZodOptional<z.ZodString>;
@@ -475,6 +477,7 @@ declare const LayoutPageSchema: z.ZodObject<{
475
477
  }, z.core.$strip>;
476
478
  widthPt: z.ZodOptional<z.ZodNumber>;
477
479
  rotationDeg: z.ZodOptional<z.ZodNumber>;
480
+ writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
478
481
  underline: z.ZodOptional<z.ZodBoolean>;
479
482
  layer: z.ZodOptional<z.ZodString>;
480
483
  actualText: z.ZodOptional<z.ZodString>;
@@ -825,6 +828,7 @@ declare const LayoutDocumentSchema: z.ZodObject<{
825
828
  }, z.core.$strip>;
826
829
  widthPt: z.ZodOptional<z.ZodNumber>;
827
830
  rotationDeg: z.ZodOptional<z.ZodNumber>;
831
+ writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
828
832
  underline: z.ZodOptional<z.ZodBoolean>;
829
833
  layer: z.ZodOptional<z.ZodString>;
830
834
  actualText: z.ZodOptional<z.ZodString>;
package/dist/layout.js CHANGED
@@ -12,6 +12,7 @@ const LayoutTextSchema = z.object({
12
12
  color: ColorSchema,
13
13
  widthPt: z.number().nonnegative().optional(),
14
14
  rotationDeg: z.number().optional(),
15
+ writingMode: z.literal("vertical").optional(),
15
16
  underline: z.boolean().optional(),
16
17
  layer: z.string().optional(),
17
18
  actualText: z.string().optional(),
package/dist/raster.cjs CHANGED
@@ -533,8 +533,8 @@ function buildTextOutlineFace(fontDict, resolver, fontResolver, fontResourceName
533
533
  const subtype = require_objects.asName(require_objects.dictGet(fontDict, "Subtype"));
534
534
  if (subtype === "Type0") {
535
535
  const encoding = resolver.resolve(require_objects.dictGet(fontDict, "Encoding"));
536
- if (encoding?.kind !== "name" || encoding.name !== "Identity-H") {
537
- unavailable("a Type0 font whose /Encoding is not Identity-H (a predefined or embedded CMap this raster walk does not decode)");
536
+ if (encoding?.kind !== "name" || encoding.name !== "Identity-H" && encoding.name !== "Identity-V") {
537
+ unavailable("a Type0 font whose /Encoding is neither Identity-H nor Identity-V (a predefined or embedded CMap this raster walk does not decode)");
538
538
  return;
539
539
  }
540
540
  const descendants = require_objects.asArray(require_objects.dictGet(fontDict, "DescendantFonts"));
@@ -623,17 +623,19 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
623
623
  let cumulative = 0;
624
624
  while (offset < item.codes.length) {
625
625
  const advance = fontResolver.metrics.glyphAdvance(item.fontResourceName, item.resources, item.codes, offset);
626
- const widthPer1000 = advance.widthPer1000;
627
626
  const byteLength = advance.byteLengthConsumed;
628
627
  placements.push({
629
628
  glyphId: face.glyphIdOf(item.codes, offset),
630
- advance: cumulative
629
+ advance: cumulative,
630
+ positionX: (advance.vertical?.positionXPer1000 ?? 0) / 1e3,
631
+ positionY: (advance.vertical?.positionYPer1000 ?? 0) / 1e3
631
632
  });
632
- cumulative += widthPer1000 / 1e3 * item.sizePt;
633
+ cumulative += (advance.vertical?.displacementPer1000 ?? advance.widthPer1000) / 1e3 * item.sizePt;
633
634
  offset += byteLength;
634
635
  }
635
636
  const startDeviceMatrix = require_matrix.multiplyMatrices(item.startMatrix, interpretToDeviceMatrix);
636
637
  const endDeviceMatrix = require_matrix.multiplyMatrices(item.endMatrix, interpretToDeviceMatrix);
638
+ const vertical = item.vertical === true;
637
639
  const startPoint = require_matrix.applyMatrix(startDeviceMatrix, {
638
640
  x: 0,
639
641
  y: 0
@@ -642,19 +644,23 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
642
644
  x: 0,
643
645
  y: 0
644
646
  });
645
- const rawEndPoint = require_matrix.applyMatrix(startDeviceMatrix, {
647
+ const rawEndPoint = require_matrix.applyMatrix(startDeviceMatrix, vertical ? {
648
+ x: 0,
649
+ y: cumulative
650
+ } : {
646
651
  x: cumulative,
647
652
  y: 0
648
653
  });
649
654
  const rawExtent = Math.hypot(rawEndPoint.x - startPoint.x, rawEndPoint.y - startPoint.y);
650
655
  const actualExtent = Math.hypot(endPoint.x - startPoint.x, endPoint.y - startPoint.y);
651
- const correction = rawExtent > 1e-9 && cumulative > 0 ? actualExtent / rawExtent : 1;
656
+ const correction = rawExtent > 1e-9 ? actualExtent / rawExtent : 1;
657
+ const firstPlacement = placements[0];
652
658
  const glyphScale = require_matrix.scaleMatrix(1 / face.unitsPerEm, 1 / face.unitsPerEm);
653
659
  for (const placement of placements) {
654
660
  if (placement.glyphId === void 0) continue;
655
661
  const outline = require_glyf_contours.decodeGlyphOutline(face.glyf, placement.glyphId);
656
662
  if (outline === void 0) continue;
657
- const trm = require_matrix.multiplyMatrices(require_matrix.translationMatrix(placement.advance * correction, 0), item.startMatrix);
663
+ const trm = require_matrix.multiplyMatrices(require_matrix.translationMatrix((firstPlacement?.positionX ?? 0) - placement.positionX, (firstPlacement?.positionY ?? 0) - placement.positionY), require_matrix.multiplyMatrices(vertical ? require_matrix.translationMatrix(0, placement.advance * correction) : require_matrix.translationMatrix(placement.advance * correction, 0), item.startMatrix));
658
664
  drawGlyphOutline(outline, require_matrix.multiplyMatrices(glyphScale, require_matrix.multiplyMatrices(trm, interpretToDeviceMatrix)), item.color, rasteriser);
659
665
  }
660
666
  }
package/dist/raster.js CHANGED
@@ -532,8 +532,8 @@ function buildTextOutlineFace(fontDict, resolver, fontResolver, fontResourceName
532
532
  const subtype = asName(dictGet(fontDict, "Subtype"));
533
533
  if (subtype === "Type0") {
534
534
  const encoding = resolver.resolve(dictGet(fontDict, "Encoding"));
535
- if (encoding?.kind !== "name" || encoding.name !== "Identity-H") {
536
- unavailable("a Type0 font whose /Encoding is not Identity-H (a predefined or embedded CMap this raster walk does not decode)");
535
+ if (encoding?.kind !== "name" || encoding.name !== "Identity-H" && encoding.name !== "Identity-V") {
536
+ unavailable("a Type0 font whose /Encoding is neither Identity-H nor Identity-V (a predefined or embedded CMap this raster walk does not decode)");
537
537
  return;
538
538
  }
539
539
  const descendants = asArray(dictGet(fontDict, "DescendantFonts"));
@@ -622,17 +622,19 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
622
622
  let cumulative = 0;
623
623
  while (offset < item.codes.length) {
624
624
  const advance = fontResolver.metrics.glyphAdvance(item.fontResourceName, item.resources, item.codes, offset);
625
- const widthPer1000 = advance.widthPer1000;
626
625
  const byteLength = advance.byteLengthConsumed;
627
626
  placements.push({
628
627
  glyphId: face.glyphIdOf(item.codes, offset),
629
- advance: cumulative
628
+ advance: cumulative,
629
+ positionX: (advance.vertical?.positionXPer1000 ?? 0) / 1e3,
630
+ positionY: (advance.vertical?.positionYPer1000 ?? 0) / 1e3
630
631
  });
631
- cumulative += widthPer1000 / 1e3 * item.sizePt;
632
+ cumulative += (advance.vertical?.displacementPer1000 ?? advance.widthPer1000) / 1e3 * item.sizePt;
632
633
  offset += byteLength;
633
634
  }
634
635
  const startDeviceMatrix = multiplyMatrices(item.startMatrix, interpretToDeviceMatrix);
635
636
  const endDeviceMatrix = multiplyMatrices(item.endMatrix, interpretToDeviceMatrix);
637
+ const vertical = item.vertical === true;
636
638
  const startPoint = applyMatrix(startDeviceMatrix, {
637
639
  x: 0,
638
640
  y: 0
@@ -641,19 +643,23 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
641
643
  x: 0,
642
644
  y: 0
643
645
  });
644
- const rawEndPoint = applyMatrix(startDeviceMatrix, {
646
+ const rawEndPoint = applyMatrix(startDeviceMatrix, vertical ? {
647
+ x: 0,
648
+ y: cumulative
649
+ } : {
645
650
  x: cumulative,
646
651
  y: 0
647
652
  });
648
653
  const rawExtent = Math.hypot(rawEndPoint.x - startPoint.x, rawEndPoint.y - startPoint.y);
649
654
  const actualExtent = Math.hypot(endPoint.x - startPoint.x, endPoint.y - startPoint.y);
650
- const correction = rawExtent > 1e-9 && cumulative > 0 ? actualExtent / rawExtent : 1;
655
+ const correction = rawExtent > 1e-9 ? actualExtent / rawExtent : 1;
656
+ const firstPlacement = placements[0];
651
657
  const glyphScale = scaleMatrix(1 / face.unitsPerEm, 1 / face.unitsPerEm);
652
658
  for (const placement of placements) {
653
659
  if (placement.glyphId === void 0) continue;
654
660
  const outline = decodeGlyphOutline(face.glyf, placement.glyphId);
655
661
  if (outline === void 0) continue;
656
- const trm = multiplyMatrices(translationMatrix(placement.advance * correction, 0), item.startMatrix);
662
+ const trm = multiplyMatrices(translationMatrix((firstPlacement?.positionX ?? 0) - placement.positionX, (firstPlacement?.positionY ?? 0) - placement.positionY), multiplyMatrices(vertical ? translationMatrix(0, placement.advance * correction) : translationMatrix(placement.advance * correction, 0), item.startMatrix));
657
663
  drawGlyphOutline(outline, multiplyMatrices(glyphScale, multiplyMatrices(trm, interpretToDeviceMatrix)), item.color, rasteriser);
658
664
  }
659
665
  }
package/dist/read.cjs CHANGED
@@ -366,6 +366,7 @@ function convertText(item, pageMatrix, fontResolver) {
366
366
  color: item.color,
367
367
  widthPt,
368
368
  rotationDeg: rotationDeg !== 0 ? rotationDeg : void 0,
369
+ ...item.vertical === true ? { writingMode: "vertical" } : {},
369
370
  ...item.layerName !== void 0 ? { layer: item.layerName } : {},
370
371
  ...item.actualText !== void 0 ? { actualText: item.actualText } : {},
371
372
  ...item.alt !== void 0 ? { alt: item.alt } : {}
package/dist/read.js CHANGED
@@ -365,6 +365,7 @@ function convertText(item, pageMatrix, fontResolver) {
365
365
  color: item.color,
366
366
  widthPt,
367
367
  rotationDeg: rotationDeg !== 0 ? rotationDeg : void 0,
368
+ ...item.vertical === true ? { writingMode: "vertical" } : {},
368
369
  ...item.layerName !== void 0 ? { layer: item.layerName } : {},
369
370
  ...item.actualText !== void 0 ? { actualText: item.actualText } : {},
370
371
  ...item.alt !== void 0 ? { alt: item.alt } : {}
@@ -0,0 +1,227 @@
1
+ Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
2
+ //#region src/text-group.ts
3
+ const DEFAULT_BASELINE_TOLERANCE_EM = 2 / 3;
4
+ const DEFAULT_WORD_GAP_EM = 1 / 8;
5
+ const DEFAULT_COLUMN_GAP_EM = 1;
6
+ const ROTATION_NOISE_DEG = 1e-6;
7
+ const DEGREES_PER_TURN = 360;
8
+ const VERTICAL_ADVANCE_OFFSET_DEG = -90;
9
+ function advanceAngleDeg(run) {
10
+ const raw = (run.rotationDeg ?? 0) + (run.writingMode === "vertical" ? VERTICAL_ADVANCE_OFFSET_DEG : 0);
11
+ return (Math.round(raw / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
12
+ }
13
+ const RADIANS_PER_TURN = 2 * Math.PI;
14
+ function baselineAxis(rotationDeg) {
15
+ const radians = rotationDeg * RADIANS_PER_TURN / DEGREES_PER_TURN;
16
+ return {
17
+ cos: Math.cos(radians),
18
+ sin: Math.sin(radians)
19
+ };
20
+ }
21
+ function project(run, rotationDeg) {
22
+ const { cos, sin } = baselineAxis(rotationDeg);
23
+ return {
24
+ run,
25
+ alongPt: run.xPt * cos + run.yPt * sin,
26
+ acrossPt: run.yPt * cos - run.xPt * sin,
27
+ sizePt: run.sizePt
28
+ };
29
+ }
30
+ function unproject(alongPt, acrossPt, rotationDeg) {
31
+ const { cos, sin } = baselineAxis(rotationDeg);
32
+ return {
33
+ xPt: alongPt * cos - acrossPt * sin,
34
+ yPt: alongPt * sin + acrossPt * cos
35
+ };
36
+ }
37
+ /**
38
+ * Whether two positioned runs sit on one visual line.
39
+ *
40
+ * The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
41
+ * @param a - one run
42
+ * @param b - the other run
43
+ * @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
44
+ * @returns true when the two baselines are within tolerance of each other
45
+ */
46
+ function runsShareBaseline(a, b, options = {}) {
47
+ const rotationDeg = advanceAngleDeg(a);
48
+ if (rotationDeg !== advanceAngleDeg(b) || a.writingMode !== b.writingMode) return false;
49
+ const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
50
+ return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
51
+ }
52
+ /**
53
+ * The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
54
+ *
55
+ * Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
56
+ * @param previous - the run to the left, in reading order
57
+ * @param next - the run that follows it
58
+ * @returns the gap in points, or undefined when it cannot be derived
59
+ */
60
+ function runGapPt(previous, next) {
61
+ const rotationDeg = advanceAngleDeg(previous);
62
+ if (rotationDeg !== advanceAngleDeg(next) || previous.writingMode !== next.writingMode) return;
63
+ const previousWidthPt = previous.widthPt;
64
+ if (previousWidthPt === void 0) return;
65
+ return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
66
+ }
67
+ function clusterIntoLines(entries, toleranceEm) {
68
+ const lines = [];
69
+ for (const entry of entries) {
70
+ let nearest;
71
+ for (const line of lines) if (runsShareBaseline(line.anchor.run, entry.run, { baselineToleranceEm: toleranceEm })) nearest = line;
72
+ if (nearest === void 0) lines.push({
73
+ anchor: entry,
74
+ entries: [entry]
75
+ });
76
+ else nearest.entries.push(entry);
77
+ }
78
+ return lines;
79
+ }
80
+ function tokeniseRun(text) {
81
+ const words = text.match(/\S+/g) ?? [];
82
+ const leadingSpace = /^\s/.test(text);
83
+ const trailingSpace = /\s$/.test(text);
84
+ return {
85
+ words,
86
+ leadingSpace,
87
+ trailingSpace,
88
+ whole: words.length === 1 && !leadingSpace && !trailingSpace
89
+ };
90
+ }
91
+ const SEPARATOR_TEXT = {
92
+ none: "",
93
+ space: " ",
94
+ column: " "
95
+ };
96
+ function separatorForGap(gapPt, smallerSizePt, wordGapEm, columnGapEm) {
97
+ if (gapPt === void 0) return "none";
98
+ if (gapPt >= columnGapEm * smallerSizePt) return "column";
99
+ if (gapPt >= wordGapEm * smallerSizePt) return "space";
100
+ return "none";
101
+ }
102
+ function boundsOf(minAlongPt, maxAlongEndPt, acrossPt, maxSizePt, rotationDeg) {
103
+ const origin = unproject(minAlongPt, acrossPt, rotationDeg);
104
+ return {
105
+ xPt: origin.xPt,
106
+ yPt: origin.yPt,
107
+ widthPt: maxAlongEndPt - minAlongPt,
108
+ heightPt: maxSizePt
109
+ };
110
+ }
111
+ function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
112
+ const working = [];
113
+ let previous;
114
+ for (const entry of line.entries) {
115
+ const tokens = tokeniseRun(entry.run.text);
116
+ const gapSeparator = previous === void 0 ? "none" : separatorForGap(runGapPt(previous.entry.run, entry.run), Math.min(previous.entry.sizePt, entry.sizePt), wordGapEm, columnGapEm);
117
+ const pendingSpace = previous?.trailingSpace === true;
118
+ let separator = "none";
119
+ if (working.length > 0) separator = gapSeparator === "none" && (pendingSpace || tokens.leadingSpace) ? "space" : gapSeparator;
120
+ const widthPt = entry.run.widthPt;
121
+ const alongEndPt = entry.alongPt + (widthPt ?? 0);
122
+ tokens.words.forEach((word, index) => {
123
+ const last = working[working.length - 1];
124
+ if (index === 0 && separator === "none" && last !== void 0) {
125
+ last.text += word;
126
+ last.runs.push(entry.run);
127
+ last.measurable = last.measurable && tokens.whole && widthPt !== void 0;
128
+ last.minAlongPt = Math.min(last.minAlongPt, entry.alongPt);
129
+ last.maxAlongEndPt = Math.max(last.maxAlongEndPt, alongEndPt);
130
+ last.maxSizePt = Math.max(last.maxSizePt, entry.sizePt);
131
+ return;
132
+ }
133
+ working.push({
134
+ text: word,
135
+ separatorBefore: index === 0 ? separator : "space",
136
+ runs: [entry.run],
137
+ measurable: tokens.whole && widthPt !== void 0,
138
+ minAlongPt: entry.alongPt,
139
+ maxAlongEndPt: alongEndPt,
140
+ maxSizePt: entry.sizePt
141
+ });
142
+ });
143
+ previous = {
144
+ entry,
145
+ trailingSpace: tokens.trailingSpace
146
+ };
147
+ }
148
+ return working.map((word) => ({
149
+ text: word.text,
150
+ separatorBefore: word.separatorBefore,
151
+ runs: word.runs,
152
+ ...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
153
+ }));
154
+ }
155
+ function buildLine(line, rotationDeg, vertical, wordGapEm, columnGapEm) {
156
+ const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
157
+ const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
158
+ let minAlongPt = Number.POSITIVE_INFINITY;
159
+ let maxAlongEndPt = Number.NEGATIVE_INFINITY;
160
+ let maxSizePt = 0;
161
+ let measurable = true;
162
+ for (const entry of line.entries) {
163
+ const widthPt = entry.run.widthPt;
164
+ if (widthPt === void 0) measurable = false;
165
+ minAlongPt = Math.min(minAlongPt, entry.alongPt);
166
+ maxAlongEndPt = Math.max(maxAlongEndPt, entry.alongPt + (widthPt ?? 0));
167
+ maxSizePt = Math.max(maxSizePt, entry.sizePt);
168
+ }
169
+ return {
170
+ text,
171
+ words,
172
+ baselineYPt: line.anchor.acrossPt,
173
+ rotationDeg,
174
+ ...vertical ? { writingMode: "vertical" } : {},
175
+ ...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
176
+ };
177
+ }
178
+ /**
179
+ * Groups positioned PDF text runs into lines of words.
180
+ *
181
+ * Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
182
+ *
183
+ * Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
184
+ *
185
+ * Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
186
+ *
187
+ * Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
188
+ * @param runs - positioned text runs, in any order
189
+ * @param options - tolerance overrides; each defaults to the exported constant of the same name
190
+ * @returns the grouped lines, in reading order
191
+ */
192
+ function groupPdfTextRuns(runs, options = {}) {
193
+ const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
194
+ const wordGapEm = options.wordGapEm ?? .125;
195
+ const columnGapEm = options.columnGapEm ?? 1;
196
+ const buckets = /* @__PURE__ */ new Map();
197
+ for (const run of runs) {
198
+ if (run.text.length === 0) continue;
199
+ const rotationDeg = advanceAngleDeg(run);
200
+ const vertical = run.writingMode === "vertical";
201
+ const key = `${String(rotationDeg)}|${String(vertical)}`;
202
+ const bucket = buckets.get(key);
203
+ if (bucket === void 0) buckets.set(key, {
204
+ rotationDeg,
205
+ vertical,
206
+ runs: [run]
207
+ });
208
+ else bucket.runs.push(run);
209
+ }
210
+ const result = [];
211
+ const ordered = [...buckets.values()].sort((a, b) => a.rotationDeg - b.rotationDeg || Number(a.vertical) - Number(b.vertical));
212
+ for (const bucket of ordered) {
213
+ const entries = bucket.runs.map((run) => project(run, bucket.rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
214
+ for (const line of clusterIntoLines(entries, toleranceEm)) {
215
+ line.entries.sort((a, b) => a.alongPt - b.alongPt);
216
+ result.push(buildLine(line, bucket.rotationDeg, bucket.vertical, wordGapEm, columnGapEm));
217
+ }
218
+ }
219
+ return result;
220
+ }
221
+ //#endregion
222
+ exports.DEFAULT_BASELINE_TOLERANCE_EM = DEFAULT_BASELINE_TOLERANCE_EM;
223
+ exports.DEFAULT_COLUMN_GAP_EM = DEFAULT_COLUMN_GAP_EM;
224
+ exports.DEFAULT_WORD_GAP_EM = DEFAULT_WORD_GAP_EM;
225
+ exports.groupPdfTextRuns = groupPdfTextRuns;
226
+ exports.runGapPt = runGapPt;
227
+ exports.runsShareBaseline = runsShareBaseline;
@@ -0,0 +1,75 @@
1
+ //#region src/text-group.d.ts
2
+ interface PdfTextRunGeometry {
3
+ readonly text: string;
4
+ readonly xPt: number;
5
+ readonly yPt: number;
6
+ readonly sizePt: number;
7
+ readonly widthPt?: number;
8
+ readonly rotationDeg?: number;
9
+ readonly writingMode?: "vertical";
10
+ }
11
+ interface PdfTextBox {
12
+ readonly xPt: number;
13
+ readonly yPt: number;
14
+ readonly widthPt: number;
15
+ readonly heightPt: number;
16
+ }
17
+ type PdfWordSeparator = "none" | "space" | "column";
18
+ interface PdfTextWord<TRun extends PdfTextRunGeometry> {
19
+ readonly text: string;
20
+ readonly separatorBefore: PdfWordSeparator;
21
+ readonly runs: readonly TRun[];
22
+ readonly bounds?: PdfTextBox;
23
+ }
24
+ interface PdfTextLine<TRun extends PdfTextRunGeometry> {
25
+ readonly text: string;
26
+ readonly words: readonly PdfTextWord<TRun>[];
27
+ readonly baselineYPt: number;
28
+ readonly rotationDeg: number;
29
+ readonly writingMode?: "vertical";
30
+ readonly bounds?: PdfTextBox;
31
+ }
32
+ interface PdfTextGroupingOptions {
33
+ readonly baselineToleranceEm?: number;
34
+ readonly wordGapEm?: number;
35
+ readonly columnGapEm?: number;
36
+ }
37
+ declare const DEFAULT_BASELINE_TOLERANCE_EM: number;
38
+ declare const DEFAULT_WORD_GAP_EM: number;
39
+ declare const DEFAULT_COLUMN_GAP_EM = 1;
40
+ /**
41
+ * Whether two positioned runs sit on one visual line.
42
+ *
43
+ * The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
44
+ * @param a - one run
45
+ * @param b - the other run
46
+ * @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
47
+ * @returns true when the two baselines are within tolerance of each other
48
+ */
49
+ declare function runsShareBaseline(a: PdfTextRunGeometry, b: PdfTextRunGeometry, options?: PdfTextGroupingOptions): boolean;
50
+ /**
51
+ * The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
52
+ *
53
+ * Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
54
+ * @param previous - the run to the left, in reading order
55
+ * @param next - the run that follows it
56
+ * @returns the gap in points, or undefined when it cannot be derived
57
+ */
58
+ declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry): number | undefined;
59
+ /**
60
+ * Groups positioned PDF text runs into lines of words.
61
+ *
62
+ * Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
63
+ *
64
+ * Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
65
+ *
66
+ * Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
67
+ *
68
+ * Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
69
+ * @param runs - positioned text runs, in any order
70
+ * @param options - tolerance overrides; each defaults to the exported constant of the same name
71
+ * @returns the grouped lines, in reading order
72
+ */
73
+ declare function groupPdfTextRuns<TRun extends PdfTextRunGeometry>(runs: readonly TRun[], options?: PdfTextGroupingOptions): PdfTextLine<TRun>[];
74
+ //#endregion
75
+ export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline };