pdf-codec.js 5.0.1 → 5.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -0
- package/dist/codec.d.cts +1 -0
- package/dist/codec.d.ts +1 -0
- package/dist/font-read.cjs +94 -12
- package/dist/font-read.d.cts +7 -1
- package/dist/font-read.d.ts +7 -1
- package/dist/font-read.js +94 -12
- package/dist/index.cjs +7 -0
- package/dist/index.d.cts +2 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/interpret.cjs +23 -7
- package/dist/interpret.d.cts +9 -1
- package/dist/interpret.d.ts +9 -1
- package/dist/interpret.js +23 -7
- package/dist/layout.cjs +1 -0
- package/dist/layout.d.cts +4 -0
- package/dist/layout.d.ts +4 -0
- package/dist/layout.js +1 -0
- package/dist/raster.cjs +14 -8
- package/dist/raster.js +14 -8
- package/dist/read.cjs +1 -0
- package/dist/read.js +1 -0
- package/dist/text-group.cjs +227 -0
- package/dist/text-group.d.cts +75 -0
- package/dist/text-group.d.ts +75 -0
- package/dist/text-group.js +221 -0
- package/package.json +1 -1
package/dist/interpret.js
CHANGED
|
@@ -432,24 +432,38 @@ function runContentStream(bytes, resources, initialState, context, items, depth,
|
|
|
432
432
|
const widthPer1000 = glyph?.widthPer1000 ?? FALLBACK_GLYPH_WIDTH_PER_1000;
|
|
433
433
|
const byteLength = glyph?.byteLengthConsumed ?? 1;
|
|
434
434
|
const isSingleByteSpace = byteLength === 1 && codes[offset] === 32;
|
|
435
|
-
const
|
|
436
|
-
|
|
435
|
+
const spacing = gs.charSpace + (isSingleByteSpace ? gs.wordSpace : 0);
|
|
436
|
+
const vertical = glyph?.vertical;
|
|
437
|
+
text.tm = vertical === void 0 ? multiplyMatrices(translationMatrix((widthPer1000 / 1e3 * gs.fontSizePt + spacing) * gs.horizScale, 0), text.tm) : multiplyMatrices(translationMatrix(0, vertical.displacementPer1000 / 1e3 * gs.fontSizePt + spacing), text.tm);
|
|
437
438
|
offset += byteLength;
|
|
438
439
|
}
|
|
439
440
|
};
|
|
441
|
+
const positionVectorOf = (codes) => {
|
|
442
|
+
const fontResourceName = gs.fontResourceName;
|
|
443
|
+
if (fontResourceName === void 0) return;
|
|
444
|
+
const vertical = context.fontMetrics.glyphAdvance(fontResourceName, resources, codes, 0)?.vertical;
|
|
445
|
+
if (vertical === void 0) return;
|
|
446
|
+
return {
|
|
447
|
+
x: vertical.positionXPer1000 / 1e3,
|
|
448
|
+
y: vertical.positionYPer1000 / 1e3
|
|
449
|
+
};
|
|
450
|
+
};
|
|
440
451
|
const showTextArray = (elements) => {
|
|
441
452
|
const fontResourceName = gs.fontResourceName;
|
|
442
453
|
if (fontResourceName === void 0) return;
|
|
454
|
+
const vertical = context.fontMetrics.isVertical(fontResourceName, resources);
|
|
443
455
|
const startMatrix = computeTrm(gs, text);
|
|
444
456
|
const chunks = [];
|
|
445
457
|
let totalLength = 0;
|
|
458
|
+
let firstGlyphPosition;
|
|
446
459
|
for (const el of elements) if (el.kind === "string") {
|
|
460
|
+
firstGlyphPosition ??= vertical ? positionVectorOf(el.bytes) : void 0;
|
|
447
461
|
chunks.push(el.bytes);
|
|
448
462
|
totalLength += el.bytes.length;
|
|
449
463
|
advanceThroughString(el.bytes);
|
|
450
464
|
} else if (el.kind === "number") {
|
|
451
|
-
const adjustment = -(el.value / 1e3) * gs.fontSizePt
|
|
452
|
-
text.tm = multiplyMatrices(translationMatrix(adjustment, 0), text.tm);
|
|
465
|
+
const adjustment = -(el.value / 1e3) * gs.fontSizePt;
|
|
466
|
+
text.tm = multiplyMatrices(vertical ? translationMatrix(0, adjustment) : translationMatrix(adjustment * gs.horizScale, 0), text.tm);
|
|
453
467
|
}
|
|
454
468
|
if (totalLength === 0) return;
|
|
455
469
|
const combined = new Uint8Array(totalLength);
|
|
@@ -459,15 +473,17 @@ function runContentStream(bytes, resources, initialState, context, items, depth,
|
|
|
459
473
|
at += chunk.length;
|
|
460
474
|
}
|
|
461
475
|
const endMatrix = computeTrm(gs, text);
|
|
476
|
+
const shift = firstGlyphPosition === void 0 ? void 0 : translationMatrix(-firstGlyphPosition.x, -firstGlyphPosition.y);
|
|
462
477
|
pushItem({
|
|
463
478
|
kind: "text",
|
|
464
479
|
codes: combined,
|
|
465
480
|
fontResourceName,
|
|
466
481
|
resources,
|
|
467
|
-
startMatrix,
|
|
468
|
-
endMatrix,
|
|
482
|
+
startMatrix: shift === void 0 ? startMatrix : multiplyMatrices(shift, startMatrix),
|
|
483
|
+
endMatrix: shift === void 0 ? endMatrix : multiplyMatrices(shift, endMatrix),
|
|
469
484
|
sizePt: gs.fontSizePt,
|
|
470
|
-
color: gs.fillColor
|
|
485
|
+
color: gs.fillColor,
|
|
486
|
+
...vertical ? { vertical: true } : {}
|
|
471
487
|
});
|
|
472
488
|
};
|
|
473
489
|
const showText = (bytes) => {
|
package/dist/layout.cjs
CHANGED
|
@@ -13,6 +13,7 @@ const LayoutTextSchema = zod.z.object({
|
|
|
13
13
|
color: document_schema_js.ColorSchema,
|
|
14
14
|
widthPt: zod.z.number().nonnegative().optional(),
|
|
15
15
|
rotationDeg: zod.z.number().optional(),
|
|
16
|
+
writingMode: zod.z.literal("vertical").optional(),
|
|
16
17
|
underline: zod.z.boolean().optional(),
|
|
17
18
|
layer: zod.z.string().optional(),
|
|
18
19
|
actualText: zod.z.string().optional(),
|
package/dist/layout.d.cts
CHANGED
|
@@ -25,6 +25,7 @@ declare const LayoutTextSchema: z.ZodObject<{
|
|
|
25
25
|
}, z.core.$strip>;
|
|
26
26
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
27
27
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
28
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
28
29
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
29
30
|
layer: z.ZodOptional<z.ZodString>;
|
|
30
31
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -243,6 +244,7 @@ declare const LayoutItemSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
|
243
244
|
}, z.core.$strip>;
|
|
244
245
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
245
246
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
247
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
246
248
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
247
249
|
layer: z.ZodOptional<z.ZodString>;
|
|
248
250
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -475,6 +477,7 @@ declare const LayoutPageSchema: z.ZodObject<{
|
|
|
475
477
|
}, z.core.$strip>;
|
|
476
478
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
477
479
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
480
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
478
481
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
479
482
|
layer: z.ZodOptional<z.ZodString>;
|
|
480
483
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -825,6 +828,7 @@ declare const LayoutDocumentSchema: z.ZodObject<{
|
|
|
825
828
|
}, z.core.$strip>;
|
|
826
829
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
827
830
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
831
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
828
832
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
829
833
|
layer: z.ZodOptional<z.ZodString>;
|
|
830
834
|
actualText: z.ZodOptional<z.ZodString>;
|
package/dist/layout.d.ts
CHANGED
|
@@ -25,6 +25,7 @@ declare const LayoutTextSchema: z.ZodObject<{
|
|
|
25
25
|
}, z.core.$strip>;
|
|
26
26
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
27
27
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
28
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
28
29
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
29
30
|
layer: z.ZodOptional<z.ZodString>;
|
|
30
31
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -243,6 +244,7 @@ declare const LayoutItemSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
|
|
|
243
244
|
}, z.core.$strip>;
|
|
244
245
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
245
246
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
247
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
246
248
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
247
249
|
layer: z.ZodOptional<z.ZodString>;
|
|
248
250
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -475,6 +477,7 @@ declare const LayoutPageSchema: z.ZodObject<{
|
|
|
475
477
|
}, z.core.$strip>;
|
|
476
478
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
477
479
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
480
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
478
481
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
479
482
|
layer: z.ZodOptional<z.ZodString>;
|
|
480
483
|
actualText: z.ZodOptional<z.ZodString>;
|
|
@@ -825,6 +828,7 @@ declare const LayoutDocumentSchema: z.ZodObject<{
|
|
|
825
828
|
}, z.core.$strip>;
|
|
826
829
|
widthPt: z.ZodOptional<z.ZodNumber>;
|
|
827
830
|
rotationDeg: z.ZodOptional<z.ZodNumber>;
|
|
831
|
+
writingMode: z.ZodOptional<z.ZodLiteral<"vertical">>;
|
|
828
832
|
underline: z.ZodOptional<z.ZodBoolean>;
|
|
829
833
|
layer: z.ZodOptional<z.ZodString>;
|
|
830
834
|
actualText: z.ZodOptional<z.ZodString>;
|
package/dist/layout.js
CHANGED
|
@@ -12,6 +12,7 @@ const LayoutTextSchema = z.object({
|
|
|
12
12
|
color: ColorSchema,
|
|
13
13
|
widthPt: z.number().nonnegative().optional(),
|
|
14
14
|
rotationDeg: z.number().optional(),
|
|
15
|
+
writingMode: z.literal("vertical").optional(),
|
|
15
16
|
underline: z.boolean().optional(),
|
|
16
17
|
layer: z.string().optional(),
|
|
17
18
|
actualText: z.string().optional(),
|
package/dist/raster.cjs
CHANGED
|
@@ -533,8 +533,8 @@ function buildTextOutlineFace(fontDict, resolver, fontResolver, fontResourceName
|
|
|
533
533
|
const subtype = require_objects.asName(require_objects.dictGet(fontDict, "Subtype"));
|
|
534
534
|
if (subtype === "Type0") {
|
|
535
535
|
const encoding = resolver.resolve(require_objects.dictGet(fontDict, "Encoding"));
|
|
536
|
-
if (encoding?.kind !== "name" || encoding.name !== "Identity-H") {
|
|
537
|
-
unavailable("a Type0 font whose /Encoding is
|
|
536
|
+
if (encoding?.kind !== "name" || encoding.name !== "Identity-H" && encoding.name !== "Identity-V") {
|
|
537
|
+
unavailable("a Type0 font whose /Encoding is neither Identity-H nor Identity-V (a predefined or embedded CMap this raster walk does not decode)");
|
|
538
538
|
return;
|
|
539
539
|
}
|
|
540
540
|
const descendants = require_objects.asArray(require_objects.dictGet(fontDict, "DescendantFonts"));
|
|
@@ -623,17 +623,19 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
|
|
|
623
623
|
let cumulative = 0;
|
|
624
624
|
while (offset < item.codes.length) {
|
|
625
625
|
const advance = fontResolver.metrics.glyphAdvance(item.fontResourceName, item.resources, item.codes, offset);
|
|
626
|
-
const widthPer1000 = advance.widthPer1000;
|
|
627
626
|
const byteLength = advance.byteLengthConsumed;
|
|
628
627
|
placements.push({
|
|
629
628
|
glyphId: face.glyphIdOf(item.codes, offset),
|
|
630
|
-
advance: cumulative
|
|
629
|
+
advance: cumulative,
|
|
630
|
+
positionX: (advance.vertical?.positionXPer1000 ?? 0) / 1e3,
|
|
631
|
+
positionY: (advance.vertical?.positionYPer1000 ?? 0) / 1e3
|
|
631
632
|
});
|
|
632
|
-
cumulative += widthPer1000 / 1e3 * item.sizePt;
|
|
633
|
+
cumulative += (advance.vertical?.displacementPer1000 ?? advance.widthPer1000) / 1e3 * item.sizePt;
|
|
633
634
|
offset += byteLength;
|
|
634
635
|
}
|
|
635
636
|
const startDeviceMatrix = require_matrix.multiplyMatrices(item.startMatrix, interpretToDeviceMatrix);
|
|
636
637
|
const endDeviceMatrix = require_matrix.multiplyMatrices(item.endMatrix, interpretToDeviceMatrix);
|
|
638
|
+
const vertical = item.vertical === true;
|
|
637
639
|
const startPoint = require_matrix.applyMatrix(startDeviceMatrix, {
|
|
638
640
|
x: 0,
|
|
639
641
|
y: 0
|
|
@@ -642,19 +644,23 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
|
|
|
642
644
|
x: 0,
|
|
643
645
|
y: 0
|
|
644
646
|
});
|
|
645
|
-
const rawEndPoint = require_matrix.applyMatrix(startDeviceMatrix, {
|
|
647
|
+
const rawEndPoint = require_matrix.applyMatrix(startDeviceMatrix, vertical ? {
|
|
648
|
+
x: 0,
|
|
649
|
+
y: cumulative
|
|
650
|
+
} : {
|
|
646
651
|
x: cumulative,
|
|
647
652
|
y: 0
|
|
648
653
|
});
|
|
649
654
|
const rawExtent = Math.hypot(rawEndPoint.x - startPoint.x, rawEndPoint.y - startPoint.y);
|
|
650
655
|
const actualExtent = Math.hypot(endPoint.x - startPoint.x, endPoint.y - startPoint.y);
|
|
651
|
-
const correction = rawExtent > 1e-9
|
|
656
|
+
const correction = rawExtent > 1e-9 ? actualExtent / rawExtent : 1;
|
|
657
|
+
const firstPlacement = placements[0];
|
|
652
658
|
const glyphScale = require_matrix.scaleMatrix(1 / face.unitsPerEm, 1 / face.unitsPerEm);
|
|
653
659
|
for (const placement of placements) {
|
|
654
660
|
if (placement.glyphId === void 0) continue;
|
|
655
661
|
const outline = require_glyf_contours.decodeGlyphOutline(face.glyf, placement.glyphId);
|
|
656
662
|
if (outline === void 0) continue;
|
|
657
|
-
const trm = require_matrix.multiplyMatrices(require_matrix.translationMatrix(placement.advance * correction, 0), item.startMatrix);
|
|
663
|
+
const trm = require_matrix.multiplyMatrices(require_matrix.translationMatrix((firstPlacement?.positionX ?? 0) - placement.positionX, (firstPlacement?.positionY ?? 0) - placement.positionY), require_matrix.multiplyMatrices(vertical ? require_matrix.translationMatrix(0, placement.advance * correction) : require_matrix.translationMatrix(placement.advance * correction, 0), item.startMatrix));
|
|
658
664
|
drawGlyphOutline(outline, require_matrix.multiplyMatrices(glyphScale, require_matrix.multiplyMatrices(trm, interpretToDeviceMatrix)), item.color, rasteriser);
|
|
659
665
|
}
|
|
660
666
|
}
|
package/dist/raster.js
CHANGED
|
@@ -532,8 +532,8 @@ function buildTextOutlineFace(fontDict, resolver, fontResolver, fontResourceName
|
|
|
532
532
|
const subtype = asName(dictGet(fontDict, "Subtype"));
|
|
533
533
|
if (subtype === "Type0") {
|
|
534
534
|
const encoding = resolver.resolve(dictGet(fontDict, "Encoding"));
|
|
535
|
-
if (encoding?.kind !== "name" || encoding.name !== "Identity-H") {
|
|
536
|
-
unavailable("a Type0 font whose /Encoding is
|
|
535
|
+
if (encoding?.kind !== "name" || encoding.name !== "Identity-H" && encoding.name !== "Identity-V") {
|
|
536
|
+
unavailable("a Type0 font whose /Encoding is neither Identity-H nor Identity-V (a predefined or embedded CMap this raster walk does not decode)");
|
|
537
537
|
return;
|
|
538
538
|
}
|
|
539
539
|
const descendants = asArray(dictGet(fontDict, "DescendantFonts"));
|
|
@@ -622,17 +622,19 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
|
|
|
622
622
|
let cumulative = 0;
|
|
623
623
|
while (offset < item.codes.length) {
|
|
624
624
|
const advance = fontResolver.metrics.glyphAdvance(item.fontResourceName, item.resources, item.codes, offset);
|
|
625
|
-
const widthPer1000 = advance.widthPer1000;
|
|
626
625
|
const byteLength = advance.byteLengthConsumed;
|
|
627
626
|
placements.push({
|
|
628
627
|
glyphId: face.glyphIdOf(item.codes, offset),
|
|
629
|
-
advance: cumulative
|
|
628
|
+
advance: cumulative,
|
|
629
|
+
positionX: (advance.vertical?.positionXPer1000 ?? 0) / 1e3,
|
|
630
|
+
positionY: (advance.vertical?.positionYPer1000 ?? 0) / 1e3
|
|
630
631
|
});
|
|
631
|
-
cumulative += widthPer1000 / 1e3 * item.sizePt;
|
|
632
|
+
cumulative += (advance.vertical?.displacementPer1000 ?? advance.widthPer1000) / 1e3 * item.sizePt;
|
|
632
633
|
offset += byteLength;
|
|
633
634
|
}
|
|
634
635
|
const startDeviceMatrix = multiplyMatrices(item.startMatrix, interpretToDeviceMatrix);
|
|
635
636
|
const endDeviceMatrix = multiplyMatrices(item.endMatrix, interpretToDeviceMatrix);
|
|
637
|
+
const vertical = item.vertical === true;
|
|
636
638
|
const startPoint = applyMatrix(startDeviceMatrix, {
|
|
637
639
|
x: 0,
|
|
638
640
|
y: 0
|
|
@@ -641,19 +643,23 @@ function drawTextRun(item, fontResolver, resolver, outlineFaces, interpretToDevi
|
|
|
641
643
|
x: 0,
|
|
642
644
|
y: 0
|
|
643
645
|
});
|
|
644
|
-
const rawEndPoint = applyMatrix(startDeviceMatrix, {
|
|
646
|
+
const rawEndPoint = applyMatrix(startDeviceMatrix, vertical ? {
|
|
647
|
+
x: 0,
|
|
648
|
+
y: cumulative
|
|
649
|
+
} : {
|
|
645
650
|
x: cumulative,
|
|
646
651
|
y: 0
|
|
647
652
|
});
|
|
648
653
|
const rawExtent = Math.hypot(rawEndPoint.x - startPoint.x, rawEndPoint.y - startPoint.y);
|
|
649
654
|
const actualExtent = Math.hypot(endPoint.x - startPoint.x, endPoint.y - startPoint.y);
|
|
650
|
-
const correction = rawExtent > 1e-9
|
|
655
|
+
const correction = rawExtent > 1e-9 ? actualExtent / rawExtent : 1;
|
|
656
|
+
const firstPlacement = placements[0];
|
|
651
657
|
const glyphScale = scaleMatrix(1 / face.unitsPerEm, 1 / face.unitsPerEm);
|
|
652
658
|
for (const placement of placements) {
|
|
653
659
|
if (placement.glyphId === void 0) continue;
|
|
654
660
|
const outline = decodeGlyphOutline(face.glyf, placement.glyphId);
|
|
655
661
|
if (outline === void 0) continue;
|
|
656
|
-
const trm = multiplyMatrices(translationMatrix(placement.advance * correction, 0), item.startMatrix);
|
|
662
|
+
const trm = multiplyMatrices(translationMatrix((firstPlacement?.positionX ?? 0) - placement.positionX, (firstPlacement?.positionY ?? 0) - placement.positionY), multiplyMatrices(vertical ? translationMatrix(0, placement.advance * correction) : translationMatrix(placement.advance * correction, 0), item.startMatrix));
|
|
657
663
|
drawGlyphOutline(outline, multiplyMatrices(glyphScale, multiplyMatrices(trm, interpretToDeviceMatrix)), item.color, rasteriser);
|
|
658
664
|
}
|
|
659
665
|
}
|
package/dist/read.cjs
CHANGED
|
@@ -366,6 +366,7 @@ function convertText(item, pageMatrix, fontResolver) {
|
|
|
366
366
|
color: item.color,
|
|
367
367
|
widthPt,
|
|
368
368
|
rotationDeg: rotationDeg !== 0 ? rotationDeg : void 0,
|
|
369
|
+
...item.vertical === true ? { writingMode: "vertical" } : {},
|
|
369
370
|
...item.layerName !== void 0 ? { layer: item.layerName } : {},
|
|
370
371
|
...item.actualText !== void 0 ? { actualText: item.actualText } : {},
|
|
371
372
|
...item.alt !== void 0 ? { alt: item.alt } : {}
|
package/dist/read.js
CHANGED
|
@@ -365,6 +365,7 @@ function convertText(item, pageMatrix, fontResolver) {
|
|
|
365
365
|
color: item.color,
|
|
366
366
|
widthPt,
|
|
367
367
|
rotationDeg: rotationDeg !== 0 ? rotationDeg : void 0,
|
|
368
|
+
...item.vertical === true ? { writingMode: "vertical" } : {},
|
|
368
369
|
...item.layerName !== void 0 ? { layer: item.layerName } : {},
|
|
369
370
|
...item.actualText !== void 0 ? { actualText: item.actualText } : {},
|
|
370
371
|
...item.alt !== void 0 ? { alt: item.alt } : {}
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
|
|
2
|
+
//#region src/text-group.ts
|
|
3
|
+
const DEFAULT_BASELINE_TOLERANCE_EM = 2 / 3;
|
|
4
|
+
const DEFAULT_WORD_GAP_EM = 1 / 8;
|
|
5
|
+
const DEFAULT_COLUMN_GAP_EM = 1;
|
|
6
|
+
const ROTATION_NOISE_DEG = 1e-6;
|
|
7
|
+
const DEGREES_PER_TURN = 360;
|
|
8
|
+
const VERTICAL_ADVANCE_OFFSET_DEG = -90;
|
|
9
|
+
function advanceAngleDeg(run) {
|
|
10
|
+
const raw = (run.rotationDeg ?? 0) + (run.writingMode === "vertical" ? VERTICAL_ADVANCE_OFFSET_DEG : 0);
|
|
11
|
+
return (Math.round(raw / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
|
|
12
|
+
}
|
|
13
|
+
const RADIANS_PER_TURN = 2 * Math.PI;
|
|
14
|
+
function baselineAxis(rotationDeg) {
|
|
15
|
+
const radians = rotationDeg * RADIANS_PER_TURN / DEGREES_PER_TURN;
|
|
16
|
+
return {
|
|
17
|
+
cos: Math.cos(radians),
|
|
18
|
+
sin: Math.sin(radians)
|
|
19
|
+
};
|
|
20
|
+
}
|
|
21
|
+
function project(run, rotationDeg) {
|
|
22
|
+
const { cos, sin } = baselineAxis(rotationDeg);
|
|
23
|
+
return {
|
|
24
|
+
run,
|
|
25
|
+
alongPt: run.xPt * cos + run.yPt * sin,
|
|
26
|
+
acrossPt: run.yPt * cos - run.xPt * sin,
|
|
27
|
+
sizePt: run.sizePt
|
|
28
|
+
};
|
|
29
|
+
}
|
|
30
|
+
function unproject(alongPt, acrossPt, rotationDeg) {
|
|
31
|
+
const { cos, sin } = baselineAxis(rotationDeg);
|
|
32
|
+
return {
|
|
33
|
+
xPt: alongPt * cos - acrossPt * sin,
|
|
34
|
+
yPt: alongPt * sin + acrossPt * cos
|
|
35
|
+
};
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Whether two positioned runs sit on one visual line.
|
|
39
|
+
*
|
|
40
|
+
* The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
|
|
41
|
+
* @param a - one run
|
|
42
|
+
* @param b - the other run
|
|
43
|
+
* @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
|
|
44
|
+
* @returns true when the two baselines are within tolerance of each other
|
|
45
|
+
*/
|
|
46
|
+
function runsShareBaseline(a, b, options = {}) {
|
|
47
|
+
const rotationDeg = advanceAngleDeg(a);
|
|
48
|
+
if (rotationDeg !== advanceAngleDeg(b) || a.writingMode !== b.writingMode) return false;
|
|
49
|
+
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
50
|
+
return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
|
|
51
|
+
}
|
|
52
|
+
/**
|
|
53
|
+
* The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
|
|
54
|
+
*
|
|
55
|
+
* Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
|
|
56
|
+
* @param previous - the run to the left, in reading order
|
|
57
|
+
* @param next - the run that follows it
|
|
58
|
+
* @returns the gap in points, or undefined when it cannot be derived
|
|
59
|
+
*/
|
|
60
|
+
function runGapPt(previous, next) {
|
|
61
|
+
const rotationDeg = advanceAngleDeg(previous);
|
|
62
|
+
if (rotationDeg !== advanceAngleDeg(next) || previous.writingMode !== next.writingMode) return;
|
|
63
|
+
const previousWidthPt = previous.widthPt;
|
|
64
|
+
if (previousWidthPt === void 0) return;
|
|
65
|
+
return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
|
|
66
|
+
}
|
|
67
|
+
function clusterIntoLines(entries, toleranceEm) {
|
|
68
|
+
const lines = [];
|
|
69
|
+
for (const entry of entries) {
|
|
70
|
+
let nearest;
|
|
71
|
+
for (const line of lines) if (runsShareBaseline(line.anchor.run, entry.run, { baselineToleranceEm: toleranceEm })) nearest = line;
|
|
72
|
+
if (nearest === void 0) lines.push({
|
|
73
|
+
anchor: entry,
|
|
74
|
+
entries: [entry]
|
|
75
|
+
});
|
|
76
|
+
else nearest.entries.push(entry);
|
|
77
|
+
}
|
|
78
|
+
return lines;
|
|
79
|
+
}
|
|
80
|
+
function tokeniseRun(text) {
|
|
81
|
+
const words = text.match(/\S+/g) ?? [];
|
|
82
|
+
const leadingSpace = /^\s/.test(text);
|
|
83
|
+
const trailingSpace = /\s$/.test(text);
|
|
84
|
+
return {
|
|
85
|
+
words,
|
|
86
|
+
leadingSpace,
|
|
87
|
+
trailingSpace,
|
|
88
|
+
whole: words.length === 1 && !leadingSpace && !trailingSpace
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
const SEPARATOR_TEXT = {
|
|
92
|
+
none: "",
|
|
93
|
+
space: " ",
|
|
94
|
+
column: " "
|
|
95
|
+
};
|
|
96
|
+
function separatorForGap(gapPt, smallerSizePt, wordGapEm, columnGapEm) {
|
|
97
|
+
if (gapPt === void 0) return "none";
|
|
98
|
+
if (gapPt >= columnGapEm * smallerSizePt) return "column";
|
|
99
|
+
if (gapPt >= wordGapEm * smallerSizePt) return "space";
|
|
100
|
+
return "none";
|
|
101
|
+
}
|
|
102
|
+
function boundsOf(minAlongPt, maxAlongEndPt, acrossPt, maxSizePt, rotationDeg) {
|
|
103
|
+
const origin = unproject(minAlongPt, acrossPt, rotationDeg);
|
|
104
|
+
return {
|
|
105
|
+
xPt: origin.xPt,
|
|
106
|
+
yPt: origin.yPt,
|
|
107
|
+
widthPt: maxAlongEndPt - minAlongPt,
|
|
108
|
+
heightPt: maxSizePt
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
112
|
+
const working = [];
|
|
113
|
+
let previous;
|
|
114
|
+
for (const entry of line.entries) {
|
|
115
|
+
const tokens = tokeniseRun(entry.run.text);
|
|
116
|
+
const gapSeparator = previous === void 0 ? "none" : separatorForGap(runGapPt(previous.entry.run, entry.run), Math.min(previous.entry.sizePt, entry.sizePt), wordGapEm, columnGapEm);
|
|
117
|
+
const pendingSpace = previous?.trailingSpace === true;
|
|
118
|
+
let separator = "none";
|
|
119
|
+
if (working.length > 0) separator = gapSeparator === "none" && (pendingSpace || tokens.leadingSpace) ? "space" : gapSeparator;
|
|
120
|
+
const widthPt = entry.run.widthPt;
|
|
121
|
+
const alongEndPt = entry.alongPt + (widthPt ?? 0);
|
|
122
|
+
tokens.words.forEach((word, index) => {
|
|
123
|
+
const last = working[working.length - 1];
|
|
124
|
+
if (index === 0 && separator === "none" && last !== void 0) {
|
|
125
|
+
last.text += word;
|
|
126
|
+
last.runs.push(entry.run);
|
|
127
|
+
last.measurable = last.measurable && tokens.whole && widthPt !== void 0;
|
|
128
|
+
last.minAlongPt = Math.min(last.minAlongPt, entry.alongPt);
|
|
129
|
+
last.maxAlongEndPt = Math.max(last.maxAlongEndPt, alongEndPt);
|
|
130
|
+
last.maxSizePt = Math.max(last.maxSizePt, entry.sizePt);
|
|
131
|
+
return;
|
|
132
|
+
}
|
|
133
|
+
working.push({
|
|
134
|
+
text: word,
|
|
135
|
+
separatorBefore: index === 0 ? separator : "space",
|
|
136
|
+
runs: [entry.run],
|
|
137
|
+
measurable: tokens.whole && widthPt !== void 0,
|
|
138
|
+
minAlongPt: entry.alongPt,
|
|
139
|
+
maxAlongEndPt: alongEndPt,
|
|
140
|
+
maxSizePt: entry.sizePt
|
|
141
|
+
});
|
|
142
|
+
});
|
|
143
|
+
previous = {
|
|
144
|
+
entry,
|
|
145
|
+
trailingSpace: tokens.trailingSpace
|
|
146
|
+
};
|
|
147
|
+
}
|
|
148
|
+
return working.map((word) => ({
|
|
149
|
+
text: word.text,
|
|
150
|
+
separatorBefore: word.separatorBefore,
|
|
151
|
+
runs: word.runs,
|
|
152
|
+
...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
|
|
153
|
+
}));
|
|
154
|
+
}
|
|
155
|
+
function buildLine(line, rotationDeg, vertical, wordGapEm, columnGapEm) {
|
|
156
|
+
const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
|
|
157
|
+
const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
|
|
158
|
+
let minAlongPt = Number.POSITIVE_INFINITY;
|
|
159
|
+
let maxAlongEndPt = Number.NEGATIVE_INFINITY;
|
|
160
|
+
let maxSizePt = 0;
|
|
161
|
+
let measurable = true;
|
|
162
|
+
for (const entry of line.entries) {
|
|
163
|
+
const widthPt = entry.run.widthPt;
|
|
164
|
+
if (widthPt === void 0) measurable = false;
|
|
165
|
+
minAlongPt = Math.min(minAlongPt, entry.alongPt);
|
|
166
|
+
maxAlongEndPt = Math.max(maxAlongEndPt, entry.alongPt + (widthPt ?? 0));
|
|
167
|
+
maxSizePt = Math.max(maxSizePt, entry.sizePt);
|
|
168
|
+
}
|
|
169
|
+
return {
|
|
170
|
+
text,
|
|
171
|
+
words,
|
|
172
|
+
baselineYPt: line.anchor.acrossPt,
|
|
173
|
+
rotationDeg,
|
|
174
|
+
...vertical ? { writingMode: "vertical" } : {},
|
|
175
|
+
...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
|
|
176
|
+
};
|
|
177
|
+
}
|
|
178
|
+
/**
|
|
179
|
+
* Groups positioned PDF text runs into lines of words.
|
|
180
|
+
*
|
|
181
|
+
* Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
|
|
182
|
+
*
|
|
183
|
+
* Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
|
|
184
|
+
*
|
|
185
|
+
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
186
|
+
*
|
|
187
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
188
|
+
* @param runs - positioned text runs, in any order
|
|
189
|
+
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
190
|
+
* @returns the grouped lines, in reading order
|
|
191
|
+
*/
|
|
192
|
+
function groupPdfTextRuns(runs, options = {}) {
|
|
193
|
+
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
194
|
+
const wordGapEm = options.wordGapEm ?? .125;
|
|
195
|
+
const columnGapEm = options.columnGapEm ?? 1;
|
|
196
|
+
const buckets = /* @__PURE__ */ new Map();
|
|
197
|
+
for (const run of runs) {
|
|
198
|
+
if (run.text.length === 0) continue;
|
|
199
|
+
const rotationDeg = advanceAngleDeg(run);
|
|
200
|
+
const vertical = run.writingMode === "vertical";
|
|
201
|
+
const key = `${String(rotationDeg)}|${String(vertical)}`;
|
|
202
|
+
const bucket = buckets.get(key);
|
|
203
|
+
if (bucket === void 0) buckets.set(key, {
|
|
204
|
+
rotationDeg,
|
|
205
|
+
vertical,
|
|
206
|
+
runs: [run]
|
|
207
|
+
});
|
|
208
|
+
else bucket.runs.push(run);
|
|
209
|
+
}
|
|
210
|
+
const result = [];
|
|
211
|
+
const ordered = [...buckets.values()].sort((a, b) => a.rotationDeg - b.rotationDeg || Number(a.vertical) - Number(b.vertical));
|
|
212
|
+
for (const bucket of ordered) {
|
|
213
|
+
const entries = bucket.runs.map((run) => project(run, bucket.rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
|
|
214
|
+
for (const line of clusterIntoLines(entries, toleranceEm)) {
|
|
215
|
+
line.entries.sort((a, b) => a.alongPt - b.alongPt);
|
|
216
|
+
result.push(buildLine(line, bucket.rotationDeg, bucket.vertical, wordGapEm, columnGapEm));
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
return result;
|
|
220
|
+
}
|
|
221
|
+
//#endregion
|
|
222
|
+
exports.DEFAULT_BASELINE_TOLERANCE_EM = DEFAULT_BASELINE_TOLERANCE_EM;
|
|
223
|
+
exports.DEFAULT_COLUMN_GAP_EM = DEFAULT_COLUMN_GAP_EM;
|
|
224
|
+
exports.DEFAULT_WORD_GAP_EM = DEFAULT_WORD_GAP_EM;
|
|
225
|
+
exports.groupPdfTextRuns = groupPdfTextRuns;
|
|
226
|
+
exports.runGapPt = runGapPt;
|
|
227
|
+
exports.runsShareBaseline = runsShareBaseline;
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
//#region src/text-group.d.ts
|
|
2
|
+
interface PdfTextRunGeometry {
|
|
3
|
+
readonly text: string;
|
|
4
|
+
readonly xPt: number;
|
|
5
|
+
readonly yPt: number;
|
|
6
|
+
readonly sizePt: number;
|
|
7
|
+
readonly widthPt?: number;
|
|
8
|
+
readonly rotationDeg?: number;
|
|
9
|
+
readonly writingMode?: "vertical";
|
|
10
|
+
}
|
|
11
|
+
interface PdfTextBox {
|
|
12
|
+
readonly xPt: number;
|
|
13
|
+
readonly yPt: number;
|
|
14
|
+
readonly widthPt: number;
|
|
15
|
+
readonly heightPt: number;
|
|
16
|
+
}
|
|
17
|
+
type PdfWordSeparator = "none" | "space" | "column";
|
|
18
|
+
interface PdfTextWord<TRun extends PdfTextRunGeometry> {
|
|
19
|
+
readonly text: string;
|
|
20
|
+
readonly separatorBefore: PdfWordSeparator;
|
|
21
|
+
readonly runs: readonly TRun[];
|
|
22
|
+
readonly bounds?: PdfTextBox;
|
|
23
|
+
}
|
|
24
|
+
interface PdfTextLine<TRun extends PdfTextRunGeometry> {
|
|
25
|
+
readonly text: string;
|
|
26
|
+
readonly words: readonly PdfTextWord<TRun>[];
|
|
27
|
+
readonly baselineYPt: number;
|
|
28
|
+
readonly rotationDeg: number;
|
|
29
|
+
readonly writingMode?: "vertical";
|
|
30
|
+
readonly bounds?: PdfTextBox;
|
|
31
|
+
}
|
|
32
|
+
interface PdfTextGroupingOptions {
|
|
33
|
+
readonly baselineToleranceEm?: number;
|
|
34
|
+
readonly wordGapEm?: number;
|
|
35
|
+
readonly columnGapEm?: number;
|
|
36
|
+
}
|
|
37
|
+
declare const DEFAULT_BASELINE_TOLERANCE_EM: number;
|
|
38
|
+
declare const DEFAULT_WORD_GAP_EM: number;
|
|
39
|
+
declare const DEFAULT_COLUMN_GAP_EM = 1;
|
|
40
|
+
/**
|
|
41
|
+
* Whether two positioned runs sit on one visual line.
|
|
42
|
+
*
|
|
43
|
+
* The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
|
|
44
|
+
* @param a - one run
|
|
45
|
+
* @param b - the other run
|
|
46
|
+
* @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
|
|
47
|
+
* @returns true when the two baselines are within tolerance of each other
|
|
48
|
+
*/
|
|
49
|
+
declare function runsShareBaseline(a: PdfTextRunGeometry, b: PdfTextRunGeometry, options?: PdfTextGroupingOptions): boolean;
|
|
50
|
+
/**
|
|
51
|
+
* The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
|
|
52
|
+
*
|
|
53
|
+
* Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
|
|
54
|
+
* @param previous - the run to the left, in reading order
|
|
55
|
+
* @param next - the run that follows it
|
|
56
|
+
* @returns the gap in points, or undefined when it cannot be derived
|
|
57
|
+
*/
|
|
58
|
+
declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry): number | undefined;
|
|
59
|
+
/**
|
|
60
|
+
* Groups positioned PDF text runs into lines of words.
|
|
61
|
+
*
|
|
62
|
+
* Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
|
|
63
|
+
*
|
|
64
|
+
* Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
|
|
65
|
+
*
|
|
66
|
+
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
67
|
+
*
|
|
68
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
69
|
+
* @param runs - positioned text runs, in any order
|
|
70
|
+
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
71
|
+
* @returns the grouped lines, in reading order
|
|
72
|
+
*/
|
|
73
|
+
declare function groupPdfTextRuns<TRun extends PdfTextRunGeometry>(runs: readonly TRun[], options?: PdfTextGroupingOptions): PdfTextLine<TRun>[];
|
|
74
|
+
//#endregion
|
|
75
|
+
export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline };
|