pdf-codec.js 5.0.0 → 5.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -0
- package/dist/index.cjs +7 -0
- package/dist/index.d.cts +2 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/text-group.cjs +217 -0
- package/dist/text-group.d.cts +73 -0
- package/dist/text-group.d.ts +73 -0
- package/dist/text-group.js +211 -0
- package/package.json +3 -3
package/README.md
CHANGED
|
@@ -310,6 +310,31 @@ Reading a program's built-in encoding is what stops a symbol-encoded subset —
|
|
|
310
310
|
|
|
311
311
|
**Where nothing states an answer, the answer is the replacement character plus a `text/unmapped-encoding` diagnostic, never a guess.** Two cases reach it in practice: a symbolic font with no embedded program and no `/ToUnicode`, and a subsetted font that both strips its glyph names and maps its glyphs only from private-use code points — a private-use code point identifies a glyph inside one font and says nothing about the character it draws, so it is treated as no answer rather than a wrong one.
|
|
312
312
|
|
|
313
|
+
## Grouping text runs into lines and words
|
|
314
|
+
|
|
315
|
+
`readPdf` reports one positioned run per text-showing operator, and a PDF says nothing about lines or words: a line of prose can arrive as one run, as one run per word, or as one run per glyph, entirely at the producer's discretion. `groupPdfTextRuns` turns those runs back into lines of words, and `pdf-codec/text-group` serves it without the write half or the vendored fonts, alongside `pdf-codec/read`.
|
|
316
|
+
|
|
317
|
+
```ts
|
|
318
|
+
import { readPdf } from "pdf-codec/read";
|
|
319
|
+
import { groupPdfTextRuns } from "pdf-codec/text-group";
|
|
320
|
+
|
|
321
|
+
const page = readPdf(bytes).pages[0];
|
|
322
|
+
const runs = page.items.filter((item) => item.kind === "text");
|
|
323
|
+
for (const line of groupPdfTextRuns(runs)) {
|
|
324
|
+
console.log(line.text);
|
|
325
|
+
}
|
|
326
|
+
```
|
|
327
|
+
|
|
328
|
+
Two rules decide everything, and both were got wrong often enough by hand ([#1317](https://github.com/ExaDev/documents.js/issues/1317)) to be worth stating.
|
|
329
|
+
|
|
330
|
+
Two runs are on one line when their baselines sit within a fraction of an em of each other, measured against **the smaller of the two**. Taken from the larger, a 30pt heading's own window reaches the 9pt line beneath it, the merged line's runs sort by x into a sequence whose gaps are negative, a negative gap reads as one word, and the two lines concatenate into exactly the run-together text the grouping exists to prevent. The default fraction is two thirds, bounded on both sides by real typography: two consecutive lines are never closer than solid setting puts them, one full em, so it has to stay under 1; a superscript raised by at most a third of its parent's size at no less than half that size shifts by at most two thirds of its own em, so anything tighter cuts footnote markers off the line they annotate.
|
|
331
|
+
|
|
332
|
+
A run whose `widthPt` is absent states no advance, so where it ends is unknown and the gap after it is underivable. `runGapPt` reports `undefined` for that, and an undefined gap is **no evidence of a space**. Reading it as zero puts the previous run's end at its own start, turns its whole advance into an apparent gap, and produces "Com plete ly".
|
|
333
|
+
|
|
334
|
+
Gaps that are derivable become a word space at an eighth of an em (the narrowest genuine one: the standard fourteen faces set their space glyph at 250/1000 em, and justified setting compresses to no less than half of that) and a column boundary past a full em (wider than the em space, the widest single space character there is), so a table's row reads as `"North\t4.2m\t11%"` rather than as one sentence. `runsShareBaseline` and `runGapPt` are exported on their own for a consumer with its own pipeline around them, and every threshold is overridable.
|
|
335
|
+
|
|
336
|
+
Limits, each inherent to what a PDF states rather than to the implementation: grouping is by baseline alone with no page segmentation, so on a multi-column page whose columns sit at different vertical offsets a line of one can fall within tolerance of a line of the next and the two are reported as one line with a column boundary between them, which is the cut a caller needs (`document-outline.js`'s `segmentPdfRegions` finds the gutter first if you want them genuinely apart); text comes out in visual order, so a right-to-left script needs the Unicode Bidi Algorithm applied to the result; vertical writing modes are not recognised, because the content interpreter does not read a CMap's `WMode` ([#1358](https://github.com/ExaDev/documents.js/issues/1358)); and a word hyphenated across a line end stays split, hyphen intact, because rejoining it is a paragraph-level judgement this package leaves to its consumers.
|
|
337
|
+
|
|
313
338
|
## JBIG2 scope
|
|
314
339
|
|
|
315
340
|
`src/image/jbig2*.ts` is a hand-written ITU-T T.88 decoder covering what real scanned PDFs actually contain.
|
package/dist/index.cjs
CHANGED
|
@@ -19,8 +19,12 @@ const require_measure = require("./measure.cjs");
|
|
|
19
19
|
const require_write = require("./write.cjs");
|
|
20
20
|
const require_codec = require("./codec.cjs");
|
|
21
21
|
const require_font_face = require("./font-face.cjs");
|
|
22
|
+
const require_text_group = require("./text-group.cjs");
|
|
22
23
|
const require_raster = require("./raster.cjs");
|
|
24
|
+
exports.DEFAULT_BASELINE_TOLERANCE_EM = require_text_group.DEFAULT_BASELINE_TOLERANCE_EM;
|
|
25
|
+
exports.DEFAULT_COLUMN_GAP_EM = require_text_group.DEFAULT_COLUMN_GAP_EM;
|
|
23
26
|
exports.DEFAULT_VERTICAL_METRIC_POLICY = require_measure.DEFAULT_VERTICAL_METRIC_POLICY;
|
|
27
|
+
exports.DEFAULT_WORD_GAP_EM = require_text_group.DEFAULT_WORD_GAP_EM;
|
|
24
28
|
exports.FontFaceParseError = require_font_face.FontFaceParseError;
|
|
25
29
|
exports.Jbig2ParseError = require_image_jbig2_errors.Jbig2ParseError;
|
|
26
30
|
exports.Jbig2UnsupportedError = require_image_jbig2_errors.Jbig2UnsupportedError;
|
|
@@ -66,6 +70,7 @@ exports.decodeCcittFax = require_image_ccitt.decodeCcittFax;
|
|
|
66
70
|
exports.decodeJbig2Embedded = require_image_jbig2.decodeJbig2Embedded;
|
|
67
71
|
exports.decodeJpeg2000 = require_image_jpeg2000.decodeJpeg2000;
|
|
68
72
|
exports.filterScanlines = require_image_png_filter.filterScanlines;
|
|
73
|
+
exports.groupPdfTextRuns = require_text_group.groupPdfTextRuns;
|
|
69
74
|
exports.loadMathFont = require_math_font.loadMathFont;
|
|
70
75
|
exports.looksLikeBareCodestream = require_image_jp2_boxes.looksLikeBareCodestream;
|
|
71
76
|
exports.parseJp2Container = require_image_jp2_boxes.parseJp2Container;
|
|
@@ -76,6 +81,8 @@ exports.readPdf = require_read.readPdf;
|
|
|
76
81
|
exports.renderPdfPage = require_raster.renderPdfPage;
|
|
77
82
|
exports.resolveFaceWithRegistry = require_font_registry.resolveFaceWithRegistry;
|
|
78
83
|
exports.resolveStandardFont = require_fonts.resolveStandardFont;
|
|
84
|
+
exports.runGapPt = require_text_group.runGapPt;
|
|
85
|
+
exports.runsShareBaseline = require_text_group.runsShareBaseline;
|
|
79
86
|
exports.scaleMathStretchConstruction = require_math_stretch.scaleMathStretchConstruction;
|
|
80
87
|
exports.unfilterScanlines = require_image_png_filter.unfilterScanlines;
|
|
81
88
|
exports.writePdf = require_write.writePdf;
|
package/dist/index.d.cts
CHANGED
|
@@ -23,7 +23,8 @@ import { a as MathGlyphPart, c as MathVariants, n as MathGlyphAssembly, o as Mat
|
|
|
23
23
|
import { a as scaleMathStretchConstruction, i as assembleStretchyGlyph, n as MathStretchOptions, r as MathStretchPlacement, t as MathStretchConstruction } from "./math-stretch-C2JanDAz.cjs";
|
|
24
24
|
import { LoadedMathFont, MathFont, MathFontDescriptorMetrics, loadMathFont } from "./math-font.cjs";
|
|
25
25
|
import { DEFAULT_VERTICAL_METRIC_POLICY, FontMeasurerOptions, StandardFontMeasurerOptions, VerticalMetricPolicy, createFontMeasurer, createStandardFontMeasurer } from "./measure.cjs";
|
|
26
|
+
import { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline } from "./text-group.cjs";
|
|
26
27
|
import { PageRasteriser, RasterDrawOp, RasterFillRectOp, RasterFillSpec, RasterImageOp, RasterMatrix, RasterPageGeometry, RasterPathOp, RasterPathSegment, RasterRegionPt, RasterStrokeSpec, RasterSubpath, RenderPdfPageOptions, renderPdfPage } from "./raster.cjs";
|
|
27
28
|
import { FontRegistryOptions, FontSubstitution, MathAssembledGlyphs, MathBox, MathColor, MathFontMetrics, MathGlyphMetrics, MathGlyphPlacement, MathGlyphRun, MathLayoutItem, MathRule, MathStretchAxis, MathStretchGlyph, MathStretchResult, MathStroke, Point, PositionedFormula, ProvidedFont, StyledFragment, StyledRun, TextMeasurer, UnderlineMetrics, WrapOptions, WrappedLine } from "document-schema.js";
|
|
28
29
|
export * from "byte-codec";
|
|
29
|
-
export { type CcittFaxImage, type CcittFaxOptions, DEFAULT_VERTICAL_METRIC_POLICY, type EmbeddedFace, type EmbeddedFaceMetrics, type EmbeddedFaceSubstitution, type FontFace, FontFaceParseError, type FontMeasurerOptions, type FontMetrics, type FontRegistry, type FontRegistryOptions, type FontSubstitution, type GlyphInkBounds, type Jbig2DecodeOptions, type Jbig2Image, Jbig2ParseError, Jbig2UnsupportedError, type Jp2ChannelDefinition, type Jp2ColourSpace, type Jp2Container, type Jp2ImageHeader, type Jpeg2000ComponentMetadata, type Jpeg2000DecodeOptions, type Jpeg2000Image, type Jpeg2000Metadata, Jpeg2000ParseError, type Jpeg2000ProgressionOrder, type Jpeg2000QuantizationStyle, type Jpeg2000Transform, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotation, LayoutAnnotationQuad, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachment, LayoutAttachmentSchema, LayoutDestination, LayoutDestinationSchema, LayoutDestinationTarget, LayoutDestinationTargetSchema, LayoutDocument, LayoutDocumentSchema, LayoutEllipse, LayoutEllipseSchema, LayoutFormField, LayoutFormFieldSchema, LayoutFormWidget, LayoutFormWidgetSchema, LayoutImage, LayoutImageAsset, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLink, LayoutInternalLinkSchema, LayoutItem, LayoutItemSchema, LayoutLayer, LayoutLayerSchema, LayoutLine, LayoutLineSchema, LayoutLink, LayoutLinkSchema, LayoutOutlineItem, LayoutOutlineItemSchema, LayoutPage, LayoutPageSchema, LayoutPath, LayoutPathSchema, LayoutPathSegment, LayoutPathSegmentSchema, LayoutRect, LayoutRectSchema, LayoutStructureElement, LayoutStructureElementSchema, LayoutSubpath, LayoutSubpathSchema, LayoutText, LayoutTextSchema, type LoadedMathFont, type MathAssembledGlyphs, type MathBox, type MathColor, type MathFont, type MathFontDescriptorMetrics, type MathFontMetrics, type MathGlyphAssembly, type MathGlyphConstruction, type MathGlyphMetrics, type MathGlyphPart, type MathGlyphPlacement, type MathGlyphRun, type MathGlyphVariant, type MathLayoutItem, type MathRule, type MathStretchAxis, type MathStretchConstruction, type MathStretchGlyph, type MathStretchOptions, type MathStretchPlacement, type MathStretchResult, type MathStroke, type MathVariants, NOOP_DIAGNOSTIC_SINK, type PageRasteriser, PdfBytesSchema, type PdfDiagnostic, type PdfDiagnosticSeverity, type PdfDiagnosticSink, PdfEncryptedError, PdfEncryptionError, type PdfEncryptionOptions, type PdfEncryptionPermissions, type PdfEncryptionScheme, PdfParseError, PdfPasswordRequiredError, type Point, type PositionedFormula, type ProvidedFont, type RasterDrawOp, type RasterFillRectOp, type RasterFillSpec, type RasterImageOp, type RasterMatrix, type RasterPageGeometry, type RasterPathOp, type RasterPathSegment, type RasterRegionPt, type RasterStrokeSpec, type RasterSubpath, type ReadPdfOptions, type RenderPdfPageOptions, type ResolvedFace, type ResolvedFont, STANDARD_METRICS, type StandardFontMeasurerOptions, type StandardFontName, type StyledFragment, type StyledRun, type TextMeasurer, type UnderlineMetrics, type VerticalMetricPolicy, type WinAnsiSubstitution, type WrapOptions, type WrappedLine, type WritePdfOptions, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, scaleMathStretchConstruction, unfilterScanlines, writePdf };
|
|
30
|
+
export { type CcittFaxImage, type CcittFaxOptions, DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_VERTICAL_METRIC_POLICY, DEFAULT_WORD_GAP_EM, type EmbeddedFace, type EmbeddedFaceMetrics, type EmbeddedFaceSubstitution, type FontFace, FontFaceParseError, type FontMeasurerOptions, type FontMetrics, type FontRegistry, type FontRegistryOptions, type FontSubstitution, type GlyphInkBounds, type Jbig2DecodeOptions, type Jbig2Image, Jbig2ParseError, Jbig2UnsupportedError, type Jp2ChannelDefinition, type Jp2ColourSpace, type Jp2Container, type Jp2ImageHeader, type Jpeg2000ComponentMetadata, type Jpeg2000DecodeOptions, type Jpeg2000Image, type Jpeg2000Metadata, Jpeg2000ParseError, type Jpeg2000ProgressionOrder, type Jpeg2000QuantizationStyle, type Jpeg2000Transform, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotation, LayoutAnnotationQuad, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachment, LayoutAttachmentSchema, LayoutDestination, LayoutDestinationSchema, LayoutDestinationTarget, LayoutDestinationTargetSchema, LayoutDocument, LayoutDocumentSchema, LayoutEllipse, LayoutEllipseSchema, LayoutFormField, LayoutFormFieldSchema, LayoutFormWidget, LayoutFormWidgetSchema, LayoutImage, LayoutImageAsset, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLink, LayoutInternalLinkSchema, LayoutItem, LayoutItemSchema, LayoutLayer, LayoutLayerSchema, LayoutLine, LayoutLineSchema, LayoutLink, LayoutLinkSchema, LayoutOutlineItem, LayoutOutlineItemSchema, LayoutPage, LayoutPageSchema, LayoutPath, LayoutPathSchema, LayoutPathSegment, LayoutPathSegmentSchema, LayoutRect, LayoutRectSchema, LayoutStructureElement, LayoutStructureElementSchema, LayoutSubpath, LayoutSubpathSchema, LayoutText, LayoutTextSchema, type LoadedMathFont, type MathAssembledGlyphs, type MathBox, type MathColor, type MathFont, type MathFontDescriptorMetrics, type MathFontMetrics, type MathGlyphAssembly, type MathGlyphConstruction, type MathGlyphMetrics, type MathGlyphPart, type MathGlyphPlacement, type MathGlyphRun, type MathGlyphVariant, type MathLayoutItem, type MathRule, type MathStretchAxis, type MathStretchConstruction, type MathStretchGlyph, type MathStretchOptions, type MathStretchPlacement, type MathStretchResult, type MathStroke, type MathVariants, NOOP_DIAGNOSTIC_SINK, type PageRasteriser, PdfBytesSchema, type PdfDiagnostic, type PdfDiagnosticSeverity, type PdfDiagnosticSink, PdfEncryptedError, PdfEncryptionError, type PdfEncryptionOptions, type PdfEncryptionPermissions, type PdfEncryptionScheme, PdfParseError, PdfPasswordRequiredError, type PdfTextBox, type PdfTextGroupingOptions, type PdfTextLine, type PdfTextRunGeometry, type PdfTextWord, type PdfWordSeparator, type Point, type PositionedFormula, type ProvidedFont, type RasterDrawOp, type RasterFillRectOp, type RasterFillSpec, type RasterImageOp, type RasterMatrix, type RasterPageGeometry, type RasterPathOp, type RasterPathSegment, type RasterRegionPt, type RasterStrokeSpec, type RasterSubpath, type ReadPdfOptions, type RenderPdfPageOptions, type ResolvedFace, type ResolvedFont, STANDARD_METRICS, type StandardFontMeasurerOptions, type StandardFontName, type StyledFragment, type StyledRun, type TextMeasurer, type UnderlineMetrics, type VerticalMetricPolicy, type WinAnsiSubstitution, type WrapOptions, type WrappedLine, type WritePdfOptions, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, groupPdfTextRuns, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, runGapPt, runsShareBaseline, scaleMathStretchConstruction, unfilterScanlines, writePdf };
|
package/dist/index.d.ts
CHANGED
|
@@ -23,7 +23,8 @@ import { a as MathGlyphPart, c as MathVariants, n as MathGlyphAssembly, o as Mat
|
|
|
23
23
|
import { a as scaleMathStretchConstruction, i as assembleStretchyGlyph, n as MathStretchOptions, r as MathStretchPlacement, t as MathStretchConstruction } from "./math-stretch-BbC2o4Br.js";
|
|
24
24
|
import { LoadedMathFont, MathFont, MathFontDescriptorMetrics, loadMathFont } from "./math-font.js";
|
|
25
25
|
import { DEFAULT_VERTICAL_METRIC_POLICY, FontMeasurerOptions, StandardFontMeasurerOptions, VerticalMetricPolicy, createFontMeasurer, createStandardFontMeasurer } from "./measure.js";
|
|
26
|
+
import { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline } from "./text-group.js";
|
|
26
27
|
import { PageRasteriser, RasterDrawOp, RasterFillRectOp, RasterFillSpec, RasterImageOp, RasterMatrix, RasterPageGeometry, RasterPathOp, RasterPathSegment, RasterRegionPt, RasterStrokeSpec, RasterSubpath, RenderPdfPageOptions, renderPdfPage } from "./raster.js";
|
|
27
28
|
import { FontRegistryOptions, FontSubstitution, MathAssembledGlyphs, MathBox, MathColor, MathFontMetrics, MathGlyphMetrics, MathGlyphPlacement, MathGlyphRun, MathLayoutItem, MathRule, MathStretchAxis, MathStretchGlyph, MathStretchResult, MathStroke, Point, PositionedFormula, ProvidedFont, StyledFragment, StyledRun, TextMeasurer, UnderlineMetrics, WrapOptions, WrappedLine } from "document-schema.js";
|
|
28
29
|
export * from "byte-codec";
|
|
29
|
-
export { type CcittFaxImage, type CcittFaxOptions, DEFAULT_VERTICAL_METRIC_POLICY, type EmbeddedFace, type EmbeddedFaceMetrics, type EmbeddedFaceSubstitution, type FontFace, FontFaceParseError, type FontMeasurerOptions, type FontMetrics, type FontRegistry, type FontRegistryOptions, type FontSubstitution, type GlyphInkBounds, type Jbig2DecodeOptions, type Jbig2Image, Jbig2ParseError, Jbig2UnsupportedError, type Jp2ChannelDefinition, type Jp2ColourSpace, type Jp2Container, type Jp2ImageHeader, type Jpeg2000ComponentMetadata, type Jpeg2000DecodeOptions, type Jpeg2000Image, type Jpeg2000Metadata, Jpeg2000ParseError, type Jpeg2000ProgressionOrder, type Jpeg2000QuantizationStyle, type Jpeg2000Transform, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotation, LayoutAnnotationQuad, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachment, LayoutAttachmentSchema, LayoutDestination, LayoutDestinationSchema, LayoutDestinationTarget, LayoutDestinationTargetSchema, LayoutDocument, LayoutDocumentSchema, LayoutEllipse, LayoutEllipseSchema, LayoutFormField, LayoutFormFieldSchema, LayoutFormWidget, LayoutFormWidgetSchema, LayoutImage, LayoutImageAsset, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLink, LayoutInternalLinkSchema, LayoutItem, LayoutItemSchema, LayoutLayer, LayoutLayerSchema, LayoutLine, LayoutLineSchema, LayoutLink, LayoutLinkSchema, LayoutOutlineItem, LayoutOutlineItemSchema, LayoutPage, LayoutPageSchema, LayoutPath, LayoutPathSchema, LayoutPathSegment, LayoutPathSegmentSchema, LayoutRect, LayoutRectSchema, LayoutStructureElement, LayoutStructureElementSchema, LayoutSubpath, LayoutSubpathSchema, LayoutText, LayoutTextSchema, type LoadedMathFont, type MathAssembledGlyphs, type MathBox, type MathColor, type MathFont, type MathFontDescriptorMetrics, type MathFontMetrics, type MathGlyphAssembly, type MathGlyphConstruction, type MathGlyphMetrics, type MathGlyphPart, type MathGlyphPlacement, type MathGlyphRun, type MathGlyphVariant, type MathLayoutItem, type MathRule, type MathStretchAxis, type MathStretchConstruction, type MathStretchGlyph, type MathStretchOptions, type MathStretchPlacement, type MathStretchResult, type MathStroke, type MathVariants, NOOP_DIAGNOSTIC_SINK, type PageRasteriser, PdfBytesSchema, type PdfDiagnostic, type PdfDiagnosticSeverity, type PdfDiagnosticSink, PdfEncryptedError, PdfEncryptionError, type PdfEncryptionOptions, type PdfEncryptionPermissions, type PdfEncryptionScheme, PdfParseError, PdfPasswordRequiredError, type Point, type PositionedFormula, type ProvidedFont, type RasterDrawOp, type RasterFillRectOp, type RasterFillSpec, type RasterImageOp, type RasterMatrix, type RasterPageGeometry, type RasterPathOp, type RasterPathSegment, type RasterRegionPt, type RasterStrokeSpec, type RasterSubpath, type ReadPdfOptions, type RenderPdfPageOptions, type ResolvedFace, type ResolvedFont, STANDARD_METRICS, type StandardFontMeasurerOptions, type StandardFontName, type StyledFragment, type StyledRun, type TextMeasurer, type UnderlineMetrics, type VerticalMetricPolicy, type WinAnsiSubstitution, type WrapOptions, type WrappedLine, type WritePdfOptions, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, scaleMathStretchConstruction, unfilterScanlines, writePdf };
|
|
30
|
+
export { type CcittFaxImage, type CcittFaxOptions, DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_VERTICAL_METRIC_POLICY, DEFAULT_WORD_GAP_EM, type EmbeddedFace, type EmbeddedFaceMetrics, type EmbeddedFaceSubstitution, type FontFace, FontFaceParseError, type FontMeasurerOptions, type FontMetrics, type FontRegistry, type FontRegistryOptions, type FontSubstitution, type GlyphInkBounds, type Jbig2DecodeOptions, type Jbig2Image, Jbig2ParseError, Jbig2UnsupportedError, type Jp2ChannelDefinition, type Jp2ColourSpace, type Jp2Container, type Jp2ImageHeader, type Jpeg2000ComponentMetadata, type Jpeg2000DecodeOptions, type Jpeg2000Image, type Jpeg2000Metadata, Jpeg2000ParseError, type Jpeg2000ProgressionOrder, type Jpeg2000QuantizationStyle, type Jpeg2000Transform, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotation, LayoutAnnotationQuad, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachment, LayoutAttachmentSchema, LayoutDestination, LayoutDestinationSchema, LayoutDestinationTarget, LayoutDestinationTargetSchema, LayoutDocument, LayoutDocumentSchema, LayoutEllipse, LayoutEllipseSchema, LayoutFormField, LayoutFormFieldSchema, LayoutFormWidget, LayoutFormWidgetSchema, LayoutImage, LayoutImageAsset, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLink, LayoutInternalLinkSchema, LayoutItem, LayoutItemSchema, LayoutLayer, LayoutLayerSchema, LayoutLine, LayoutLineSchema, LayoutLink, LayoutLinkSchema, LayoutOutlineItem, LayoutOutlineItemSchema, LayoutPage, LayoutPageSchema, LayoutPath, LayoutPathSchema, LayoutPathSegment, LayoutPathSegmentSchema, LayoutRect, LayoutRectSchema, LayoutStructureElement, LayoutStructureElementSchema, LayoutSubpath, LayoutSubpathSchema, LayoutText, LayoutTextSchema, type LoadedMathFont, type MathAssembledGlyphs, type MathBox, type MathColor, type MathFont, type MathFontDescriptorMetrics, type MathFontMetrics, type MathGlyphAssembly, type MathGlyphConstruction, type MathGlyphMetrics, type MathGlyphPart, type MathGlyphPlacement, type MathGlyphRun, type MathGlyphVariant, type MathLayoutItem, type MathRule, type MathStretchAxis, type MathStretchConstruction, type MathStretchGlyph, type MathStretchOptions, type MathStretchPlacement, type MathStretchResult, type MathStroke, type MathVariants, NOOP_DIAGNOSTIC_SINK, type PageRasteriser, PdfBytesSchema, type PdfDiagnostic, type PdfDiagnosticSeverity, type PdfDiagnosticSink, PdfEncryptedError, PdfEncryptionError, type PdfEncryptionOptions, type PdfEncryptionPermissions, type PdfEncryptionScheme, PdfParseError, PdfPasswordRequiredError, type PdfTextBox, type PdfTextGroupingOptions, type PdfTextLine, type PdfTextRunGeometry, type PdfTextWord, type PdfWordSeparator, type Point, type PositionedFormula, type ProvidedFont, type RasterDrawOp, type RasterFillRectOp, type RasterFillSpec, type RasterImageOp, type RasterMatrix, type RasterPageGeometry, type RasterPathOp, type RasterPathSegment, type RasterRegionPt, type RasterStrokeSpec, type RasterSubpath, type ReadPdfOptions, type RenderPdfPageOptions, type ResolvedFace, type ResolvedFont, STANDARD_METRICS, type StandardFontMeasurerOptions, type StandardFontName, type StyledFragment, type StyledRun, type TextMeasurer, type UnderlineMetrics, type VerticalMetricPolicy, type WinAnsiSubstitution, type WrapOptions, type WrappedLine, type WritePdfOptions, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, groupPdfTextRuns, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, runGapPt, runsShareBaseline, scaleMathStretchConstruction, unfilterScanlines, writePdf };
|
package/dist/index.js
CHANGED
|
@@ -18,6 +18,7 @@ import { DEFAULT_VERTICAL_METRIC_POLICY, createFontMeasurer, createStandardFontM
|
|
|
18
18
|
import { writePdf } from "./write.js";
|
|
19
19
|
import { PdfBytesSchema, pdfCodec } from "./codec.js";
|
|
20
20
|
import { FontFaceParseError, readFontFace } from "./font-face.js";
|
|
21
|
+
import { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, groupPdfTextRuns, runGapPt, runsShareBaseline } from "./text-group.js";
|
|
21
22
|
import { renderPdfPage } from "./raster.js";
|
|
22
23
|
export * from "byte-codec";
|
|
23
|
-
export { DEFAULT_VERTICAL_METRIC_POLICY, FontFaceParseError, Jbig2ParseError, Jbig2UnsupportedError, Jpeg2000ParseError, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachmentSchema, LayoutDestinationSchema, LayoutDestinationTargetSchema, LayoutDocumentSchema, LayoutEllipseSchema, LayoutFormFieldSchema, LayoutFormWidgetSchema, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLinkSchema, LayoutItemSchema, LayoutLayerSchema, LayoutLineSchema, LayoutLinkSchema, LayoutOutlineItemSchema, LayoutPageSchema, LayoutPathSchema, LayoutPathSegmentSchema, LayoutRectSchema, LayoutStructureElementSchema, LayoutSubpathSchema, LayoutTextSchema, NOOP_DIAGNOSTIC_SINK, PdfBytesSchema, PdfEncryptedError, PdfEncryptionError, PdfParseError, PdfPasswordRequiredError, STANDARD_METRICS, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, scaleMathStretchConstruction, unfilterScanlines, writePdf };
|
|
24
|
+
export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_VERTICAL_METRIC_POLICY, DEFAULT_WORD_GAP_EM, FontFaceParseError, Jbig2ParseError, Jbig2UnsupportedError, Jpeg2000ParseError, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachmentSchema, LayoutDestinationSchema, LayoutDestinationTargetSchema, LayoutDocumentSchema, LayoutEllipseSchema, LayoutFormFieldSchema, LayoutFormWidgetSchema, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLinkSchema, LayoutItemSchema, LayoutLayerSchema, LayoutLineSchema, LayoutLinkSchema, LayoutOutlineItemSchema, LayoutPageSchema, LayoutPathSchema, LayoutPathSegmentSchema, LayoutRectSchema, LayoutStructureElementSchema, LayoutSubpathSchema, LayoutTextSchema, NOOP_DIAGNOSTIC_SINK, PdfBytesSchema, PdfEncryptedError, PdfEncryptionError, PdfParseError, PdfPasswordRequiredError, STANDARD_METRICS, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, groupPdfTextRuns, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, runGapPt, runsShareBaseline, scaleMathStretchConstruction, unfilterScanlines, writePdf };
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
|
|
2
|
+
//#region src/text-group.ts
|
|
3
|
+
const DEFAULT_BASELINE_TOLERANCE_EM = 2 / 3;
|
|
4
|
+
const DEFAULT_WORD_GAP_EM = 1 / 8;
|
|
5
|
+
const DEFAULT_COLUMN_GAP_EM = 1;
|
|
6
|
+
const ROTATION_NOISE_DEG = 1e-6;
|
|
7
|
+
const DEGREES_PER_TURN = 360;
|
|
8
|
+
function normaliseRotationDeg(rotationDeg) {
|
|
9
|
+
return (Math.round((rotationDeg ?? 0) / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
|
|
10
|
+
}
|
|
11
|
+
const RADIANS_PER_TURN = 2 * Math.PI;
|
|
12
|
+
function baselineAxis(rotationDeg) {
|
|
13
|
+
const radians = rotationDeg * RADIANS_PER_TURN / DEGREES_PER_TURN;
|
|
14
|
+
return {
|
|
15
|
+
cos: Math.cos(radians),
|
|
16
|
+
sin: Math.sin(radians)
|
|
17
|
+
};
|
|
18
|
+
}
|
|
19
|
+
function project(run, rotationDeg) {
|
|
20
|
+
const { cos, sin } = baselineAxis(rotationDeg);
|
|
21
|
+
return {
|
|
22
|
+
run,
|
|
23
|
+
alongPt: run.xPt * cos + run.yPt * sin,
|
|
24
|
+
acrossPt: run.yPt * cos - run.xPt * sin,
|
|
25
|
+
sizePt: run.sizePt
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
function unproject(alongPt, acrossPt, rotationDeg) {
|
|
29
|
+
const { cos, sin } = baselineAxis(rotationDeg);
|
|
30
|
+
return {
|
|
31
|
+
xPt: alongPt * cos - acrossPt * sin,
|
|
32
|
+
yPt: alongPt * sin + acrossPt * cos
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Whether two positioned runs sit on one visual line.
|
|
37
|
+
*
|
|
38
|
+
* The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
|
|
39
|
+
* @param a - one run
|
|
40
|
+
* @param b - the other run
|
|
41
|
+
* @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
|
|
42
|
+
* @returns true when the two baselines are within tolerance of each other
|
|
43
|
+
*/
|
|
44
|
+
function runsShareBaseline(a, b, options = {}) {
|
|
45
|
+
const rotationDeg = normaliseRotationDeg(a.rotationDeg);
|
|
46
|
+
if (rotationDeg !== normaliseRotationDeg(b.rotationDeg)) return false;
|
|
47
|
+
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
48
|
+
return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
|
|
52
|
+
*
|
|
53
|
+
* Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
|
|
54
|
+
* @param previous - the run to the left, in reading order
|
|
55
|
+
* @param next - the run that follows it
|
|
56
|
+
* @returns the gap in points, or undefined when it cannot be derived
|
|
57
|
+
*/
|
|
58
|
+
function runGapPt(previous, next) {
|
|
59
|
+
const rotationDeg = normaliseRotationDeg(previous.rotationDeg);
|
|
60
|
+
if (rotationDeg !== normaliseRotationDeg(next.rotationDeg)) return;
|
|
61
|
+
const previousWidthPt = previous.widthPt;
|
|
62
|
+
if (previousWidthPt === void 0) return;
|
|
63
|
+
return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
|
|
64
|
+
}
|
|
65
|
+
function clusterIntoLines(entries, toleranceEm) {
|
|
66
|
+
const lines = [];
|
|
67
|
+
for (const entry of entries) {
|
|
68
|
+
let nearest;
|
|
69
|
+
for (const line of lines) if (runsShareBaseline(line.anchor.run, entry.run, { baselineToleranceEm: toleranceEm })) nearest = line;
|
|
70
|
+
if (nearest === void 0) lines.push({
|
|
71
|
+
anchor: entry,
|
|
72
|
+
entries: [entry]
|
|
73
|
+
});
|
|
74
|
+
else nearest.entries.push(entry);
|
|
75
|
+
}
|
|
76
|
+
return lines;
|
|
77
|
+
}
|
|
78
|
+
function tokeniseRun(text) {
|
|
79
|
+
const words = text.match(/\S+/g) ?? [];
|
|
80
|
+
const leadingSpace = /^\s/.test(text);
|
|
81
|
+
const trailingSpace = /\s$/.test(text);
|
|
82
|
+
return {
|
|
83
|
+
words,
|
|
84
|
+
leadingSpace,
|
|
85
|
+
trailingSpace,
|
|
86
|
+
whole: words.length === 1 && !leadingSpace && !trailingSpace
|
|
87
|
+
};
|
|
88
|
+
}
|
|
89
|
+
const SEPARATOR_TEXT = {
|
|
90
|
+
none: "",
|
|
91
|
+
space: " ",
|
|
92
|
+
column: " "
|
|
93
|
+
};
|
|
94
|
+
function separatorForGap(gapPt, smallerSizePt, wordGapEm, columnGapEm) {
|
|
95
|
+
if (gapPt === void 0) return "none";
|
|
96
|
+
if (gapPt >= columnGapEm * smallerSizePt) return "column";
|
|
97
|
+
if (gapPt >= wordGapEm * smallerSizePt) return "space";
|
|
98
|
+
return "none";
|
|
99
|
+
}
|
|
100
|
+
function boundsOf(minAlongPt, maxAlongEndPt, acrossPt, maxSizePt, rotationDeg) {
|
|
101
|
+
const origin = unproject(minAlongPt, acrossPt, rotationDeg);
|
|
102
|
+
return {
|
|
103
|
+
xPt: origin.xPt,
|
|
104
|
+
yPt: origin.yPt,
|
|
105
|
+
widthPt: maxAlongEndPt - minAlongPt,
|
|
106
|
+
heightPt: maxSizePt
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
110
|
+
const working = [];
|
|
111
|
+
let previous;
|
|
112
|
+
for (const entry of line.entries) {
|
|
113
|
+
const tokens = tokeniseRun(entry.run.text);
|
|
114
|
+
const gapSeparator = previous === void 0 ? "none" : separatorForGap(runGapPt(previous.entry.run, entry.run), Math.min(previous.entry.sizePt, entry.sizePt), wordGapEm, columnGapEm);
|
|
115
|
+
const pendingSpace = previous?.trailingSpace === true;
|
|
116
|
+
let separator = "none";
|
|
117
|
+
if (working.length > 0) separator = gapSeparator === "none" && (pendingSpace || tokens.leadingSpace) ? "space" : gapSeparator;
|
|
118
|
+
const widthPt = entry.run.widthPt;
|
|
119
|
+
const alongEndPt = entry.alongPt + (widthPt ?? 0);
|
|
120
|
+
tokens.words.forEach((word, index) => {
|
|
121
|
+
const last = working[working.length - 1];
|
|
122
|
+
if (index === 0 && separator === "none" && last !== void 0) {
|
|
123
|
+
last.text += word;
|
|
124
|
+
last.runs.push(entry.run);
|
|
125
|
+
last.measurable = last.measurable && tokens.whole && widthPt !== void 0;
|
|
126
|
+
last.minAlongPt = Math.min(last.minAlongPt, entry.alongPt);
|
|
127
|
+
last.maxAlongEndPt = Math.max(last.maxAlongEndPt, alongEndPt);
|
|
128
|
+
last.maxSizePt = Math.max(last.maxSizePt, entry.sizePt);
|
|
129
|
+
return;
|
|
130
|
+
}
|
|
131
|
+
working.push({
|
|
132
|
+
text: word,
|
|
133
|
+
separatorBefore: index === 0 ? separator : "space",
|
|
134
|
+
runs: [entry.run],
|
|
135
|
+
measurable: tokens.whole && widthPt !== void 0,
|
|
136
|
+
minAlongPt: entry.alongPt,
|
|
137
|
+
maxAlongEndPt: alongEndPt,
|
|
138
|
+
maxSizePt: entry.sizePt
|
|
139
|
+
});
|
|
140
|
+
});
|
|
141
|
+
previous = {
|
|
142
|
+
entry,
|
|
143
|
+
trailingSpace: tokens.trailingSpace
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
return working.map((word) => ({
|
|
147
|
+
text: word.text,
|
|
148
|
+
separatorBefore: word.separatorBefore,
|
|
149
|
+
runs: word.runs,
|
|
150
|
+
...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
|
|
151
|
+
}));
|
|
152
|
+
}
|
|
153
|
+
function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
154
|
+
const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
|
|
155
|
+
const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
|
|
156
|
+
let minAlongPt = Number.POSITIVE_INFINITY;
|
|
157
|
+
let maxAlongEndPt = Number.NEGATIVE_INFINITY;
|
|
158
|
+
let maxSizePt = 0;
|
|
159
|
+
let measurable = true;
|
|
160
|
+
for (const entry of line.entries) {
|
|
161
|
+
const widthPt = entry.run.widthPt;
|
|
162
|
+
if (widthPt === void 0) measurable = false;
|
|
163
|
+
minAlongPt = Math.min(minAlongPt, entry.alongPt);
|
|
164
|
+
maxAlongEndPt = Math.max(maxAlongEndPt, entry.alongPt + (widthPt ?? 0));
|
|
165
|
+
maxSizePt = Math.max(maxSizePt, entry.sizePt);
|
|
166
|
+
}
|
|
167
|
+
return {
|
|
168
|
+
text,
|
|
169
|
+
words,
|
|
170
|
+
baselineYPt: line.anchor.acrossPt,
|
|
171
|
+
rotationDeg,
|
|
172
|
+
...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
|
|
173
|
+
};
|
|
174
|
+
}
|
|
175
|
+
/**
|
|
176
|
+
* Groups positioned PDF text runs into lines of words.
|
|
177
|
+
*
|
|
178
|
+
* Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
|
|
179
|
+
*
|
|
180
|
+
* Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
|
|
181
|
+
*
|
|
182
|
+
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
183
|
+
*
|
|
184
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. Vertical writing modes are not recognised, because this package's content interpreter does not read a CMap's WMode and so reports vertically set text with horizontal advances (ExaDev/documents.js#1358), which puts the positions beyond anything grouping could repair. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
185
|
+
* @param runs - positioned text runs, in any order
|
|
186
|
+
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
187
|
+
* @returns the grouped lines, in reading order
|
|
188
|
+
*/
|
|
189
|
+
function groupPdfTextRuns(runs, options = {}) {
|
|
190
|
+
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
191
|
+
const wordGapEm = options.wordGapEm ?? .125;
|
|
192
|
+
const columnGapEm = options.columnGapEm ?? 1;
|
|
193
|
+
const byRotation = /* @__PURE__ */ new Map();
|
|
194
|
+
for (const run of runs) {
|
|
195
|
+
if (run.text.length === 0) continue;
|
|
196
|
+
const rotationDeg = normaliseRotationDeg(run.rotationDeg);
|
|
197
|
+
const bucket = byRotation.get(rotationDeg);
|
|
198
|
+
if (bucket === void 0) byRotation.set(rotationDeg, [run]);
|
|
199
|
+
else bucket.push(run);
|
|
200
|
+
}
|
|
201
|
+
const result = [];
|
|
202
|
+
for (const rotationDeg of [...byRotation.keys()].sort((a, b) => a - b)) {
|
|
203
|
+
const entries = (byRotation.get(rotationDeg) ?? []).map((run) => project(run, rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
|
|
204
|
+
for (const line of clusterIntoLines(entries, toleranceEm)) {
|
|
205
|
+
line.entries.sort((a, b) => a.alongPt - b.alongPt);
|
|
206
|
+
result.push(buildLine(line, rotationDeg, wordGapEm, columnGapEm));
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
return result;
|
|
210
|
+
}
|
|
211
|
+
//#endregion
|
|
212
|
+
exports.DEFAULT_BASELINE_TOLERANCE_EM = DEFAULT_BASELINE_TOLERANCE_EM;
|
|
213
|
+
exports.DEFAULT_COLUMN_GAP_EM = DEFAULT_COLUMN_GAP_EM;
|
|
214
|
+
exports.DEFAULT_WORD_GAP_EM = DEFAULT_WORD_GAP_EM;
|
|
215
|
+
exports.groupPdfTextRuns = groupPdfTextRuns;
|
|
216
|
+
exports.runGapPt = runGapPt;
|
|
217
|
+
exports.runsShareBaseline = runsShareBaseline;
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
//#region src/text-group.d.ts
|
|
2
|
+
interface PdfTextRunGeometry {
|
|
3
|
+
readonly text: string;
|
|
4
|
+
readonly xPt: number;
|
|
5
|
+
readonly yPt: number;
|
|
6
|
+
readonly sizePt: number;
|
|
7
|
+
readonly widthPt?: number;
|
|
8
|
+
readonly rotationDeg?: number;
|
|
9
|
+
}
|
|
10
|
+
interface PdfTextBox {
|
|
11
|
+
readonly xPt: number;
|
|
12
|
+
readonly yPt: number;
|
|
13
|
+
readonly widthPt: number;
|
|
14
|
+
readonly heightPt: number;
|
|
15
|
+
}
|
|
16
|
+
type PdfWordSeparator = "none" | "space" | "column";
|
|
17
|
+
interface PdfTextWord<TRun extends PdfTextRunGeometry> {
|
|
18
|
+
readonly text: string;
|
|
19
|
+
readonly separatorBefore: PdfWordSeparator;
|
|
20
|
+
readonly runs: readonly TRun[];
|
|
21
|
+
readonly bounds?: PdfTextBox;
|
|
22
|
+
}
|
|
23
|
+
interface PdfTextLine<TRun extends PdfTextRunGeometry> {
|
|
24
|
+
readonly text: string;
|
|
25
|
+
readonly words: readonly PdfTextWord<TRun>[];
|
|
26
|
+
readonly baselineYPt: number;
|
|
27
|
+
readonly rotationDeg: number;
|
|
28
|
+
readonly bounds?: PdfTextBox;
|
|
29
|
+
}
|
|
30
|
+
interface PdfTextGroupingOptions {
|
|
31
|
+
readonly baselineToleranceEm?: number;
|
|
32
|
+
readonly wordGapEm?: number;
|
|
33
|
+
readonly columnGapEm?: number;
|
|
34
|
+
}
|
|
35
|
+
declare const DEFAULT_BASELINE_TOLERANCE_EM: number;
|
|
36
|
+
declare const DEFAULT_WORD_GAP_EM: number;
|
|
37
|
+
declare const DEFAULT_COLUMN_GAP_EM = 1;
|
|
38
|
+
/**
|
|
39
|
+
* Whether two positioned runs sit on one visual line.
|
|
40
|
+
*
|
|
41
|
+
* The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
|
|
42
|
+
* @param a - one run
|
|
43
|
+
* @param b - the other run
|
|
44
|
+
* @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
|
|
45
|
+
* @returns true when the two baselines are within tolerance of each other
|
|
46
|
+
*/
|
|
47
|
+
declare function runsShareBaseline(a: PdfTextRunGeometry, b: PdfTextRunGeometry, options?: PdfTextGroupingOptions): boolean;
|
|
48
|
+
/**
|
|
49
|
+
* The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
|
|
50
|
+
*
|
|
51
|
+
* Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
|
|
52
|
+
* @param previous - the run to the left, in reading order
|
|
53
|
+
* @param next - the run that follows it
|
|
54
|
+
* @returns the gap in points, or undefined when it cannot be derived
|
|
55
|
+
*/
|
|
56
|
+
declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry): number | undefined;
|
|
57
|
+
/**
|
|
58
|
+
* Groups positioned PDF text runs into lines of words.
|
|
59
|
+
*
|
|
60
|
+
* Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
|
|
61
|
+
*
|
|
62
|
+
* Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
|
|
63
|
+
*
|
|
64
|
+
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
65
|
+
*
|
|
66
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. Vertical writing modes are not recognised, because this package's content interpreter does not read a CMap's WMode and so reports vertically set text with horizontal advances (ExaDev/documents.js#1358), which puts the positions beyond anything grouping could repair. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
67
|
+
* @param runs - positioned text runs, in any order
|
|
68
|
+
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
69
|
+
* @returns the grouped lines, in reading order
|
|
70
|
+
*/
|
|
71
|
+
declare function groupPdfTextRuns<TRun extends PdfTextRunGeometry>(runs: readonly TRun[], options?: PdfTextGroupingOptions): PdfTextLine<TRun>[];
|
|
72
|
+
//#endregion
|
|
73
|
+
export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline };
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
//#region src/text-group.d.ts
|
|
2
|
+
interface PdfTextRunGeometry {
|
|
3
|
+
readonly text: string;
|
|
4
|
+
readonly xPt: number;
|
|
5
|
+
readonly yPt: number;
|
|
6
|
+
readonly sizePt: number;
|
|
7
|
+
readonly widthPt?: number;
|
|
8
|
+
readonly rotationDeg?: number;
|
|
9
|
+
}
|
|
10
|
+
interface PdfTextBox {
|
|
11
|
+
readonly xPt: number;
|
|
12
|
+
readonly yPt: number;
|
|
13
|
+
readonly widthPt: number;
|
|
14
|
+
readonly heightPt: number;
|
|
15
|
+
}
|
|
16
|
+
type PdfWordSeparator = "none" | "space" | "column";
|
|
17
|
+
interface PdfTextWord<TRun extends PdfTextRunGeometry> {
|
|
18
|
+
readonly text: string;
|
|
19
|
+
readonly separatorBefore: PdfWordSeparator;
|
|
20
|
+
readonly runs: readonly TRun[];
|
|
21
|
+
readonly bounds?: PdfTextBox;
|
|
22
|
+
}
|
|
23
|
+
interface PdfTextLine<TRun extends PdfTextRunGeometry> {
|
|
24
|
+
readonly text: string;
|
|
25
|
+
readonly words: readonly PdfTextWord<TRun>[];
|
|
26
|
+
readonly baselineYPt: number;
|
|
27
|
+
readonly rotationDeg: number;
|
|
28
|
+
readonly bounds?: PdfTextBox;
|
|
29
|
+
}
|
|
30
|
+
interface PdfTextGroupingOptions {
|
|
31
|
+
readonly baselineToleranceEm?: number;
|
|
32
|
+
readonly wordGapEm?: number;
|
|
33
|
+
readonly columnGapEm?: number;
|
|
34
|
+
}
|
|
35
|
+
declare const DEFAULT_BASELINE_TOLERANCE_EM: number;
|
|
36
|
+
declare const DEFAULT_WORD_GAP_EM: number;
|
|
37
|
+
declare const DEFAULT_COLUMN_GAP_EM = 1;
|
|
38
|
+
/**
|
|
39
|
+
* Whether two positioned runs sit on one visual line.
|
|
40
|
+
*
|
|
41
|
+
* The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
|
|
42
|
+
* @param a - one run
|
|
43
|
+
* @param b - the other run
|
|
44
|
+
* @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
|
|
45
|
+
* @returns true when the two baselines are within tolerance of each other
|
|
46
|
+
*/
|
|
47
|
+
declare function runsShareBaseline(a: PdfTextRunGeometry, b: PdfTextRunGeometry, options?: PdfTextGroupingOptions): boolean;
|
|
48
|
+
/**
|
|
49
|
+
* The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
|
|
50
|
+
*
|
|
51
|
+
* Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
|
|
52
|
+
* @param previous - the run to the left, in reading order
|
|
53
|
+
* @param next - the run that follows it
|
|
54
|
+
* @returns the gap in points, or undefined when it cannot be derived
|
|
55
|
+
*/
|
|
56
|
+
declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry): number | undefined;
|
|
57
|
+
/**
|
|
58
|
+
* Groups positioned PDF text runs into lines of words.
|
|
59
|
+
*
|
|
60
|
+
* Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
|
|
61
|
+
*
|
|
62
|
+
* Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
|
|
63
|
+
*
|
|
64
|
+
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
65
|
+
*
|
|
66
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. Vertical writing modes are not recognised, because this package's content interpreter does not read a CMap's WMode and so reports vertically set text with horizontal advances (ExaDev/documents.js#1358), which puts the positions beyond anything grouping could repair. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
67
|
+
* @param runs - positioned text runs, in any order
|
|
68
|
+
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
69
|
+
* @returns the grouped lines, in reading order
|
|
70
|
+
*/
|
|
71
|
+
declare function groupPdfTextRuns<TRun extends PdfTextRunGeometry>(runs: readonly TRun[], options?: PdfTextGroupingOptions): PdfTextLine<TRun>[];
|
|
72
|
+
//#endregion
|
|
73
|
+
export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline };
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
//#region src/text-group.ts
|
|
2
|
+
const DEFAULT_BASELINE_TOLERANCE_EM = 2 / 3;
|
|
3
|
+
const DEFAULT_WORD_GAP_EM = 1 / 8;
|
|
4
|
+
const DEFAULT_COLUMN_GAP_EM = 1;
|
|
5
|
+
const ROTATION_NOISE_DEG = 1e-6;
|
|
6
|
+
const DEGREES_PER_TURN = 360;
|
|
7
|
+
function normaliseRotationDeg(rotationDeg) {
|
|
8
|
+
return (Math.round((rotationDeg ?? 0) / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
|
|
9
|
+
}
|
|
10
|
+
const RADIANS_PER_TURN = 2 * Math.PI;
|
|
11
|
+
function baselineAxis(rotationDeg) {
|
|
12
|
+
const radians = rotationDeg * RADIANS_PER_TURN / DEGREES_PER_TURN;
|
|
13
|
+
return {
|
|
14
|
+
cos: Math.cos(radians),
|
|
15
|
+
sin: Math.sin(radians)
|
|
16
|
+
};
|
|
17
|
+
}
|
|
18
|
+
function project(run, rotationDeg) {
|
|
19
|
+
const { cos, sin } = baselineAxis(rotationDeg);
|
|
20
|
+
return {
|
|
21
|
+
run,
|
|
22
|
+
alongPt: run.xPt * cos + run.yPt * sin,
|
|
23
|
+
acrossPt: run.yPt * cos - run.xPt * sin,
|
|
24
|
+
sizePt: run.sizePt
|
|
25
|
+
};
|
|
26
|
+
}
|
|
27
|
+
function unproject(alongPt, acrossPt, rotationDeg) {
|
|
28
|
+
const { cos, sin } = baselineAxis(rotationDeg);
|
|
29
|
+
return {
|
|
30
|
+
xPt: alongPt * cos - acrossPt * sin,
|
|
31
|
+
yPt: alongPt * sin + acrossPt * cos
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* Whether two positioned runs sit on one visual line.
|
|
36
|
+
*
|
|
37
|
+
* The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
|
|
38
|
+
* @param a - one run
|
|
39
|
+
* @param b - the other run
|
|
40
|
+
* @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
|
|
41
|
+
* @returns true when the two baselines are within tolerance of each other
|
|
42
|
+
*/
|
|
43
|
+
function runsShareBaseline(a, b, options = {}) {
|
|
44
|
+
const rotationDeg = normaliseRotationDeg(a.rotationDeg);
|
|
45
|
+
if (rotationDeg !== normaliseRotationDeg(b.rotationDeg)) return false;
|
|
46
|
+
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
47
|
+
return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
|
|
51
|
+
*
|
|
52
|
+
* Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
|
|
53
|
+
* @param previous - the run to the left, in reading order
|
|
54
|
+
* @param next - the run that follows it
|
|
55
|
+
* @returns the gap in points, or undefined when it cannot be derived
|
|
56
|
+
*/
|
|
57
|
+
function runGapPt(previous, next) {
|
|
58
|
+
const rotationDeg = normaliseRotationDeg(previous.rotationDeg);
|
|
59
|
+
if (rotationDeg !== normaliseRotationDeg(next.rotationDeg)) return;
|
|
60
|
+
const previousWidthPt = previous.widthPt;
|
|
61
|
+
if (previousWidthPt === void 0) return;
|
|
62
|
+
return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
|
|
63
|
+
}
|
|
64
|
+
function clusterIntoLines(entries, toleranceEm) {
|
|
65
|
+
const lines = [];
|
|
66
|
+
for (const entry of entries) {
|
|
67
|
+
let nearest;
|
|
68
|
+
for (const line of lines) if (runsShareBaseline(line.anchor.run, entry.run, { baselineToleranceEm: toleranceEm })) nearest = line;
|
|
69
|
+
if (nearest === void 0) lines.push({
|
|
70
|
+
anchor: entry,
|
|
71
|
+
entries: [entry]
|
|
72
|
+
});
|
|
73
|
+
else nearest.entries.push(entry);
|
|
74
|
+
}
|
|
75
|
+
return lines;
|
|
76
|
+
}
|
|
77
|
+
function tokeniseRun(text) {
|
|
78
|
+
const words = text.match(/\S+/g) ?? [];
|
|
79
|
+
const leadingSpace = /^\s/.test(text);
|
|
80
|
+
const trailingSpace = /\s$/.test(text);
|
|
81
|
+
return {
|
|
82
|
+
words,
|
|
83
|
+
leadingSpace,
|
|
84
|
+
trailingSpace,
|
|
85
|
+
whole: words.length === 1 && !leadingSpace && !trailingSpace
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
const SEPARATOR_TEXT = {
|
|
89
|
+
none: "",
|
|
90
|
+
space: " ",
|
|
91
|
+
column: " "
|
|
92
|
+
};
|
|
93
|
+
function separatorForGap(gapPt, smallerSizePt, wordGapEm, columnGapEm) {
|
|
94
|
+
if (gapPt === void 0) return "none";
|
|
95
|
+
if (gapPt >= columnGapEm * smallerSizePt) return "column";
|
|
96
|
+
if (gapPt >= wordGapEm * smallerSizePt) return "space";
|
|
97
|
+
return "none";
|
|
98
|
+
}
|
|
99
|
+
function boundsOf(minAlongPt, maxAlongEndPt, acrossPt, maxSizePt, rotationDeg) {
|
|
100
|
+
const origin = unproject(minAlongPt, acrossPt, rotationDeg);
|
|
101
|
+
return {
|
|
102
|
+
xPt: origin.xPt,
|
|
103
|
+
yPt: origin.yPt,
|
|
104
|
+
widthPt: maxAlongEndPt - minAlongPt,
|
|
105
|
+
heightPt: maxSizePt
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
109
|
+
const working = [];
|
|
110
|
+
let previous;
|
|
111
|
+
for (const entry of line.entries) {
|
|
112
|
+
const tokens = tokeniseRun(entry.run.text);
|
|
113
|
+
const gapSeparator = previous === void 0 ? "none" : separatorForGap(runGapPt(previous.entry.run, entry.run), Math.min(previous.entry.sizePt, entry.sizePt), wordGapEm, columnGapEm);
|
|
114
|
+
const pendingSpace = previous?.trailingSpace === true;
|
|
115
|
+
let separator = "none";
|
|
116
|
+
if (working.length > 0) separator = gapSeparator === "none" && (pendingSpace || tokens.leadingSpace) ? "space" : gapSeparator;
|
|
117
|
+
const widthPt = entry.run.widthPt;
|
|
118
|
+
const alongEndPt = entry.alongPt + (widthPt ?? 0);
|
|
119
|
+
tokens.words.forEach((word, index) => {
|
|
120
|
+
const last = working[working.length - 1];
|
|
121
|
+
if (index === 0 && separator === "none" && last !== void 0) {
|
|
122
|
+
last.text += word;
|
|
123
|
+
last.runs.push(entry.run);
|
|
124
|
+
last.measurable = last.measurable && tokens.whole && widthPt !== void 0;
|
|
125
|
+
last.minAlongPt = Math.min(last.minAlongPt, entry.alongPt);
|
|
126
|
+
last.maxAlongEndPt = Math.max(last.maxAlongEndPt, alongEndPt);
|
|
127
|
+
last.maxSizePt = Math.max(last.maxSizePt, entry.sizePt);
|
|
128
|
+
return;
|
|
129
|
+
}
|
|
130
|
+
working.push({
|
|
131
|
+
text: word,
|
|
132
|
+
separatorBefore: index === 0 ? separator : "space",
|
|
133
|
+
runs: [entry.run],
|
|
134
|
+
measurable: tokens.whole && widthPt !== void 0,
|
|
135
|
+
minAlongPt: entry.alongPt,
|
|
136
|
+
maxAlongEndPt: alongEndPt,
|
|
137
|
+
maxSizePt: entry.sizePt
|
|
138
|
+
});
|
|
139
|
+
});
|
|
140
|
+
previous = {
|
|
141
|
+
entry,
|
|
142
|
+
trailingSpace: tokens.trailingSpace
|
|
143
|
+
};
|
|
144
|
+
}
|
|
145
|
+
return working.map((word) => ({
|
|
146
|
+
text: word.text,
|
|
147
|
+
separatorBefore: word.separatorBefore,
|
|
148
|
+
runs: word.runs,
|
|
149
|
+
...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
|
|
150
|
+
}));
|
|
151
|
+
}
|
|
152
|
+
function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
153
|
+
const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
|
|
154
|
+
const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
|
|
155
|
+
let minAlongPt = Number.POSITIVE_INFINITY;
|
|
156
|
+
let maxAlongEndPt = Number.NEGATIVE_INFINITY;
|
|
157
|
+
let maxSizePt = 0;
|
|
158
|
+
let measurable = true;
|
|
159
|
+
for (const entry of line.entries) {
|
|
160
|
+
const widthPt = entry.run.widthPt;
|
|
161
|
+
if (widthPt === void 0) measurable = false;
|
|
162
|
+
minAlongPt = Math.min(minAlongPt, entry.alongPt);
|
|
163
|
+
maxAlongEndPt = Math.max(maxAlongEndPt, entry.alongPt + (widthPt ?? 0));
|
|
164
|
+
maxSizePt = Math.max(maxSizePt, entry.sizePt);
|
|
165
|
+
}
|
|
166
|
+
return {
|
|
167
|
+
text,
|
|
168
|
+
words,
|
|
169
|
+
baselineYPt: line.anchor.acrossPt,
|
|
170
|
+
rotationDeg,
|
|
171
|
+
...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
|
|
172
|
+
};
|
|
173
|
+
}
|
|
174
|
+
/**
|
|
175
|
+
* Groups positioned PDF text runs into lines of words.
|
|
176
|
+
*
|
|
177
|
+
* Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
|
|
178
|
+
*
|
|
179
|
+
* Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
|
|
180
|
+
*
|
|
181
|
+
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
182
|
+
*
|
|
183
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. Vertical writing modes are not recognised, because this package's content interpreter does not read a CMap's WMode and so reports vertically set text with horizontal advances (ExaDev/documents.js#1358), which puts the positions beyond anything grouping could repair. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
184
|
+
* @param runs - positioned text runs, in any order
|
|
185
|
+
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
186
|
+
* @returns the grouped lines, in reading order
|
|
187
|
+
*/
|
|
188
|
+
function groupPdfTextRuns(runs, options = {}) {
|
|
189
|
+
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
190
|
+
const wordGapEm = options.wordGapEm ?? .125;
|
|
191
|
+
const columnGapEm = options.columnGapEm ?? 1;
|
|
192
|
+
const byRotation = /* @__PURE__ */ new Map();
|
|
193
|
+
for (const run of runs) {
|
|
194
|
+
if (run.text.length === 0) continue;
|
|
195
|
+
const rotationDeg = normaliseRotationDeg(run.rotationDeg);
|
|
196
|
+
const bucket = byRotation.get(rotationDeg);
|
|
197
|
+
if (bucket === void 0) byRotation.set(rotationDeg, [run]);
|
|
198
|
+
else bucket.push(run);
|
|
199
|
+
}
|
|
200
|
+
const result = [];
|
|
201
|
+
for (const rotationDeg of [...byRotation.keys()].sort((a, b) => a - b)) {
|
|
202
|
+
const entries = (byRotation.get(rotationDeg) ?? []).map((run) => project(run, rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
|
|
203
|
+
for (const line of clusterIntoLines(entries, toleranceEm)) {
|
|
204
|
+
line.entries.sort((a, b) => a.alongPt - b.alongPt);
|
|
205
|
+
result.push(buildLine(line, rotationDeg, wordGapEm, columnGapEm));
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
return result;
|
|
209
|
+
}
|
|
210
|
+
//#endregion
|
|
211
|
+
export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, groupPdfTextRuns, runGapPt, runsShareBaseline };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pdf-codec.js",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.1.0",
|
|
4
4
|
"description": "Hand-written, dependency-minimal PDF codec: parses arbitrary real-world PDFs and generates new ones, built on its own codec-owned LayoutDocument item model and Zod 4 codecs.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"repository": {
|
|
@@ -72,8 +72,8 @@
|
|
|
72
72
|
],
|
|
73
73
|
"license": "MIT",
|
|
74
74
|
"dependencies": {
|
|
75
|
-
"byte-codec": "1.6.
|
|
76
|
-
"document-schema.js": "7.11.
|
|
75
|
+
"byte-codec": "1.6.3",
|
|
76
|
+
"document-schema.js": "7.11.5",
|
|
77
77
|
"fflate": "0.8.3",
|
|
78
78
|
"zod": "4.4.3"
|
|
79
79
|
},
|