pdf-codec.js 5.0.0 → 5.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -310,6 +310,31 @@ Reading a program's built-in encoding is what stops a symbol-encoded subset —
310
310
 
311
311
  **Where nothing states an answer, the answer is the replacement character plus a `text/unmapped-encoding` diagnostic, never a guess.** Two cases reach it in practice: a symbolic font with no embedded program and no `/ToUnicode`, and a subsetted font that both strips its glyph names and maps its glyphs only from private-use code points — a private-use code point identifies a glyph inside one font and says nothing about the character it draws, so it is treated as no answer rather than a wrong one.
312
312
 
313
+ ## Grouping text runs into lines and words
314
+
315
+ `readPdf` reports one positioned run per text-showing operator, and a PDF says nothing about lines or words: a line of prose can arrive as one run, as one run per word, or as one run per glyph, entirely at the producer's discretion. `groupPdfTextRuns` turns those runs back into lines of words, and `pdf-codec/text-group` serves it without the write half or the vendored fonts, alongside `pdf-codec/read`.
316
+
317
+ ```ts
318
+ import { readPdf } from "pdf-codec/read";
319
+ import { groupPdfTextRuns } from "pdf-codec/text-group";
320
+
321
+ const page = readPdf(bytes).pages[0];
322
+ const runs = page.items.filter((item) => item.kind === "text");
323
+ for (const line of groupPdfTextRuns(runs)) {
324
+ console.log(line.text);
325
+ }
326
+ ```
327
+
328
+ Two rules decide everything, and both were got wrong often enough by hand ([#1317](https://github.com/ExaDev/documents.js/issues/1317)) to be worth stating.
329
+
330
+ Two runs are on one line when their baselines sit within a fraction of an em of each other, measured against **the smaller of the two**. Taken from the larger, a 30pt heading's own window reaches the 9pt line beneath it, the merged line's runs sort by x into a sequence whose gaps are negative, a negative gap reads as one word, and the two lines concatenate into exactly the run-together text the grouping exists to prevent. The default fraction is two thirds, bounded on both sides by real typography: two consecutive lines are never closer than solid setting puts them, one full em, so it has to stay under 1; a superscript raised by at most a third of its parent's size at no less than half that size shifts by at most two thirds of its own em, so anything tighter cuts footnote markers off the line they annotate.
331
+
332
+ A run whose `widthPt` is absent states no advance, so where it ends is unknown and the gap after it is underivable. `runGapPt` reports `undefined` for that, and an undefined gap is **no evidence of a space**. Reading it as zero puts the previous run's end at its own start, turns its whole advance into an apparent gap, and produces "Com plete ly".
333
+
334
+ Gaps that are derivable become a word space at an eighth of an em (the narrowest genuine one: the standard fourteen faces set their space glyph at 250/1000 em, and justified setting compresses to no less than half of that) and a column boundary past a full em (wider than the em space, the widest single space character there is), so a table's row reads as `"North\t4.2m\t11%"` rather than as one sentence. `runsShareBaseline` and `runGapPt` are exported on their own for a consumer with its own pipeline around them, and every threshold is overridable.
335
+
336
+ Limits, each inherent to what a PDF states rather than to the implementation: grouping is by baseline alone with no page segmentation, so on a multi-column page whose columns sit at different vertical offsets a line of one can fall within tolerance of a line of the next and the two are reported as one line with a column boundary between them, which is the cut a caller needs (`document-outline.js`'s `segmentPdfRegions` finds the gutter first if you want them genuinely apart); text comes out in visual order, so a right-to-left script needs the Unicode Bidi Algorithm applied to the result; vertical writing modes are not recognised, because the content interpreter does not read a CMap's `WMode` ([#1358](https://github.com/ExaDev/documents.js/issues/1358)); and a word hyphenated across a line end stays split, hyphen intact, because rejoining it is a paragraph-level judgement this package leaves to its consumers.
337
+
313
338
  ## JBIG2 scope
314
339
 
315
340
  `src/image/jbig2*.ts` is a hand-written ITU-T T.88 decoder covering what real scanned PDFs actually contain.
package/dist/index.cjs CHANGED
@@ -19,8 +19,12 @@ const require_measure = require("./measure.cjs");
19
19
  const require_write = require("./write.cjs");
20
20
  const require_codec = require("./codec.cjs");
21
21
  const require_font_face = require("./font-face.cjs");
22
+ const require_text_group = require("./text-group.cjs");
22
23
  const require_raster = require("./raster.cjs");
24
+ exports.DEFAULT_BASELINE_TOLERANCE_EM = require_text_group.DEFAULT_BASELINE_TOLERANCE_EM;
25
+ exports.DEFAULT_COLUMN_GAP_EM = require_text_group.DEFAULT_COLUMN_GAP_EM;
23
26
  exports.DEFAULT_VERTICAL_METRIC_POLICY = require_measure.DEFAULT_VERTICAL_METRIC_POLICY;
27
+ exports.DEFAULT_WORD_GAP_EM = require_text_group.DEFAULT_WORD_GAP_EM;
24
28
  exports.FontFaceParseError = require_font_face.FontFaceParseError;
25
29
  exports.Jbig2ParseError = require_image_jbig2_errors.Jbig2ParseError;
26
30
  exports.Jbig2UnsupportedError = require_image_jbig2_errors.Jbig2UnsupportedError;
@@ -66,6 +70,7 @@ exports.decodeCcittFax = require_image_ccitt.decodeCcittFax;
66
70
  exports.decodeJbig2Embedded = require_image_jbig2.decodeJbig2Embedded;
67
71
  exports.decodeJpeg2000 = require_image_jpeg2000.decodeJpeg2000;
68
72
  exports.filterScanlines = require_image_png_filter.filterScanlines;
73
+ exports.groupPdfTextRuns = require_text_group.groupPdfTextRuns;
69
74
  exports.loadMathFont = require_math_font.loadMathFont;
70
75
  exports.looksLikeBareCodestream = require_image_jp2_boxes.looksLikeBareCodestream;
71
76
  exports.parseJp2Container = require_image_jp2_boxes.parseJp2Container;
@@ -76,6 +81,8 @@ exports.readPdf = require_read.readPdf;
76
81
  exports.renderPdfPage = require_raster.renderPdfPage;
77
82
  exports.resolveFaceWithRegistry = require_font_registry.resolveFaceWithRegistry;
78
83
  exports.resolveStandardFont = require_fonts.resolveStandardFont;
84
+ exports.runGapPt = require_text_group.runGapPt;
85
+ exports.runsShareBaseline = require_text_group.runsShareBaseline;
79
86
  exports.scaleMathStretchConstruction = require_math_stretch.scaleMathStretchConstruction;
80
87
  exports.unfilterScanlines = require_image_png_filter.unfilterScanlines;
81
88
  exports.writePdf = require_write.writePdf;
package/dist/index.d.cts CHANGED
@@ -23,7 +23,8 @@ import { a as MathGlyphPart, c as MathVariants, n as MathGlyphAssembly, o as Mat
23
23
  import { a as scaleMathStretchConstruction, i as assembleStretchyGlyph, n as MathStretchOptions, r as MathStretchPlacement, t as MathStretchConstruction } from "./math-stretch-C2JanDAz.cjs";
24
24
  import { LoadedMathFont, MathFont, MathFontDescriptorMetrics, loadMathFont } from "./math-font.cjs";
25
25
  import { DEFAULT_VERTICAL_METRIC_POLICY, FontMeasurerOptions, StandardFontMeasurerOptions, VerticalMetricPolicy, createFontMeasurer, createStandardFontMeasurer } from "./measure.cjs";
26
+ import { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline } from "./text-group.cjs";
26
27
  import { PageRasteriser, RasterDrawOp, RasterFillRectOp, RasterFillSpec, RasterImageOp, RasterMatrix, RasterPageGeometry, RasterPathOp, RasterPathSegment, RasterRegionPt, RasterStrokeSpec, RasterSubpath, RenderPdfPageOptions, renderPdfPage } from "./raster.cjs";
27
28
  import { FontRegistryOptions, FontSubstitution, MathAssembledGlyphs, MathBox, MathColor, MathFontMetrics, MathGlyphMetrics, MathGlyphPlacement, MathGlyphRun, MathLayoutItem, MathRule, MathStretchAxis, MathStretchGlyph, MathStretchResult, MathStroke, Point, PositionedFormula, ProvidedFont, StyledFragment, StyledRun, TextMeasurer, UnderlineMetrics, WrapOptions, WrappedLine } from "document-schema.js";
28
29
  export * from "byte-codec";
29
- export { type CcittFaxImage, type CcittFaxOptions, DEFAULT_VERTICAL_METRIC_POLICY, type EmbeddedFace, type EmbeddedFaceMetrics, type EmbeddedFaceSubstitution, type FontFace, FontFaceParseError, type FontMeasurerOptions, type FontMetrics, type FontRegistry, type FontRegistryOptions, type FontSubstitution, type GlyphInkBounds, type Jbig2DecodeOptions, type Jbig2Image, Jbig2ParseError, Jbig2UnsupportedError, type Jp2ChannelDefinition, type Jp2ColourSpace, type Jp2Container, type Jp2ImageHeader, type Jpeg2000ComponentMetadata, type Jpeg2000DecodeOptions, type Jpeg2000Image, type Jpeg2000Metadata, Jpeg2000ParseError, type Jpeg2000ProgressionOrder, type Jpeg2000QuantizationStyle, type Jpeg2000Transform, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotation, LayoutAnnotationQuad, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachment, LayoutAttachmentSchema, LayoutDestination, LayoutDestinationSchema, LayoutDestinationTarget, LayoutDestinationTargetSchema, LayoutDocument, LayoutDocumentSchema, LayoutEllipse, LayoutEllipseSchema, LayoutFormField, LayoutFormFieldSchema, LayoutFormWidget, LayoutFormWidgetSchema, LayoutImage, LayoutImageAsset, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLink, LayoutInternalLinkSchema, LayoutItem, LayoutItemSchema, LayoutLayer, LayoutLayerSchema, LayoutLine, LayoutLineSchema, LayoutLink, LayoutLinkSchema, LayoutOutlineItem, LayoutOutlineItemSchema, LayoutPage, LayoutPageSchema, LayoutPath, LayoutPathSchema, LayoutPathSegment, LayoutPathSegmentSchema, LayoutRect, LayoutRectSchema, LayoutStructureElement, LayoutStructureElementSchema, LayoutSubpath, LayoutSubpathSchema, LayoutText, LayoutTextSchema, type LoadedMathFont, type MathAssembledGlyphs, type MathBox, type MathColor, type MathFont, type MathFontDescriptorMetrics, type MathFontMetrics, type MathGlyphAssembly, type MathGlyphConstruction, type MathGlyphMetrics, type MathGlyphPart, type MathGlyphPlacement, type MathGlyphRun, type MathGlyphVariant, type MathLayoutItem, type MathRule, type MathStretchAxis, type MathStretchConstruction, type MathStretchGlyph, type MathStretchOptions, type MathStretchPlacement, type MathStretchResult, type MathStroke, type MathVariants, NOOP_DIAGNOSTIC_SINK, type PageRasteriser, PdfBytesSchema, type PdfDiagnostic, type PdfDiagnosticSeverity, type PdfDiagnosticSink, PdfEncryptedError, PdfEncryptionError, type PdfEncryptionOptions, type PdfEncryptionPermissions, type PdfEncryptionScheme, PdfParseError, PdfPasswordRequiredError, type Point, type PositionedFormula, type ProvidedFont, type RasterDrawOp, type RasterFillRectOp, type RasterFillSpec, type RasterImageOp, type RasterMatrix, type RasterPageGeometry, type RasterPathOp, type RasterPathSegment, type RasterRegionPt, type RasterStrokeSpec, type RasterSubpath, type ReadPdfOptions, type RenderPdfPageOptions, type ResolvedFace, type ResolvedFont, STANDARD_METRICS, type StandardFontMeasurerOptions, type StandardFontName, type StyledFragment, type StyledRun, type TextMeasurer, type UnderlineMetrics, type VerticalMetricPolicy, type WinAnsiSubstitution, type WrapOptions, type WrappedLine, type WritePdfOptions, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, scaleMathStretchConstruction, unfilterScanlines, writePdf };
30
+ export { type CcittFaxImage, type CcittFaxOptions, DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_VERTICAL_METRIC_POLICY, DEFAULT_WORD_GAP_EM, type EmbeddedFace, type EmbeddedFaceMetrics, type EmbeddedFaceSubstitution, type FontFace, FontFaceParseError, type FontMeasurerOptions, type FontMetrics, type FontRegistry, type FontRegistryOptions, type FontSubstitution, type GlyphInkBounds, type Jbig2DecodeOptions, type Jbig2Image, Jbig2ParseError, Jbig2UnsupportedError, type Jp2ChannelDefinition, type Jp2ColourSpace, type Jp2Container, type Jp2ImageHeader, type Jpeg2000ComponentMetadata, type Jpeg2000DecodeOptions, type Jpeg2000Image, type Jpeg2000Metadata, Jpeg2000ParseError, type Jpeg2000ProgressionOrder, type Jpeg2000QuantizationStyle, type Jpeg2000Transform, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotation, LayoutAnnotationQuad, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachment, LayoutAttachmentSchema, LayoutDestination, LayoutDestinationSchema, LayoutDestinationTarget, LayoutDestinationTargetSchema, LayoutDocument, LayoutDocumentSchema, LayoutEllipse, LayoutEllipseSchema, LayoutFormField, LayoutFormFieldSchema, LayoutFormWidget, LayoutFormWidgetSchema, LayoutImage, LayoutImageAsset, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLink, LayoutInternalLinkSchema, LayoutItem, LayoutItemSchema, LayoutLayer, LayoutLayerSchema, LayoutLine, LayoutLineSchema, LayoutLink, LayoutLinkSchema, LayoutOutlineItem, LayoutOutlineItemSchema, LayoutPage, LayoutPageSchema, LayoutPath, LayoutPathSchema, LayoutPathSegment, LayoutPathSegmentSchema, LayoutRect, LayoutRectSchema, LayoutStructureElement, LayoutStructureElementSchema, LayoutSubpath, LayoutSubpathSchema, LayoutText, LayoutTextSchema, type LoadedMathFont, type MathAssembledGlyphs, type MathBox, type MathColor, type MathFont, type MathFontDescriptorMetrics, type MathFontMetrics, type MathGlyphAssembly, type MathGlyphConstruction, type MathGlyphMetrics, type MathGlyphPart, type MathGlyphPlacement, type MathGlyphRun, type MathGlyphVariant, type MathLayoutItem, type MathRule, type MathStretchAxis, type MathStretchConstruction, type MathStretchGlyph, type MathStretchOptions, type MathStretchPlacement, type MathStretchResult, type MathStroke, type MathVariants, NOOP_DIAGNOSTIC_SINK, type PageRasteriser, PdfBytesSchema, type PdfDiagnostic, type PdfDiagnosticSeverity, type PdfDiagnosticSink, PdfEncryptedError, PdfEncryptionError, type PdfEncryptionOptions, type PdfEncryptionPermissions, type PdfEncryptionScheme, PdfParseError, PdfPasswordRequiredError, type PdfTextBox, type PdfTextGroupingOptions, type PdfTextLine, type PdfTextRunGeometry, type PdfTextWord, type PdfWordSeparator, type Point, type PositionedFormula, type ProvidedFont, type RasterDrawOp, type RasterFillRectOp, type RasterFillSpec, type RasterImageOp, type RasterMatrix, type RasterPageGeometry, type RasterPathOp, type RasterPathSegment, type RasterRegionPt, type RasterStrokeSpec, type RasterSubpath, type ReadPdfOptions, type RenderPdfPageOptions, type ResolvedFace, type ResolvedFont, STANDARD_METRICS, type StandardFontMeasurerOptions, type StandardFontName, type StyledFragment, type StyledRun, type TextMeasurer, type UnderlineMetrics, type VerticalMetricPolicy, type WinAnsiSubstitution, type WrapOptions, type WrappedLine, type WritePdfOptions, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, groupPdfTextRuns, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, runGapPt, runsShareBaseline, scaleMathStretchConstruction, unfilterScanlines, writePdf };
package/dist/index.d.ts CHANGED
@@ -23,7 +23,8 @@ import { a as MathGlyphPart, c as MathVariants, n as MathGlyphAssembly, o as Mat
23
23
  import { a as scaleMathStretchConstruction, i as assembleStretchyGlyph, n as MathStretchOptions, r as MathStretchPlacement, t as MathStretchConstruction } from "./math-stretch-BbC2o4Br.js";
24
24
  import { LoadedMathFont, MathFont, MathFontDescriptorMetrics, loadMathFont } from "./math-font.js";
25
25
  import { DEFAULT_VERTICAL_METRIC_POLICY, FontMeasurerOptions, StandardFontMeasurerOptions, VerticalMetricPolicy, createFontMeasurer, createStandardFontMeasurer } from "./measure.js";
26
+ import { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline } from "./text-group.js";
26
27
  import { PageRasteriser, RasterDrawOp, RasterFillRectOp, RasterFillSpec, RasterImageOp, RasterMatrix, RasterPageGeometry, RasterPathOp, RasterPathSegment, RasterRegionPt, RasterStrokeSpec, RasterSubpath, RenderPdfPageOptions, renderPdfPage } from "./raster.js";
27
28
  import { FontRegistryOptions, FontSubstitution, MathAssembledGlyphs, MathBox, MathColor, MathFontMetrics, MathGlyphMetrics, MathGlyphPlacement, MathGlyphRun, MathLayoutItem, MathRule, MathStretchAxis, MathStretchGlyph, MathStretchResult, MathStroke, Point, PositionedFormula, ProvidedFont, StyledFragment, StyledRun, TextMeasurer, UnderlineMetrics, WrapOptions, WrappedLine } from "document-schema.js";
28
29
  export * from "byte-codec";
29
- export { type CcittFaxImage, type CcittFaxOptions, DEFAULT_VERTICAL_METRIC_POLICY, type EmbeddedFace, type EmbeddedFaceMetrics, type EmbeddedFaceSubstitution, type FontFace, FontFaceParseError, type FontMeasurerOptions, type FontMetrics, type FontRegistry, type FontRegistryOptions, type FontSubstitution, type GlyphInkBounds, type Jbig2DecodeOptions, type Jbig2Image, Jbig2ParseError, Jbig2UnsupportedError, type Jp2ChannelDefinition, type Jp2ColourSpace, type Jp2Container, type Jp2ImageHeader, type Jpeg2000ComponentMetadata, type Jpeg2000DecodeOptions, type Jpeg2000Image, type Jpeg2000Metadata, Jpeg2000ParseError, type Jpeg2000ProgressionOrder, type Jpeg2000QuantizationStyle, type Jpeg2000Transform, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotation, LayoutAnnotationQuad, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachment, LayoutAttachmentSchema, LayoutDestination, LayoutDestinationSchema, LayoutDestinationTarget, LayoutDestinationTargetSchema, LayoutDocument, LayoutDocumentSchema, LayoutEllipse, LayoutEllipseSchema, LayoutFormField, LayoutFormFieldSchema, LayoutFormWidget, LayoutFormWidgetSchema, LayoutImage, LayoutImageAsset, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLink, LayoutInternalLinkSchema, LayoutItem, LayoutItemSchema, LayoutLayer, LayoutLayerSchema, LayoutLine, LayoutLineSchema, LayoutLink, LayoutLinkSchema, LayoutOutlineItem, LayoutOutlineItemSchema, LayoutPage, LayoutPageSchema, LayoutPath, LayoutPathSchema, LayoutPathSegment, LayoutPathSegmentSchema, LayoutRect, LayoutRectSchema, LayoutStructureElement, LayoutStructureElementSchema, LayoutSubpath, LayoutSubpathSchema, LayoutText, LayoutTextSchema, type LoadedMathFont, type MathAssembledGlyphs, type MathBox, type MathColor, type MathFont, type MathFontDescriptorMetrics, type MathFontMetrics, type MathGlyphAssembly, type MathGlyphConstruction, type MathGlyphMetrics, type MathGlyphPart, type MathGlyphPlacement, type MathGlyphRun, type MathGlyphVariant, type MathLayoutItem, type MathRule, type MathStretchAxis, type MathStretchConstruction, type MathStretchGlyph, type MathStretchOptions, type MathStretchPlacement, type MathStretchResult, type MathStroke, type MathVariants, NOOP_DIAGNOSTIC_SINK, type PageRasteriser, PdfBytesSchema, type PdfDiagnostic, type PdfDiagnosticSeverity, type PdfDiagnosticSink, PdfEncryptedError, PdfEncryptionError, type PdfEncryptionOptions, type PdfEncryptionPermissions, type PdfEncryptionScheme, PdfParseError, PdfPasswordRequiredError, type Point, type PositionedFormula, type ProvidedFont, type RasterDrawOp, type RasterFillRectOp, type RasterFillSpec, type RasterImageOp, type RasterMatrix, type RasterPageGeometry, type RasterPathOp, type RasterPathSegment, type RasterRegionPt, type RasterStrokeSpec, type RasterSubpath, type ReadPdfOptions, type RenderPdfPageOptions, type ResolvedFace, type ResolvedFont, STANDARD_METRICS, type StandardFontMeasurerOptions, type StandardFontName, type StyledFragment, type StyledRun, type TextMeasurer, type UnderlineMetrics, type VerticalMetricPolicy, type WinAnsiSubstitution, type WrapOptions, type WrappedLine, type WritePdfOptions, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, scaleMathStretchConstruction, unfilterScanlines, writePdf };
30
+ export { type CcittFaxImage, type CcittFaxOptions, DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_VERTICAL_METRIC_POLICY, DEFAULT_WORD_GAP_EM, type EmbeddedFace, type EmbeddedFaceMetrics, type EmbeddedFaceSubstitution, type FontFace, FontFaceParseError, type FontMeasurerOptions, type FontMetrics, type FontRegistry, type FontRegistryOptions, type FontSubstitution, type GlyphInkBounds, type Jbig2DecodeOptions, type Jbig2Image, Jbig2ParseError, Jbig2UnsupportedError, type Jp2ChannelDefinition, type Jp2ColourSpace, type Jp2Container, type Jp2ImageHeader, type Jpeg2000ComponentMetadata, type Jpeg2000DecodeOptions, type Jpeg2000Image, type Jpeg2000Metadata, Jpeg2000ParseError, type Jpeg2000ProgressionOrder, type Jpeg2000QuantizationStyle, type Jpeg2000Transform, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotation, LayoutAnnotationQuad, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachment, LayoutAttachmentSchema, LayoutDestination, LayoutDestinationSchema, LayoutDestinationTarget, LayoutDestinationTargetSchema, LayoutDocument, LayoutDocumentSchema, LayoutEllipse, LayoutEllipseSchema, LayoutFormField, LayoutFormFieldSchema, LayoutFormWidget, LayoutFormWidgetSchema, LayoutImage, LayoutImageAsset, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLink, LayoutInternalLinkSchema, LayoutItem, LayoutItemSchema, LayoutLayer, LayoutLayerSchema, LayoutLine, LayoutLineSchema, LayoutLink, LayoutLinkSchema, LayoutOutlineItem, LayoutOutlineItemSchema, LayoutPage, LayoutPageSchema, LayoutPath, LayoutPathSchema, LayoutPathSegment, LayoutPathSegmentSchema, LayoutRect, LayoutRectSchema, LayoutStructureElement, LayoutStructureElementSchema, LayoutSubpath, LayoutSubpathSchema, LayoutText, LayoutTextSchema, type LoadedMathFont, type MathAssembledGlyphs, type MathBox, type MathColor, type MathFont, type MathFontDescriptorMetrics, type MathFontMetrics, type MathGlyphAssembly, type MathGlyphConstruction, type MathGlyphMetrics, type MathGlyphPart, type MathGlyphPlacement, type MathGlyphRun, type MathGlyphVariant, type MathLayoutItem, type MathRule, type MathStretchAxis, type MathStretchConstruction, type MathStretchGlyph, type MathStretchOptions, type MathStretchPlacement, type MathStretchResult, type MathStroke, type MathVariants, NOOP_DIAGNOSTIC_SINK, type PageRasteriser, PdfBytesSchema, type PdfDiagnostic, type PdfDiagnosticSeverity, type PdfDiagnosticSink, PdfEncryptedError, PdfEncryptionError, type PdfEncryptionOptions, type PdfEncryptionPermissions, type PdfEncryptionScheme, PdfParseError, PdfPasswordRequiredError, type PdfTextBox, type PdfTextGroupingOptions, type PdfTextLine, type PdfTextRunGeometry, type PdfTextWord, type PdfWordSeparator, type Point, type PositionedFormula, type ProvidedFont, type RasterDrawOp, type RasterFillRectOp, type RasterFillSpec, type RasterImageOp, type RasterMatrix, type RasterPageGeometry, type RasterPathOp, type RasterPathSegment, type RasterRegionPt, type RasterStrokeSpec, type RasterSubpath, type ReadPdfOptions, type RenderPdfPageOptions, type ResolvedFace, type ResolvedFont, STANDARD_METRICS, type StandardFontMeasurerOptions, type StandardFontName, type StyledFragment, type StyledRun, type TextMeasurer, type UnderlineMetrics, type VerticalMetricPolicy, type WinAnsiSubstitution, type WrapOptions, type WrappedLine, type WritePdfOptions, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, groupPdfTextRuns, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, runGapPt, runsShareBaseline, scaleMathStretchConstruction, unfilterScanlines, writePdf };
package/dist/index.js CHANGED
@@ -18,6 +18,7 @@ import { DEFAULT_VERTICAL_METRIC_POLICY, createFontMeasurer, createStandardFontM
18
18
  import { writePdf } from "./write.js";
19
19
  import { PdfBytesSchema, pdfCodec } from "./codec.js";
20
20
  import { FontFaceParseError, readFontFace } from "./font-face.js";
21
+ import { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, groupPdfTextRuns, runGapPt, runsShareBaseline } from "./text-group.js";
21
22
  import { renderPdfPage } from "./raster.js";
22
23
  export * from "byte-codec";
23
- export { DEFAULT_VERTICAL_METRIC_POLICY, FontFaceParseError, Jbig2ParseError, Jbig2UnsupportedError, Jpeg2000ParseError, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachmentSchema, LayoutDestinationSchema, LayoutDestinationTargetSchema, LayoutDocumentSchema, LayoutEllipseSchema, LayoutFormFieldSchema, LayoutFormWidgetSchema, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLinkSchema, LayoutItemSchema, LayoutLayerSchema, LayoutLineSchema, LayoutLinkSchema, LayoutOutlineItemSchema, LayoutPageSchema, LayoutPathSchema, LayoutPathSegmentSchema, LayoutRectSchema, LayoutStructureElementSchema, LayoutSubpathSchema, LayoutTextSchema, NOOP_DIAGNOSTIC_SINK, PdfBytesSchema, PdfEncryptedError, PdfEncryptionError, PdfParseError, PdfPasswordRequiredError, STANDARD_METRICS, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, scaleMathStretchConstruction, unfilterScanlines, writePdf };
24
+ export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_VERTICAL_METRIC_POLICY, DEFAULT_WORD_GAP_EM, FontFaceParseError, Jbig2ParseError, Jbig2UnsupportedError, Jpeg2000ParseError, Jpeg2000UnsupportedError, LAYOUT_FORMAT_VERSION, LayoutAnnotationQuadSchema, LayoutAnnotationSchema, LayoutAttachmentSchema, LayoutDestinationSchema, LayoutDestinationTargetSchema, LayoutDocumentSchema, LayoutEllipseSchema, LayoutFormFieldSchema, LayoutFormWidgetSchema, LayoutImageAssetSchema, LayoutImageSchema, LayoutInternalLinkSchema, LayoutItemSchema, LayoutLayerSchema, LayoutLineSchema, LayoutLinkSchema, LayoutOutlineItemSchema, LayoutPageSchema, LayoutPathSchema, LayoutPathSegmentSchema, LayoutRectSchema, LayoutStructureElementSchema, LayoutSubpathSchema, LayoutTextSchema, NOOP_DIAGNOSTIC_SINK, PdfBytesSchema, PdfEncryptedError, PdfEncryptionError, PdfParseError, PdfPasswordRequiredError, STANDARD_METRICS, assembleStretchyGlyph, createFontMeasurer, createFontRegistry, createStandardFontMeasurer, decodeCcittFax, decodeJbig2Embedded, decodeJpeg2000, filterScanlines, groupPdfTextRuns, loadMathFont, looksLikeBareCodestream, parseJp2Container, pdfCodec, readFontFace, readJpeg2000Metadata, readPdf, renderPdfPage, resolveFaceWithRegistry, resolveStandardFont, runGapPt, runsShareBaseline, scaleMathStretchConstruction, unfilterScanlines, writePdf };
@@ -0,0 +1,217 @@
1
+ Object.defineProperty(exports, Symbol.toStringTag, { value: "Module" });
2
+ //#region src/text-group.ts
3
+ const DEFAULT_BASELINE_TOLERANCE_EM = 2 / 3;
4
+ const DEFAULT_WORD_GAP_EM = 1 / 8;
5
+ const DEFAULT_COLUMN_GAP_EM = 1;
6
+ const ROTATION_NOISE_DEG = 1e-6;
7
+ const DEGREES_PER_TURN = 360;
8
+ function normaliseRotationDeg(rotationDeg) {
9
+ return (Math.round((rotationDeg ?? 0) / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
10
+ }
11
+ const RADIANS_PER_TURN = 2 * Math.PI;
12
+ function baselineAxis(rotationDeg) {
13
+ const radians = rotationDeg * RADIANS_PER_TURN / DEGREES_PER_TURN;
14
+ return {
15
+ cos: Math.cos(radians),
16
+ sin: Math.sin(radians)
17
+ };
18
+ }
19
+ function project(run, rotationDeg) {
20
+ const { cos, sin } = baselineAxis(rotationDeg);
21
+ return {
22
+ run,
23
+ alongPt: run.xPt * cos + run.yPt * sin,
24
+ acrossPt: run.yPt * cos - run.xPt * sin,
25
+ sizePt: run.sizePt
26
+ };
27
+ }
28
+ function unproject(alongPt, acrossPt, rotationDeg) {
29
+ const { cos, sin } = baselineAxis(rotationDeg);
30
+ return {
31
+ xPt: alongPt * cos - acrossPt * sin,
32
+ yPt: alongPt * sin + acrossPt * cos
33
+ };
34
+ }
35
+ /**
36
+ * Whether two positioned runs sit on one visual line.
37
+ *
38
+ * The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
39
+ * @param a - one run
40
+ * @param b - the other run
41
+ * @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
42
+ * @returns true when the two baselines are within tolerance of each other
43
+ */
44
+ function runsShareBaseline(a, b, options = {}) {
45
+ const rotationDeg = normaliseRotationDeg(a.rotationDeg);
46
+ if (rotationDeg !== normaliseRotationDeg(b.rotationDeg)) return false;
47
+ const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
48
+ return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
49
+ }
50
+ /**
51
+ * The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
52
+ *
53
+ * Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
54
+ * @param previous - the run to the left, in reading order
55
+ * @param next - the run that follows it
56
+ * @returns the gap in points, or undefined when it cannot be derived
57
+ */
58
+ function runGapPt(previous, next) {
59
+ const rotationDeg = normaliseRotationDeg(previous.rotationDeg);
60
+ if (rotationDeg !== normaliseRotationDeg(next.rotationDeg)) return;
61
+ const previousWidthPt = previous.widthPt;
62
+ if (previousWidthPt === void 0) return;
63
+ return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
64
+ }
65
+ function clusterIntoLines(entries, toleranceEm) {
66
+ const lines = [];
67
+ for (const entry of entries) {
68
+ let nearest;
69
+ for (const line of lines) if (runsShareBaseline(line.anchor.run, entry.run, { baselineToleranceEm: toleranceEm })) nearest = line;
70
+ if (nearest === void 0) lines.push({
71
+ anchor: entry,
72
+ entries: [entry]
73
+ });
74
+ else nearest.entries.push(entry);
75
+ }
76
+ return lines;
77
+ }
78
+ function tokeniseRun(text) {
79
+ const words = text.match(/\S+/g) ?? [];
80
+ const leadingSpace = /^\s/.test(text);
81
+ const trailingSpace = /\s$/.test(text);
82
+ return {
83
+ words,
84
+ leadingSpace,
85
+ trailingSpace,
86
+ whole: words.length === 1 && !leadingSpace && !trailingSpace
87
+ };
88
+ }
89
+ const SEPARATOR_TEXT = {
90
+ none: "",
91
+ space: " ",
92
+ column: " "
93
+ };
94
+ function separatorForGap(gapPt, smallerSizePt, wordGapEm, columnGapEm) {
95
+ if (gapPt === void 0) return "none";
96
+ if (gapPt >= columnGapEm * smallerSizePt) return "column";
97
+ if (gapPt >= wordGapEm * smallerSizePt) return "space";
98
+ return "none";
99
+ }
100
+ function boundsOf(minAlongPt, maxAlongEndPt, acrossPt, maxSizePt, rotationDeg) {
101
+ const origin = unproject(minAlongPt, acrossPt, rotationDeg);
102
+ return {
103
+ xPt: origin.xPt,
104
+ yPt: origin.yPt,
105
+ widthPt: maxAlongEndPt - minAlongPt,
106
+ heightPt: maxSizePt
107
+ };
108
+ }
109
+ function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
110
+ const working = [];
111
+ let previous;
112
+ for (const entry of line.entries) {
113
+ const tokens = tokeniseRun(entry.run.text);
114
+ const gapSeparator = previous === void 0 ? "none" : separatorForGap(runGapPt(previous.entry.run, entry.run), Math.min(previous.entry.sizePt, entry.sizePt), wordGapEm, columnGapEm);
115
+ const pendingSpace = previous?.trailingSpace === true;
116
+ let separator = "none";
117
+ if (working.length > 0) separator = gapSeparator === "none" && (pendingSpace || tokens.leadingSpace) ? "space" : gapSeparator;
118
+ const widthPt = entry.run.widthPt;
119
+ const alongEndPt = entry.alongPt + (widthPt ?? 0);
120
+ tokens.words.forEach((word, index) => {
121
+ const last = working[working.length - 1];
122
+ if (index === 0 && separator === "none" && last !== void 0) {
123
+ last.text += word;
124
+ last.runs.push(entry.run);
125
+ last.measurable = last.measurable && tokens.whole && widthPt !== void 0;
126
+ last.minAlongPt = Math.min(last.minAlongPt, entry.alongPt);
127
+ last.maxAlongEndPt = Math.max(last.maxAlongEndPt, alongEndPt);
128
+ last.maxSizePt = Math.max(last.maxSizePt, entry.sizePt);
129
+ return;
130
+ }
131
+ working.push({
132
+ text: word,
133
+ separatorBefore: index === 0 ? separator : "space",
134
+ runs: [entry.run],
135
+ measurable: tokens.whole && widthPt !== void 0,
136
+ minAlongPt: entry.alongPt,
137
+ maxAlongEndPt: alongEndPt,
138
+ maxSizePt: entry.sizePt
139
+ });
140
+ });
141
+ previous = {
142
+ entry,
143
+ trailingSpace: tokens.trailingSpace
144
+ };
145
+ }
146
+ return working.map((word) => ({
147
+ text: word.text,
148
+ separatorBefore: word.separatorBefore,
149
+ runs: word.runs,
150
+ ...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
151
+ }));
152
+ }
153
+ function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
154
+ const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
155
+ const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
156
+ let minAlongPt = Number.POSITIVE_INFINITY;
157
+ let maxAlongEndPt = Number.NEGATIVE_INFINITY;
158
+ let maxSizePt = 0;
159
+ let measurable = true;
160
+ for (const entry of line.entries) {
161
+ const widthPt = entry.run.widthPt;
162
+ if (widthPt === void 0) measurable = false;
163
+ minAlongPt = Math.min(minAlongPt, entry.alongPt);
164
+ maxAlongEndPt = Math.max(maxAlongEndPt, entry.alongPt + (widthPt ?? 0));
165
+ maxSizePt = Math.max(maxSizePt, entry.sizePt);
166
+ }
167
+ return {
168
+ text,
169
+ words,
170
+ baselineYPt: line.anchor.acrossPt,
171
+ rotationDeg,
172
+ ...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
173
+ };
174
+ }
175
+ /**
176
+ * Groups positioned PDF text runs into lines of words.
177
+ *
178
+ * Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
179
+ *
180
+ * Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
181
+ *
182
+ * Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
183
+ *
184
+ * Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. Vertical writing modes are not recognised, because this package's content interpreter does not read a CMap's WMode and so reports vertically set text with horizontal advances (ExaDev/documents.js#1358), which puts the positions beyond anything grouping could repair. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
185
+ * @param runs - positioned text runs, in any order
186
+ * @param options - tolerance overrides; each defaults to the exported constant of the same name
187
+ * @returns the grouped lines, in reading order
188
+ */
189
+ function groupPdfTextRuns(runs, options = {}) {
190
+ const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
191
+ const wordGapEm = options.wordGapEm ?? .125;
192
+ const columnGapEm = options.columnGapEm ?? 1;
193
+ const byRotation = /* @__PURE__ */ new Map();
194
+ for (const run of runs) {
195
+ if (run.text.length === 0) continue;
196
+ const rotationDeg = normaliseRotationDeg(run.rotationDeg);
197
+ const bucket = byRotation.get(rotationDeg);
198
+ if (bucket === void 0) byRotation.set(rotationDeg, [run]);
199
+ else bucket.push(run);
200
+ }
201
+ const result = [];
202
+ for (const rotationDeg of [...byRotation.keys()].sort((a, b) => a - b)) {
203
+ const entries = (byRotation.get(rotationDeg) ?? []).map((run) => project(run, rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
204
+ for (const line of clusterIntoLines(entries, toleranceEm)) {
205
+ line.entries.sort((a, b) => a.alongPt - b.alongPt);
206
+ result.push(buildLine(line, rotationDeg, wordGapEm, columnGapEm));
207
+ }
208
+ }
209
+ return result;
210
+ }
211
+ //#endregion
212
+ exports.DEFAULT_BASELINE_TOLERANCE_EM = DEFAULT_BASELINE_TOLERANCE_EM;
213
+ exports.DEFAULT_COLUMN_GAP_EM = DEFAULT_COLUMN_GAP_EM;
214
+ exports.DEFAULT_WORD_GAP_EM = DEFAULT_WORD_GAP_EM;
215
+ exports.groupPdfTextRuns = groupPdfTextRuns;
216
+ exports.runGapPt = runGapPt;
217
+ exports.runsShareBaseline = runsShareBaseline;
@@ -0,0 +1,73 @@
1
+ //#region src/text-group.d.ts
2
+ interface PdfTextRunGeometry {
3
+ readonly text: string;
4
+ readonly xPt: number;
5
+ readonly yPt: number;
6
+ readonly sizePt: number;
7
+ readonly widthPt?: number;
8
+ readonly rotationDeg?: number;
9
+ }
10
+ interface PdfTextBox {
11
+ readonly xPt: number;
12
+ readonly yPt: number;
13
+ readonly widthPt: number;
14
+ readonly heightPt: number;
15
+ }
16
+ type PdfWordSeparator = "none" | "space" | "column";
17
+ interface PdfTextWord<TRun extends PdfTextRunGeometry> {
18
+ readonly text: string;
19
+ readonly separatorBefore: PdfWordSeparator;
20
+ readonly runs: readonly TRun[];
21
+ readonly bounds?: PdfTextBox;
22
+ }
23
+ interface PdfTextLine<TRun extends PdfTextRunGeometry> {
24
+ readonly text: string;
25
+ readonly words: readonly PdfTextWord<TRun>[];
26
+ readonly baselineYPt: number;
27
+ readonly rotationDeg: number;
28
+ readonly bounds?: PdfTextBox;
29
+ }
30
+ interface PdfTextGroupingOptions {
31
+ readonly baselineToleranceEm?: number;
32
+ readonly wordGapEm?: number;
33
+ readonly columnGapEm?: number;
34
+ }
35
+ declare const DEFAULT_BASELINE_TOLERANCE_EM: number;
36
+ declare const DEFAULT_WORD_GAP_EM: number;
37
+ declare const DEFAULT_COLUMN_GAP_EM = 1;
38
+ /**
39
+ * Whether two positioned runs sit on one visual line.
40
+ *
41
+ * The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
42
+ * @param a - one run
43
+ * @param b - the other run
44
+ * @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
45
+ * @returns true when the two baselines are within tolerance of each other
46
+ */
47
+ declare function runsShareBaseline(a: PdfTextRunGeometry, b: PdfTextRunGeometry, options?: PdfTextGroupingOptions): boolean;
48
+ /**
49
+ * The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
50
+ *
51
+ * Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
52
+ * @param previous - the run to the left, in reading order
53
+ * @param next - the run that follows it
54
+ * @returns the gap in points, or undefined when it cannot be derived
55
+ */
56
+ declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry): number | undefined;
57
+ /**
58
+ * Groups positioned PDF text runs into lines of words.
59
+ *
60
+ * Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
61
+ *
62
+ * Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
63
+ *
64
+ * Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
65
+ *
66
+ * Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. Vertical writing modes are not recognised, because this package's content interpreter does not read a CMap's WMode and so reports vertically set text with horizontal advances (ExaDev/documents.js#1358), which puts the positions beyond anything grouping could repair. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
67
+ * @param runs - positioned text runs, in any order
68
+ * @param options - tolerance overrides; each defaults to the exported constant of the same name
69
+ * @returns the grouped lines, in reading order
70
+ */
71
+ declare function groupPdfTextRuns<TRun extends PdfTextRunGeometry>(runs: readonly TRun[], options?: PdfTextGroupingOptions): PdfTextLine<TRun>[];
72
+ //#endregion
73
+ export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline };
@@ -0,0 +1,73 @@
1
+ //#region src/text-group.d.ts
2
+ interface PdfTextRunGeometry {
3
+ readonly text: string;
4
+ readonly xPt: number;
5
+ readonly yPt: number;
6
+ readonly sizePt: number;
7
+ readonly widthPt?: number;
8
+ readonly rotationDeg?: number;
9
+ }
10
+ interface PdfTextBox {
11
+ readonly xPt: number;
12
+ readonly yPt: number;
13
+ readonly widthPt: number;
14
+ readonly heightPt: number;
15
+ }
16
+ type PdfWordSeparator = "none" | "space" | "column";
17
+ interface PdfTextWord<TRun extends PdfTextRunGeometry> {
18
+ readonly text: string;
19
+ readonly separatorBefore: PdfWordSeparator;
20
+ readonly runs: readonly TRun[];
21
+ readonly bounds?: PdfTextBox;
22
+ }
23
+ interface PdfTextLine<TRun extends PdfTextRunGeometry> {
24
+ readonly text: string;
25
+ readonly words: readonly PdfTextWord<TRun>[];
26
+ readonly baselineYPt: number;
27
+ readonly rotationDeg: number;
28
+ readonly bounds?: PdfTextBox;
29
+ }
30
+ interface PdfTextGroupingOptions {
31
+ readonly baselineToleranceEm?: number;
32
+ readonly wordGapEm?: number;
33
+ readonly columnGapEm?: number;
34
+ }
35
+ declare const DEFAULT_BASELINE_TOLERANCE_EM: number;
36
+ declare const DEFAULT_WORD_GAP_EM: number;
37
+ declare const DEFAULT_COLUMN_GAP_EM = 1;
38
+ /**
39
+ * Whether two positioned runs sit on one visual line.
40
+ *
41
+ * The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
42
+ * @param a - one run
43
+ * @param b - the other run
44
+ * @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
45
+ * @returns true when the two baselines are within tolerance of each other
46
+ */
47
+ declare function runsShareBaseline(a: PdfTextRunGeometry, b: PdfTextRunGeometry, options?: PdfTextGroupingOptions): boolean;
48
+ /**
49
+ * The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
50
+ *
51
+ * Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
52
+ * @param previous - the run to the left, in reading order
53
+ * @param next - the run that follows it
54
+ * @returns the gap in points, or undefined when it cannot be derived
55
+ */
56
+ declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry): number | undefined;
57
+ /**
58
+ * Groups positioned PDF text runs into lines of words.
59
+ *
60
+ * Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
61
+ *
62
+ * Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
63
+ *
64
+ * Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
65
+ *
66
+ * Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. Vertical writing modes are not recognised, because this package's content interpreter does not read a CMap's WMode and so reports vertically set text with horizontal advances (ExaDev/documents.js#1358), which puts the positions beyond anything grouping could repair. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
67
+ * @param runs - positioned text runs, in any order
68
+ * @param options - tolerance overrides; each defaults to the exported constant of the same name
69
+ * @returns the grouped lines, in reading order
70
+ */
71
+ declare function groupPdfTextRuns<TRun extends PdfTextRunGeometry>(runs: readonly TRun[], options?: PdfTextGroupingOptions): PdfTextLine<TRun>[];
72
+ //#endregion
73
+ export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline };
@@ -0,0 +1,211 @@
1
+ //#region src/text-group.ts
2
+ const DEFAULT_BASELINE_TOLERANCE_EM = 2 / 3;
3
+ const DEFAULT_WORD_GAP_EM = 1 / 8;
4
+ const DEFAULT_COLUMN_GAP_EM = 1;
5
+ const ROTATION_NOISE_DEG = 1e-6;
6
+ const DEGREES_PER_TURN = 360;
7
+ function normaliseRotationDeg(rotationDeg) {
8
+ return (Math.round((rotationDeg ?? 0) / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
9
+ }
10
+ const RADIANS_PER_TURN = 2 * Math.PI;
11
+ function baselineAxis(rotationDeg) {
12
+ const radians = rotationDeg * RADIANS_PER_TURN / DEGREES_PER_TURN;
13
+ return {
14
+ cos: Math.cos(radians),
15
+ sin: Math.sin(radians)
16
+ };
17
+ }
18
+ function project(run, rotationDeg) {
19
+ const { cos, sin } = baselineAxis(rotationDeg);
20
+ return {
21
+ run,
22
+ alongPt: run.xPt * cos + run.yPt * sin,
23
+ acrossPt: run.yPt * cos - run.xPt * sin,
24
+ sizePt: run.sizePt
25
+ };
26
+ }
27
+ function unproject(alongPt, acrossPt, rotationDeg) {
28
+ const { cos, sin } = baselineAxis(rotationDeg);
29
+ return {
30
+ xPt: alongPt * cos - acrossPt * sin,
31
+ yPt: alongPt * sin + acrossPt * cos
32
+ };
33
+ }
34
+ /**
35
+ * Whether two positioned runs sit on one visual line.
36
+ *
37
+ * The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
38
+ * @param a - one run
39
+ * @param b - the other run
40
+ * @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
41
+ * @returns true when the two baselines are within tolerance of each other
42
+ */
43
+ function runsShareBaseline(a, b, options = {}) {
44
+ const rotationDeg = normaliseRotationDeg(a.rotationDeg);
45
+ if (rotationDeg !== normaliseRotationDeg(b.rotationDeg)) return false;
46
+ const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
47
+ return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
48
+ }
49
+ /**
50
+ * The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
51
+ *
52
+ * Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
53
+ * @param previous - the run to the left, in reading order
54
+ * @param next - the run that follows it
55
+ * @returns the gap in points, or undefined when it cannot be derived
56
+ */
57
+ function runGapPt(previous, next) {
58
+ const rotationDeg = normaliseRotationDeg(previous.rotationDeg);
59
+ if (rotationDeg !== normaliseRotationDeg(next.rotationDeg)) return;
60
+ const previousWidthPt = previous.widthPt;
61
+ if (previousWidthPt === void 0) return;
62
+ return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
63
+ }
64
+ function clusterIntoLines(entries, toleranceEm) {
65
+ const lines = [];
66
+ for (const entry of entries) {
67
+ let nearest;
68
+ for (const line of lines) if (runsShareBaseline(line.anchor.run, entry.run, { baselineToleranceEm: toleranceEm })) nearest = line;
69
+ if (nearest === void 0) lines.push({
70
+ anchor: entry,
71
+ entries: [entry]
72
+ });
73
+ else nearest.entries.push(entry);
74
+ }
75
+ return lines;
76
+ }
77
+ function tokeniseRun(text) {
78
+ const words = text.match(/\S+/g) ?? [];
79
+ const leadingSpace = /^\s/.test(text);
80
+ const trailingSpace = /\s$/.test(text);
81
+ return {
82
+ words,
83
+ leadingSpace,
84
+ trailingSpace,
85
+ whole: words.length === 1 && !leadingSpace && !trailingSpace
86
+ };
87
+ }
88
+ const SEPARATOR_TEXT = {
89
+ none: "",
90
+ space: " ",
91
+ column: " "
92
+ };
93
+ function separatorForGap(gapPt, smallerSizePt, wordGapEm, columnGapEm) {
94
+ if (gapPt === void 0) return "none";
95
+ if (gapPt >= columnGapEm * smallerSizePt) return "column";
96
+ if (gapPt >= wordGapEm * smallerSizePt) return "space";
97
+ return "none";
98
+ }
99
+ function boundsOf(minAlongPt, maxAlongEndPt, acrossPt, maxSizePt, rotationDeg) {
100
+ const origin = unproject(minAlongPt, acrossPt, rotationDeg);
101
+ return {
102
+ xPt: origin.xPt,
103
+ yPt: origin.yPt,
104
+ widthPt: maxAlongEndPt - minAlongPt,
105
+ heightPt: maxSizePt
106
+ };
107
+ }
108
+ function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
109
+ const working = [];
110
+ let previous;
111
+ for (const entry of line.entries) {
112
+ const tokens = tokeniseRun(entry.run.text);
113
+ const gapSeparator = previous === void 0 ? "none" : separatorForGap(runGapPt(previous.entry.run, entry.run), Math.min(previous.entry.sizePt, entry.sizePt), wordGapEm, columnGapEm);
114
+ const pendingSpace = previous?.trailingSpace === true;
115
+ let separator = "none";
116
+ if (working.length > 0) separator = gapSeparator === "none" && (pendingSpace || tokens.leadingSpace) ? "space" : gapSeparator;
117
+ const widthPt = entry.run.widthPt;
118
+ const alongEndPt = entry.alongPt + (widthPt ?? 0);
119
+ tokens.words.forEach((word, index) => {
120
+ const last = working[working.length - 1];
121
+ if (index === 0 && separator === "none" && last !== void 0) {
122
+ last.text += word;
123
+ last.runs.push(entry.run);
124
+ last.measurable = last.measurable && tokens.whole && widthPt !== void 0;
125
+ last.minAlongPt = Math.min(last.minAlongPt, entry.alongPt);
126
+ last.maxAlongEndPt = Math.max(last.maxAlongEndPt, alongEndPt);
127
+ last.maxSizePt = Math.max(last.maxSizePt, entry.sizePt);
128
+ return;
129
+ }
130
+ working.push({
131
+ text: word,
132
+ separatorBefore: index === 0 ? separator : "space",
133
+ runs: [entry.run],
134
+ measurable: tokens.whole && widthPt !== void 0,
135
+ minAlongPt: entry.alongPt,
136
+ maxAlongEndPt: alongEndPt,
137
+ maxSizePt: entry.sizePt
138
+ });
139
+ });
140
+ previous = {
141
+ entry,
142
+ trailingSpace: tokens.trailingSpace
143
+ };
144
+ }
145
+ return working.map((word) => ({
146
+ text: word.text,
147
+ separatorBefore: word.separatorBefore,
148
+ runs: word.runs,
149
+ ...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
150
+ }));
151
+ }
152
+ function buildLine(line, rotationDeg, wordGapEm, columnGapEm) {
153
+ const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
154
+ const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
155
+ let minAlongPt = Number.POSITIVE_INFINITY;
156
+ let maxAlongEndPt = Number.NEGATIVE_INFINITY;
157
+ let maxSizePt = 0;
158
+ let measurable = true;
159
+ for (const entry of line.entries) {
160
+ const widthPt = entry.run.widthPt;
161
+ if (widthPt === void 0) measurable = false;
162
+ minAlongPt = Math.min(minAlongPt, entry.alongPt);
163
+ maxAlongEndPt = Math.max(maxAlongEndPt, entry.alongPt + (widthPt ?? 0));
164
+ maxSizePt = Math.max(maxSizePt, entry.sizePt);
165
+ }
166
+ return {
167
+ text,
168
+ words,
169
+ baselineYPt: line.anchor.acrossPt,
170
+ rotationDeg,
171
+ ...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
172
+ };
173
+ }
174
+ /**
175
+ * Groups positioned PDF text runs into lines of words.
176
+ *
177
+ * Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
178
+ *
179
+ * Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
180
+ *
181
+ * Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
182
+ *
183
+ * Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. Vertical writing modes are not recognised, because this package's content interpreter does not read a CMap's WMode and so reports vertically set text with horizontal advances (ExaDev/documents.js#1358), which puts the positions beyond anything grouping could repair. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
184
+ * @param runs - positioned text runs, in any order
185
+ * @param options - tolerance overrides; each defaults to the exported constant of the same name
186
+ * @returns the grouped lines, in reading order
187
+ */
188
+ function groupPdfTextRuns(runs, options = {}) {
189
+ const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
190
+ const wordGapEm = options.wordGapEm ?? .125;
191
+ const columnGapEm = options.columnGapEm ?? 1;
192
+ const byRotation = /* @__PURE__ */ new Map();
193
+ for (const run of runs) {
194
+ if (run.text.length === 0) continue;
195
+ const rotationDeg = normaliseRotationDeg(run.rotationDeg);
196
+ const bucket = byRotation.get(rotationDeg);
197
+ if (bucket === void 0) byRotation.set(rotationDeg, [run]);
198
+ else bucket.push(run);
199
+ }
200
+ const result = [];
201
+ for (const rotationDeg of [...byRotation.keys()].sort((a, b) => a - b)) {
202
+ const entries = (byRotation.get(rotationDeg) ?? []).map((run) => project(run, rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
203
+ for (const line of clusterIntoLines(entries, toleranceEm)) {
204
+ line.entries.sort((a, b) => a.alongPt - b.alongPt);
205
+ result.push(buildLine(line, rotationDeg, wordGapEm, columnGapEm));
206
+ }
207
+ }
208
+ return result;
209
+ }
210
+ //#endregion
211
+ export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, groupPdfTextRuns, runGapPt, runsShareBaseline };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pdf-codec.js",
3
- "version": "5.0.0",
3
+ "version": "5.1.0",
4
4
  "description": "Hand-written, dependency-minimal PDF codec: parses arbitrary real-world PDFs and generates new ones, built on its own codec-owned LayoutDocument item model and Zod 4 codecs.",
5
5
  "type": "module",
6
6
  "repository": {
@@ -72,8 +72,8 @@
72
72
  ],
73
73
  "license": "MIT",
74
74
  "dependencies": {
75
- "byte-codec": "1.6.2",
76
- "document-schema.js": "7.11.4",
75
+ "byte-codec": "1.6.3",
76
+ "document-schema.js": "7.11.5",
77
77
  "fflate": "0.8.3",
78
78
  "zod": "4.4.3"
79
79
  },