@json-to-office/jto-ops 3.3.0 → 4.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import { FontRuntimeOpts, ServicesConfig, GenerationWarning, RendererStatus, PptxBatchRasterizer, PptxRasterizer, ResolvedFont } from '@json-to-office/shared';
2
- import { QualityProfile, QualityPolicy, PreparedDocument, QualityAnalysis } from '@json-to-office/quality';
2
+ import { QualityProfile, QualityPolicy, PreparedDocument, QualityAnalysis, QualityDiagnostic } from '@json-to-office/quality';
3
3
 
4
4
  type FormatName = 'docx' | 'pptx';
5
5
  interface GeneratorOptions {
@@ -322,15 +322,30 @@ interface PdfTextWord {
322
322
  xMax: number;
323
323
  yMax: number;
324
324
  }
325
+ /**
326
+ * One line as poppler's layout analysis grouped it (`-bbox-layout` only):
327
+ * its box plus the indices into `words` of the words on it.
328
+ */
329
+ interface PdfTextLine {
330
+ xMin: number;
331
+ yMin: number;
332
+ xMax: number;
333
+ yMax: number;
334
+ words: number[];
335
+ }
325
336
  /** One PDF page: its size in points plus every word poppler segmented. */
326
337
  interface PdfTextPage {
327
338
  widthPt: number;
328
339
  heightPt: number;
340
+ /** Every word in stream order, whichever mode produced the page. */
329
341
  words: PdfTextWord[];
342
+ /** Line grouping; empty for plain `-bbox` output, which groups nothing. */
343
+ lines: PdfTextLine[];
330
344
  }
331
345
  /**
332
- * Parse `pdftotext -bbox` output (XHTML with `<page>`/`<word>` elements).
333
- * Pure — feed it a captured document for tests, or the runner's stdout.
346
+ * Parse `pdftotext -bbox` or `-bbox-layout` output (XHTML with `<page>`,
347
+ * optionally `<line>`, and `<word>` elements). Pure — feed it a captured
348
+ * document for tests, or the runner's stdout.
334
349
  */
335
350
  declare function parsePdfTextBbox(bboxXml: string): PdfTextPage[];
336
351
  /** True when a `pdftotext` binary is reachable — lets harnesses skip early. */
@@ -338,11 +353,226 @@ declare function pdftotextAvailable(): Promise<boolean>;
338
353
  /**
339
354
  * Extract per-word text geometry from a PDF on disk. One pdftotext spawn,
340
355
  * output streamed through stdout — nothing else touches the filesystem.
356
+ * `layout: true` asks poppler for its line grouping too (`-bbox-layout`).
341
357
  */
342
358
  declare function extractPdfTextGeometry(pdfPath: string, options?: {
343
359
  timeoutMs?: number;
360
+ layout?: boolean;
344
361
  }): Promise<PdfTextPage[]>;
345
362
 
363
+ /**
364
+ * Font names from a rendered PDF — the other half of substitution detection.
365
+ *
366
+ * `pdftotext -bbox` says where every word landed but not which face drew it.
367
+ * `pdffonts` (poppler, next to pdftotext) lists every font the PDF embeds or
368
+ * references; LibreOffice embeds the face it actually used, so a requested
369
+ * family that never appears in that list was substituted on the way.
370
+ *
371
+ * Names are PostScript-style: a subset tag (`BAAAAA+`), the family with
372
+ * spaces removed, and a style suffix (`-Bold`, `Medium-Regular`). Comparison
373
+ * therefore happens on a folded form — lowercase alphanumerics only — where
374
+ * "Space Grotesk" and `CAAAAA+SpaceGrotesk-Regular` meet as a prefix match.
375
+ */
376
+ /** One font row as `pdffonts` prints it. */
377
+ interface PdfFontInfo {
378
+ /** The name as printed, subset tag included. */
379
+ name: string;
380
+ /** The name without its subset tag: `DMSans-Regular`. */
381
+ baseName: string;
382
+ type: string;
383
+ embedded: boolean;
384
+ }
385
+ /**
386
+ * Parse `pdffonts` output. Pure; the header and rule lines are skipped and
387
+ * a row that does not fit the column layout is ignored rather than guessed.
388
+ */
389
+ declare function parsePdfFonts(stdout: string): PdfFontInfo[];
390
+ /**
391
+ * Whether a requested family is present among the rendered fonts. A family
392
+ * matches a PDF font whose folded base name starts with the folded family
393
+ * — `dmsans` against `dmsanslightregular` — so style suffixes never split
394
+ * one family into several.
395
+ */
396
+ declare function familyRendered(family: string, fonts: readonly PdfFontInfo[]): boolean;
397
+ /** True when a `pdffonts` binary is reachable. */
398
+ declare function pdffontsAvailable(): Promise<boolean>;
399
+ /** The fonts a PDF on disk carries. One spawn, stdout only. */
400
+ declare function extractPdfFonts(pdfPath: string, options?: {
401
+ timeoutMs?: number;
402
+ }): Promise<PdfFontInfo[]>;
403
+
404
+ /**
405
+ * Locate authored text in rendered PDF geometry.
406
+ *
407
+ * Promoted from the ground-truth harness's sentinel search: fold both sides
408
+ * into lowercase alphanumerics (NFKC splits ligatures, punctuation and bullet
409
+ * glyphs drop out), concatenate every word fragment into one stream, and
410
+ * search that. Narrow boxes hard-wrap a word mid-word and letter-spaced text
411
+ * makes poppler emit per-cluster fragments; a per-word comparison loses both,
412
+ * a stream keeps them. A match must start where a fragment starts and end
413
+ * where one ends, so "page" never matches inside "homepage".
414
+ *
415
+ * The stream spans the whole document, not one page, so a paragraph that
416
+ * breaks across a page is still one match — which is exactly the case the
417
+ * widow and orphan checks need. Running heads and footers would interleave
418
+ * the two halves, so text that repeats by design is matched first, inside
419
+ * the top and bottom bands of each page, and its rows are removed from the
420
+ * stream before anything else is; a row of nothing but digits in those
421
+ * bands is a page number and goes with them.
422
+ *
423
+ * Fields, footnote marks and cross-references render a value the author
424
+ * never typed. The needle is cut at each of them into segments that must
425
+ * follow one another in the stream with at most a short gap between.
426
+ *
427
+ * Duplicate strings — "Total" in three tables — are why a search returns
428
+ * every occurrence. `assignInventory` resolves them by reading order: the
429
+ * inventory is walked in document order and the stream is in page order, so
430
+ * the next unclaimed occurrence at or after the previous claim is the one.
431
+ */
432
+
433
+ /** Fold rendered and authored text into one comparable form. */
434
+ declare function normalizeForMatch(value: string): string;
435
+ /**
436
+ * Authored strings carry markup the page never shows: emphasis markers,
437
+ * link targets, and fields, footnote references and cross-references whose
438
+ * rendered value is unknowable here. The first two are dropped; the others
439
+ * become a `FIELD` mark that `needleSegments` splits on.
440
+ */
441
+ declare function authoredTextForMatch(text: string): string;
442
+ /** The part of a match that lies on one page, with its union box. */
443
+ interface OccurrencePart {
444
+ pageIndex: number;
445
+ words: number[];
446
+ xMin: number;
447
+ yMin: number;
448
+ xMax: number;
449
+ yMax: number;
450
+ }
451
+ interface TextOccurrence {
452
+ /** Offset into the document stream, for ordering. */
453
+ at: number;
454
+ /** First and last page the match touches; equal unless it breaks across. */
455
+ pageIndex: number;
456
+ endPageIndex: number;
457
+ parts: OccurrencePart[];
458
+ }
459
+ /** What an inventory entry needs to be matched: its text, in reading order. */
460
+ interface InventoryEntry {
461
+ path: string;
462
+ text: string;
463
+ /** Header/footer text repeats per page: every occurrence is legitimate. */
464
+ repeats?: boolean;
465
+ }
466
+ type MappingStatus = 'mapped' | 'ambiguous' | 'missing' | 'skipped';
467
+ interface InventoryMatch<T extends InventoryEntry = InventoryEntry> {
468
+ entry: T;
469
+ needle: string;
470
+ status: MappingStatus;
471
+ /** One occurrence when mapped; every occurrence when the text repeats. */
472
+ occurrences: TextOccurrence[];
473
+ /**
474
+ * Set when only a leading part of the text rendered: the rest was cut
475
+ * off by a frame, a page edge or a box. Folded character counts.
476
+ */
477
+ partial?: {
478
+ matchedChars: number;
479
+ totalChars: number;
480
+ };
481
+ }
482
+ interface InventoryAssignment<T extends InventoryEntry> {
483
+ matches: InventoryMatch<T>[];
484
+ /** `pageIndex:word` keys of every word chrome claimed — whole rows. */
485
+ chromeWords: ReadonlySet<string>;
486
+ }
487
+ /**
488
+ * Attribute rendered occurrences to inventory entries.
489
+ *
490
+ * Repeating entries go first and claim their occurrences inside the page
491
+ * bands, one row each; rows of nothing but digits in those bands — page
492
+ * numbers — go with them. The rest are visited in document order over a
493
+ * stream without those words; each claims the first unclaimed occurrence at
494
+ * or after the previous claim, which is how "Total" in the second table
495
+ * finds the second "Total". An entry with no occurrence is `missing` —
496
+ * fully clipped, or never set — and one whose every occurrence was already
497
+ * claimed is `ambiguous`.
498
+ */
499
+ declare function assignInventory<T extends InventoryEntry>(pages: readonly PdfTextPage[], inventory: readonly T[]): InventoryAssignment<T>;
500
+
501
+ /**
502
+ * The rendered-certainty pass (#344): findings from what LibreOffice
503
+ * actually laid out, mapped back to the pointers that authored them.
504
+ *
505
+ * Static rules predict; this pass measures. It reads per-word geometry from
506
+ * the preview PDF, the fonts the PDF embeds, and the document's authored
507
+ * text inventory, and reports the defects only ink can show: text past the
508
+ * page or its frame, words drawn over each other, text that never rendered
509
+ * (fully clipped, or set in a place poppler cannot read), faces substituted
510
+ * on the way, pages with nothing on them, headings stranded at a page foot
511
+ * and paragraphs split into a lone line.
512
+ *
513
+ * Every finding carries `certainty: 'rendered'` and an explicit mapping
514
+ * status. A defect whose words match no inventory entry is still reported,
515
+ * at the document root, as `unmapped` — an unmapped finding is a mapping
516
+ * gap to close, never a defect to hide.
517
+ *
518
+ * Pure: geometry and fonts are extracted by the caller (see
519
+ * `extractPdfTextGeometry`, `extractPdfFonts`), so every rule here is
520
+ * testable from captured fixtures with no converter on the host.
521
+ */
522
+
523
+ /** One authored string the pass can match rendered words back to. */
524
+ interface RenderedTextEntry extends InventoryEntry {
525
+ role: 'heading' | 'body' | 'list-item' | 'table-header' | 'table-cell' | 'statistic' | 'caption' | 'chrome' | 'slide-text';
526
+ level?: number;
527
+ /** A declared box the text must fit: a docx frame, a pptx text box. */
528
+ box?: {
529
+ widthPt?: number;
530
+ heightPt?: number;
531
+ };
532
+ }
533
+ interface RequestedFont {
534
+ family: string;
535
+ /** Where the family was requested, when the document names it. */
536
+ path?: string;
537
+ /**
538
+ * The document supplied a source for the family (a file, a URL), so the
539
+ * preview should have had it and a substitution is the document's defect.
540
+ * Without one the family is a host font — Calibri on a Mac without Office
541
+ * — and the substitution says something about the preview, not the file.
542
+ */
543
+ declared?: boolean;
544
+ }
545
+ interface RenderedAnalysisInput {
546
+ format: 'docx' | 'pptx';
547
+ pages: readonly PdfTextPage[];
548
+ inventory: readonly RenderedTextEntry[];
549
+ /** Fonts the PDF carries; `undefined` when the host could not inspect them. */
550
+ fonts?: readonly PdfFontInfo[];
551
+ requestedFonts?: readonly RequestedFont[];
552
+ }
553
+ /** How a finding reached its pointer — always on `context.mapping`. */
554
+ type RenderedMapping = 'mapped' | 'ambiguous' | 'unmapped';
555
+ interface RenderedAnalysisSummary {
556
+ pages: number;
557
+ words: number;
558
+ /** Inventory entries by mapping outcome. */
559
+ inventory: Record<MappingStatus, number>;
560
+ /** Findings by mapping outcome. */
561
+ findings: Record<RenderedMapping, number>;
562
+ fonts?: {
563
+ requested: number;
564
+ substituted: number;
565
+ };
566
+ }
567
+ interface RenderedAnalysis {
568
+ findings: QualityDiagnostic[];
569
+ summary: RenderedAnalysisSummary;
570
+ }
571
+ /** Spill below this is sub-visual: renderer rounding, descender fuzz. */
572
+ declare const VISIBLE_SPILL_PT = 2;
573
+ /** Run the pass. */
574
+ declare function analyzeRenderedDocument(input: RenderedAnalysisInput): RenderedAnalysis;
575
+
346
576
  /**
347
577
  * Make resolved fonts visible to the LibreOffice child process for the
348
578
  * duration of one PDF conversion, then clean up.
@@ -534,4 +764,4 @@ declare function emitDiagnostic(text: string, tone?: DiagnosticTone): void;
534
764
  */
535
765
  declare const stderrDiagnosticSink: DiagnosticSink;
536
766
 
537
- export { type DiagnosticSink, type DiagnosticTone, DocxFormatAdapter, type FontStageHandle, type FontStageOptions, type FontStager, FontconfigStager, type FormatAdapter, type FormatName, type GeneratorOptions, type GeneratorResult, MacOSCoreTextStager, NoopFontStager, type PdfTextPage, type PdfTextWord, PptxFormatAdapter, type RasterizerCacheStats, WindowsFontStager, clearRasterizerCache, createAdapter, createLibreOfficePptxBatchRasterizer, createLibreOfficePptxRasterizer, emitDiagnostic, extractPdfTextGeometry, getFontStager, getRasterizerCacheStats, parsePdfTextBbox, pdftotextAvailable, runWithDiagnosticSink, stderrDiagnosticSink };
767
+ export { type DiagnosticSink, type DiagnosticTone, DocxFormatAdapter, type FontStageHandle, type FontStageOptions, type FontStager, FontconfigStager, type FormatAdapter, type FormatName, type GeneratorOptions, type GeneratorResult, type InventoryEntry, type InventoryMatch, MacOSCoreTextStager, type MappingStatus, NoopFontStager, type PdfFontInfo, type PdfTextLine, type PdfTextPage, type PdfTextWord, PptxFormatAdapter, type RasterizerCacheStats, type RenderedAnalysis, type RenderedAnalysisInput, type RenderedAnalysisSummary, type RenderedMapping, type RenderedTextEntry, type RequestedFont, type TextOccurrence, VISIBLE_SPILL_PT, WindowsFontStager, analyzeRenderedDocument, assignInventory, authoredTextForMatch, clearRasterizerCache, createAdapter, createLibreOfficePptxBatchRasterizer, createLibreOfficePptxRasterizer, emitDiagnostic, extractPdfFonts, extractPdfTextGeometry, familyRendered, getFontStager, getRasterizerCacheStats, normalizeForMatch, parsePdfFonts, parsePdfTextBbox, pdffontsAvailable, pdftotextAvailable, runWithDiagnosticSink, stderrDiagnosticSink };