@json-to-office/jto-ops 4.0.0 → 4.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import { FontRuntimeOpts, ServicesConfig, GenerationWarning, RendererStatus, PptxBatchRasterizer, PptxRasterizer, ResolvedFont } from '@json-to-office/shared';
2
- import { QualityProfile, QualityPolicy, PreparedDocument, QualityAnalysis } from '@json-to-office/quality';
2
+ import { QualityProfile, QualityPolicy, PreparedDocument, QualityAnalysis, QualityRulePack, QualityDiagnostic, QualityAnalyzeOptions } from '@json-to-office/quality';
3
3
 
4
4
  type FormatName = 'docx' | 'pptx';
5
5
  interface GeneratorOptions {
@@ -322,15 +322,30 @@ interface PdfTextWord {
322
322
  xMax: number;
323
323
  yMax: number;
324
324
  }
325
+ /**
326
+ * One line as poppler's layout analysis grouped it (`-bbox-layout` only):
327
+ * its box plus the indices into `words` of the words on it.
328
+ */
329
+ interface PdfTextLine {
330
+ xMin: number;
331
+ yMin: number;
332
+ xMax: number;
333
+ yMax: number;
334
+ words: number[];
335
+ }
325
336
  /** One PDF page: its size in points plus every word poppler segmented. */
326
337
  interface PdfTextPage {
327
338
  widthPt: number;
328
339
  heightPt: number;
340
+ /** Every word in stream order, whichever mode produced the page. */
329
341
  words: PdfTextWord[];
342
+ /** Line grouping; empty for plain `-bbox` output, which groups nothing. */
343
+ lines: PdfTextLine[];
330
344
  }
331
345
  /**
332
- * Parse `pdftotext -bbox` output (XHTML with `<page>`/`<word>` elements).
333
- * Pure — feed it a captured document for tests, or the runner's stdout.
346
+ * Parse `pdftotext -bbox` or `-bbox-layout` output (XHTML with `<page>`,
347
+ * optionally `<line>`, and `<word>` elements). Pure — feed it a captured
348
+ * document for tests, or the runner's stdout.
334
349
  */
335
350
  declare function parsePdfTextBbox(bboxXml: string): PdfTextPage[];
336
351
  /** True when a `pdftotext` binary is reachable — lets harnesses skip early. */
@@ -338,11 +353,270 @@ declare function pdftotextAvailable(): Promise<boolean>;
338
353
  /**
339
354
  * Extract per-word text geometry from a PDF on disk. One pdftotext spawn,
340
355
  * output streamed through stdout — nothing else touches the filesystem.
356
+ * `layout: true` asks poppler for its line grouping too (`-bbox-layout`).
341
357
  */
342
358
  declare function extractPdfTextGeometry(pdfPath: string, options?: {
343
359
  timeoutMs?: number;
360
+ layout?: boolean;
344
361
  }): Promise<PdfTextPage[]>;
345
362
 
363
+ /**
364
+ * Font names from a rendered PDF — the other half of substitution detection.
365
+ *
366
+ * `pdftotext -bbox` says where every word landed but not which face drew it.
367
+ * `pdffonts` (poppler, next to pdftotext) lists every font the PDF embeds or
368
+ * references; LibreOffice embeds the face it actually used, so a requested
369
+ * family that never appears in that list was substituted on the way.
370
+ *
371
+ * Names are PostScript-style: a subset tag (`BAAAAA+`), the family with
372
+ * spaces removed, and a style suffix (`-Bold`, `Medium-Regular`). Comparison
373
+ * therefore happens on a folded form — lowercase alphanumerics only — where
374
+ * "Space Grotesk" and `CAAAAA+SpaceGrotesk-Regular` meet as a prefix match.
375
+ */
376
+ /** One font row as `pdffonts` prints it. */
377
+ interface PdfFontInfo {
378
+ /** The name as printed, subset tag included. */
379
+ name: string;
380
+ /** The name without its subset tag: `DMSans-Regular`. */
381
+ baseName: string;
382
+ type: string;
383
+ embedded: boolean;
384
+ }
385
+ /**
386
+ * Parse `pdffonts` output. Pure; the header and rule lines are skipped and
387
+ * a row that does not fit the column layout is ignored rather than guessed.
388
+ */
389
+ declare function parsePdfFonts(stdout: string): PdfFontInfo[];
390
+ /**
391
+ * Whether a requested family is present among the rendered fonts. A family
392
+ * matches a PDF font whose folded base name starts with the folded family
393
+ * — `dmsans` against `dmsanslightregular` — so style suffixes never split
394
+ * one family into several.
395
+ */
396
+ declare function familyRendered(family: string, fonts: readonly PdfFontInfo[]): boolean;
397
+ /** True when a `pdffonts` binary is reachable. */
398
+ declare function pdffontsAvailable(): Promise<boolean>;
399
+ /** The fonts a PDF on disk carries. One spawn, stdout only. */
400
+ declare function extractPdfFonts(pdfPath: string, options?: {
401
+ timeoutMs?: number;
402
+ }): Promise<PdfFontInfo[]>;
403
+
404
+ /**
405
+ * The rendered pass as a quality rule pack (#344).
406
+ *
407
+ * `draftRenderedFindings` measures; these rules are how the measurement
408
+ * enters the quality contract. Each rule owns one code, one description the
409
+ * design guide prints, and the defaults a profile or policy can move. All of
410
+ * them read a single `rendered/geometry` fact — the PDF's word boxes, its
411
+ * embedded fonts and the document's text inventory — and share one matching
412
+ * pass over it, memoised per fact, so eight rules cost one search.
413
+ */
414
+
415
+ type RenderedRuleId = 'rendered/clip' | 'rendered/spill' | 'rendered/overlap' | 'rendered/text-missing' | 'rendered/font-substituted' | 'rendered/empty-page' | 'rendered/heading-stranded' | 'rendered/paragraph-split';
416
+ declare const RENDERED_QUALITY_RULES: QualityRulePack;
417
+
418
+ /**
419
+ * Locate authored text in rendered PDF geometry.
420
+ *
421
+ * Promoted from the ground-truth harness's sentinel search: fold both sides
422
+ * into lowercase alphanumerics (NFKC splits ligatures, punctuation and bullet
423
+ * glyphs drop out), concatenate every word fragment into one stream, and
424
+ * search that. Narrow boxes hard-wrap a word mid-word and letter-spaced text
425
+ * makes poppler emit per-cluster fragments; a per-word comparison loses both,
426
+ * a stream keeps them. A match must start where a fragment starts and end
427
+ * where one ends, so "page" never matches inside "homepage".
428
+ *
429
+ * The stream spans the whole document, not one page, so a paragraph that
430
+ * breaks across a page is still one match — which is exactly the case the
431
+ * widow and orphan checks need. Running heads and footers would interleave
432
+ * the two halves, so text that repeats by design is matched first, inside
433
+ * the top and bottom bands of each page, and its rows are removed from the
434
+ * stream before anything else is; a row of nothing but digits in those
435
+ * bands is a page number and goes with them.
436
+ *
437
+ * Fields, footnote marks and cross-references render a value the author
438
+ * never typed. The needle is cut at each of them into segments that must
439
+ * follow one another in the stream with at most a short gap between.
440
+ *
441
+ * Duplicate strings — "Total" in three tables — are why a search returns
442
+ * every occurrence. `assignInventory` resolves them by reading order: the
443
+ * inventory is walked in document order and the stream is in page order, so
444
+ * the next unclaimed occurrence at or after the previous claim is the one.
445
+ */
446
+
447
+ /** Fold rendered and authored text into one comparable form. */
448
+ declare function normalizeForMatch(value: string): string;
449
+ /**
450
+ * Authored strings carry markup the page never shows: emphasis markers,
451
+ * link targets, and fields, footnote references and cross-references whose
452
+ * rendered value is unknowable here. The first two are dropped; the others
453
+ * become a `FIELD` mark that `needleSegments` splits on.
454
+ */
455
+ declare function authoredTextForMatch(text: string): string;
456
+ /** The part of a match that lies on one page, with its union box. */
457
+ interface OccurrencePart {
458
+ pageIndex: number;
459
+ words: number[];
460
+ xMin: number;
461
+ yMin: number;
462
+ xMax: number;
463
+ yMax: number;
464
+ }
465
+ interface TextOccurrence {
466
+ /** Offset into the document stream, for ordering. */
467
+ at: number;
468
+ /** First and last page the match touches; equal unless it breaks across. */
469
+ pageIndex: number;
470
+ endPageIndex: number;
471
+ parts: OccurrencePart[];
472
+ }
473
+ /** What an inventory entry needs to be matched: its text, in reading order. */
474
+ interface InventoryEntry {
475
+ path: string;
476
+ text: string;
477
+ /** Header/footer text repeats per page: every occurrence is legitimate. */
478
+ repeats?: boolean;
479
+ /**
480
+ * Painted by the renderer from other authored text — a contents entry —
481
+ * so it claims its occurrence when there is one and is `skipped`, never
482
+ * `missing`, when there is not.
483
+ */
484
+ optional?: boolean;
485
+ }
486
+ type MappingStatus = 'mapped' | 'ambiguous' | 'missing' | 'skipped';
487
+ interface InventoryMatch<T extends InventoryEntry = InventoryEntry> {
488
+ entry: T;
489
+ needle: string;
490
+ status: MappingStatus;
491
+ /** One occurrence when mapped; every occurrence when the text repeats. */
492
+ occurrences: TextOccurrence[];
493
+ /**
494
+ * Set when only a leading part of the text rendered: the rest was cut
495
+ * off by a frame, a page edge or a box. Folded character counts.
496
+ */
497
+ partial?: {
498
+ matchedChars: number;
499
+ totalChars: number;
500
+ };
501
+ }
502
+ interface InventoryAssignment<T extends InventoryEntry> {
503
+ matches: InventoryMatch<T>[];
504
+ /** `pageIndex:word` keys of every word chrome claimed — whole rows. */
505
+ chromeWords: ReadonlySet<string>;
506
+ }
507
+ /**
508
+ * Attribute rendered occurrences to inventory entries.
509
+ *
510
+ * Repeating entries go first and claim their occurrences inside the page
511
+ * bands, one row each; rows of nothing but digits in those bands — page
512
+ * numbers — go with them. The rest are visited in document order over a
513
+ * stream without those words; each claims the first unclaimed occurrence at
514
+ * or after the previous claim, which is how "Total" in the second table
515
+ * finds the second "Total". An entry with no occurrence is `missing` —
516
+ * fully clipped, or never set — unless it is optional, and one whose every
517
+ * occurrence was already claimed is `ambiguous`.
518
+ */
519
+ declare function assignInventory<T extends InventoryEntry>(pages: readonly PdfTextPage[], inventory: readonly T[]): InventoryAssignment<T>;
520
+
521
+ /**
522
+ * The rendered-certainty pass (#344): findings from what LibreOffice
523
+ * actually laid out, mapped back to the pointers that authored them.
524
+ *
525
+ * Static rules predict; this pass measures. It reads per-word geometry from
526
+ * the preview PDF, the fonts the PDF embeds, and the document's authored
527
+ * text inventory, and reports the defects only ink can show: text past the
528
+ * page or its frame, words drawn over each other, text that never rendered
529
+ * (fully clipped, or set in a place poppler cannot read), faces substituted
530
+ * on the way, pages with nothing on them, headings stranded at a page foot
531
+ * and paragraphs split into a lone line.
532
+ *
533
+ * Every finding carries `certainty: 'rendered'` and an explicit mapping
534
+ * status. A defect whose words match no inventory entry is still reported,
535
+ * at the document root, as `unmapped` — an unmapped finding is a mapping
536
+ * gap to close, never a defect to hide.
537
+ *
538
+ * Pure: geometry and fonts are extracted by the caller (see
539
+ * `extractPdfTextGeometry`, `extractPdfFonts`), so every rule here is
540
+ * testable from captured fixtures with no converter on the host.
541
+ *
542
+ * The checks draft findings; the quality engine turns them into
543
+ * diagnostics. Each check is one rule of `RENDERED_QUALITY_RULES`, so a
544
+ * profile can switch one off or move its severity, a policy can suppress
545
+ * one at a pointer and a gate can make one blocking — the same levers every
546
+ * static rule answers to, applied to the pass that measures.
547
+ */
548
+
549
+ /** One authored string the pass can match rendered words back to. */
550
+ interface RenderedTextEntry extends InventoryEntry {
551
+ role: 'heading' | 'body' | 'list-item' | 'table-header' | 'table-cell' | 'statistic' | 'caption' | 'toc-entry' | 'chrome' | 'slide-text';
552
+ level?: number;
553
+ /** A declared box the text must fit: a docx frame, a pptx text box. */
554
+ box?: {
555
+ widthPt?: number;
556
+ heightPt?: number;
557
+ };
558
+ }
559
+ interface RequestedFont {
560
+ family: string;
561
+ /** Where the family was requested, when the document names it. */
562
+ path?: string;
563
+ /**
564
+ * The document supplied a source for the family (a file, a URL), so the
565
+ * preview should have had it and a substitution is the document's defect.
566
+ * Without one the family is a host font — Calibri on a Mac without Office
567
+ * — and the substitution says something about the preview, not the file.
568
+ */
569
+ declared?: boolean;
570
+ }
571
+ interface RenderedAnalysisInput {
572
+ format: 'docx' | 'pptx';
573
+ /** Renderer identity, for a profile that declares renderer targets. */
574
+ renderer?: string;
575
+ pages: readonly PdfTextPage[];
576
+ inventory: readonly RenderedTextEntry[];
577
+ /** Fonts the PDF carries; `undefined` when the host could not inspect them. */
578
+ fonts?: readonly PdfFontInfo[];
579
+ requestedFonts?: readonly RequestedFont[];
580
+ }
581
+ /** How a finding reached its pointer — always on `context.mapping`. */
582
+ type RenderedMapping = 'mapped' | 'ambiguous' | 'unmapped';
583
+ interface RenderedAnalysisSummary {
584
+ pages: number;
585
+ words: number;
586
+ /** Inventory entries by mapping outcome. */
587
+ inventory: Record<MappingStatus, number>;
588
+ /** Findings by mapping outcome, after suppressions. */
589
+ findings: Record<RenderedMapping, number>;
590
+ fonts?: {
591
+ requested: number;
592
+ substituted: number;
593
+ };
594
+ /** Findings a policy suppression removed. */
595
+ suppressed: number;
596
+ /** Whether a policy gate made any finding blocking. */
597
+ blocked: boolean;
598
+ /** Whether a policy `maxDiagnostics` budget cut the findings returned. */
599
+ truncated: boolean;
600
+ /** The quality profile the pass ran under, when one applied. */
601
+ profileId?: string;
602
+ }
603
+ interface RenderedAnalysis {
604
+ /** The pass's diagnostics, as the quality engine finalised them. */
605
+ findings: readonly QualityDiagnostic[];
606
+ summary: RenderedAnalysisSummary;
607
+ /** The engine's own account: evaluated rules, rule errors, truncation. */
608
+ analysis: QualityAnalysis;
609
+ }
610
+ /** Spill below this is sub-visual: renderer rounding, descender fuzz. */
611
+ declare const VISIBLE_SPILL_PT = 2;
612
+ /**
613
+ * Run the pass. Geometry, fonts and inventory become one `rendered/geometry`
614
+ * fact; the rendered rule pack reads it under the caller's profile and
615
+ * policy, so the findings come back suppressed, re-severed and gated the
616
+ * way `jto_validate`'s would.
617
+ */
618
+ declare function analyzeRenderedDocument(input: RenderedAnalysisInput, options?: QualityAnalyzeOptions): RenderedAnalysis;
619
+
346
620
  /**
347
621
  * Make resolved fonts visible to the LibreOffice child process for the
348
622
  * duration of one PDF conversion, then clean up.
@@ -534,4 +808,4 @@ declare function emitDiagnostic(text: string, tone?: DiagnosticTone): void;
534
808
  */
535
809
  declare const stderrDiagnosticSink: DiagnosticSink;
536
810
 
537
- export { type DiagnosticSink, type DiagnosticTone, DocxFormatAdapter, type FontStageHandle, type FontStageOptions, type FontStager, FontconfigStager, type FormatAdapter, type FormatName, type GeneratorOptions, type GeneratorResult, MacOSCoreTextStager, NoopFontStager, type PdfTextPage, type PdfTextWord, PptxFormatAdapter, type RasterizerCacheStats, WindowsFontStager, clearRasterizerCache, createAdapter, createLibreOfficePptxBatchRasterizer, createLibreOfficePptxRasterizer, emitDiagnostic, extractPdfTextGeometry, getFontStager, getRasterizerCacheStats, parsePdfTextBbox, pdftotextAvailable, runWithDiagnosticSink, stderrDiagnosticSink };
811
+ export { type DiagnosticSink, type DiagnosticTone, DocxFormatAdapter, type FontStageHandle, type FontStageOptions, type FontStager, FontconfigStager, type FormatAdapter, type FormatName, type GeneratorOptions, type GeneratorResult, type InventoryEntry, type InventoryMatch, MacOSCoreTextStager, type MappingStatus, NoopFontStager, type PdfFontInfo, type PdfTextLine, type PdfTextPage, type PdfTextWord, PptxFormatAdapter, RENDERED_QUALITY_RULES, type RasterizerCacheStats, type RenderedAnalysis, type RenderedAnalysisInput, type RenderedAnalysisSummary, type RenderedMapping, type RenderedRuleId, type RenderedTextEntry, type RequestedFont, type TextOccurrence, VISIBLE_SPILL_PT, WindowsFontStager, analyzeRenderedDocument, assignInventory, authoredTextForMatch, clearRasterizerCache, createAdapter, createLibreOfficePptxBatchRasterizer, createLibreOfficePptxRasterizer, emitDiagnostic, extractPdfFonts, extractPdfTextGeometry, familyRendered, getFontStager, getRasterizerCacheStats, normalizeForMatch, parsePdfFonts, parsePdfTextBbox, pdffontsAvailable, pdftotextAvailable, runWithDiagnosticSink, stderrDiagnosticSink };