@json-to-office/jto-ops 3.3.0 → 4.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +234 -4
- package/dist/index.js +774 -11
- package/dist/index.js.map +1 -1
- package/package.json +7 -7
package/dist/index.js
CHANGED
|
@@ -1742,24 +1742,41 @@ function decodeEntities(value) {
|
|
|
1742
1742
|
);
|
|
1743
1743
|
}
|
|
1744
1744
|
var PAGE_PATTERN = /<page\s+width="([\d.]+)"\s+height="([\d.]+)">([\s\S]*?)<\/page>/g;
|
|
1745
|
-
var
|
|
1745
|
+
var LINE_OR_WORD_PATTERN = /<line\s+xMin="(-?[\d.]+)"\s+yMin="(-?[\d.]+)"\s+xMax="(-?[\d.]+)"\s+yMax="(-?[\d.]+)">|<\/line>|<word\s+xMin="(-?[\d.]+)"\s+yMin="(-?[\d.]+)"\s+xMax="(-?[\d.]+)"\s+yMax="(-?[\d.]+)">([\s\S]*?)<\/word>/g;
|
|
1746
1746
|
function parsePdfTextBbox(bboxXml) {
|
|
1747
1747
|
const pages = [];
|
|
1748
1748
|
for (const pageMatch of bboxXml.matchAll(PAGE_PATTERN)) {
|
|
1749
1749
|
const words = [];
|
|
1750
|
-
|
|
1751
|
-
|
|
1752
|
-
|
|
1753
|
-
|
|
1754
|
-
|
|
1755
|
-
|
|
1756
|
-
|
|
1757
|
-
|
|
1750
|
+
const lines = [];
|
|
1751
|
+
let open;
|
|
1752
|
+
for (const token of pageMatch[3].matchAll(LINE_OR_WORD_PATTERN)) {
|
|
1753
|
+
if (token[0] === "</line>") {
|
|
1754
|
+
if (open) lines.push(open);
|
|
1755
|
+
open = void 0;
|
|
1756
|
+
} else if (token[1] !== void 0) {
|
|
1757
|
+
open = {
|
|
1758
|
+
xMin: Number(token[1]),
|
|
1759
|
+
yMin: Number(token[2]),
|
|
1760
|
+
xMax: Number(token[3]),
|
|
1761
|
+
yMax: Number(token[4]),
|
|
1762
|
+
words: []
|
|
1763
|
+
};
|
|
1764
|
+
} else {
|
|
1765
|
+
open?.words.push(words.length);
|
|
1766
|
+
words.push({
|
|
1767
|
+
xMin: Number(token[5]),
|
|
1768
|
+
yMin: Number(token[6]),
|
|
1769
|
+
xMax: Number(token[7]),
|
|
1770
|
+
yMax: Number(token[8]),
|
|
1771
|
+
text: decodeEntities(token[9])
|
|
1772
|
+
});
|
|
1773
|
+
}
|
|
1758
1774
|
}
|
|
1759
1775
|
pages.push({
|
|
1760
1776
|
widthPt: Number(pageMatch[1]),
|
|
1761
1777
|
heightPt: Number(pageMatch[2]),
|
|
1762
|
-
words
|
|
1778
|
+
words,
|
|
1779
|
+
lines
|
|
1763
1780
|
});
|
|
1764
1781
|
}
|
|
1765
1782
|
return pages;
|
|
@@ -1827,27 +1844,773 @@ async function extractPdfTextGeometry(pdfPath, options = {}) {
|
|
|
1827
1844
|
const binary = await resolvePdftotext();
|
|
1828
1845
|
const stdout = await run(
|
|
1829
1846
|
binary,
|
|
1830
|
-
["-bbox", pdfPath, "-"],
|
|
1847
|
+
[options.layout ? "-bbox-layout" : "-bbox", pdfPath, "-"],
|
|
1831
1848
|
options.timeoutMs ?? 6e4
|
|
1832
1849
|
);
|
|
1833
1850
|
return parsePdfTextBbox(stdout);
|
|
1834
1851
|
}
|
|
1852
|
+
|
|
1853
|
+
// src/pdf-fonts.ts
|
|
1854
|
+
import { execFile as execFile3 } from "child_process";
|
|
1855
|
+
import * as path7 from "path";
|
|
1856
|
+
var ROW_PATTERN = /^(\S+)\s+(.+?)\s+(\S+)\s+(yes|no)\s+(yes|no)\s+(yes|no)\s+/;
|
|
1857
|
+
function parsePdfFonts(stdout) {
|
|
1858
|
+
const fonts = [];
|
|
1859
|
+
for (const line of stdout.split(/\r?\n/)) {
|
|
1860
|
+
if (line.startsWith("name ") || line.startsWith("---") || !line.trim()) {
|
|
1861
|
+
continue;
|
|
1862
|
+
}
|
|
1863
|
+
const match = ROW_PATTERN.exec(line);
|
|
1864
|
+
if (!match) continue;
|
|
1865
|
+
const name = match[1];
|
|
1866
|
+
fonts.push({
|
|
1867
|
+
name,
|
|
1868
|
+
baseName: name.replace(/^[A-Z]{6}\+/, ""),
|
|
1869
|
+
type: match[2].trim(),
|
|
1870
|
+
embedded: match[4] === "yes"
|
|
1871
|
+
});
|
|
1872
|
+
}
|
|
1873
|
+
return fonts;
|
|
1874
|
+
}
|
|
1875
|
+
function foldFontName(value) {
|
|
1876
|
+
return value.toLowerCase().replace(/[^a-z0-9]/g, "");
|
|
1877
|
+
}
|
|
1878
|
+
function familyRendered(family, fonts) {
|
|
1879
|
+
const folded = foldFontName(family);
|
|
1880
|
+
if (folded === "") return true;
|
|
1881
|
+
return fonts.some((font) => foldFontName(font.baseName).startsWith(folded));
|
|
1882
|
+
}
|
|
1883
|
+
function pdffontsCandidates() {
|
|
1884
|
+
const candidates = [];
|
|
1885
|
+
const configured = process.env.PDFFONTS_PATH?.trim();
|
|
1886
|
+
if (configured) candidates.push(configured);
|
|
1887
|
+
const sibling = process.env.PDFTOTEXT_PATH?.trim();
|
|
1888
|
+
if (sibling) {
|
|
1889
|
+
candidates.push(
|
|
1890
|
+
path7.join(
|
|
1891
|
+
path7.dirname(sibling),
|
|
1892
|
+
process.platform === "win32" ? "pdffonts.exe" : "pdffonts"
|
|
1893
|
+
)
|
|
1894
|
+
);
|
|
1895
|
+
}
|
|
1896
|
+
candidates.push("pdffonts");
|
|
1897
|
+
return [...new Set(candidates)];
|
|
1898
|
+
}
|
|
1899
|
+
async function run2(binary, args, timeoutMs) {
|
|
1900
|
+
return new Promise((resolve2, reject) => {
|
|
1901
|
+
execFile3(
|
|
1902
|
+
binary,
|
|
1903
|
+
args,
|
|
1904
|
+
{ timeout: timeoutMs, maxBuffer: 16 * 1024 * 1024 },
|
|
1905
|
+
(error, stdout) => {
|
|
1906
|
+
if (error) reject(error);
|
|
1907
|
+
else resolve2(stdout);
|
|
1908
|
+
}
|
|
1909
|
+
);
|
|
1910
|
+
});
|
|
1911
|
+
}
|
|
1912
|
+
var pdffontsPromise;
|
|
1913
|
+
async function resolvePdffonts() {
|
|
1914
|
+
if (!pdffontsPromise) {
|
|
1915
|
+
pdffontsPromise = (async () => {
|
|
1916
|
+
for (const candidate of pdffontsCandidates()) {
|
|
1917
|
+
try {
|
|
1918
|
+
await run2(candidate, ["-v"], 1e4);
|
|
1919
|
+
return candidate;
|
|
1920
|
+
} catch (error) {
|
|
1921
|
+
const code = error.code;
|
|
1922
|
+
if (code !== "ENOENT" && code !== "EACCES") return candidate;
|
|
1923
|
+
}
|
|
1924
|
+
}
|
|
1925
|
+
throw new Error(
|
|
1926
|
+
`Font inspection needs pdffonts (poppler), which was not found. Install poppler-utils or set PDFFONTS_PATH (searched: ${pdffontsCandidates().join(", ")}).`
|
|
1927
|
+
);
|
|
1928
|
+
})().catch((error) => {
|
|
1929
|
+
pdffontsPromise = void 0;
|
|
1930
|
+
throw error;
|
|
1931
|
+
});
|
|
1932
|
+
}
|
|
1933
|
+
return pdffontsPromise;
|
|
1934
|
+
}
|
|
1935
|
+
async function pdffontsAvailable() {
|
|
1936
|
+
try {
|
|
1937
|
+
await resolvePdffonts();
|
|
1938
|
+
return true;
|
|
1939
|
+
} catch {
|
|
1940
|
+
return false;
|
|
1941
|
+
}
|
|
1942
|
+
}
|
|
1943
|
+
async function extractPdfFonts(pdfPath, options = {}) {
|
|
1944
|
+
const binary = await resolvePdffonts();
|
|
1945
|
+
const stdout = await run2(binary, [pdfPath], options.timeoutMs ?? 3e4);
|
|
1946
|
+
return parsePdfFonts(stdout);
|
|
1947
|
+
}
|
|
1948
|
+
|
|
1949
|
+
// src/rendered-analysis.ts
|
|
1950
|
+
import {
|
|
1951
|
+
QUALITY_CODES
|
|
1952
|
+
} from "@json-to-office/quality";
|
|
1953
|
+
|
|
1954
|
+
// src/rendered-text-match.ts
|
|
1955
|
+
function normalizeForMatch(value) {
|
|
1956
|
+
return value.normalize("NFKC").toLowerCase().replace(/[^a-z0-9]/g, "");
|
|
1957
|
+
}
|
|
1958
|
+
var FIELD = "";
|
|
1959
|
+
function authoredTextForMatch(text) {
|
|
1960
|
+
return text.replace(/\{[^}\s]+\}/g, FIELD).replace(/\[\^[^\]\s]+\]/g, FIELD).replace(/\[@[^\]\s]+\]/g, FIELD).replace(/\[([^\]]*)\]\([^)]*\)/g, "$1").replace(/[*_`]+/g, "");
|
|
1961
|
+
}
|
|
1962
|
+
function needleSegments(text) {
|
|
1963
|
+
return authoredTextForMatch(text).split(FIELD).map(normalizeForMatch).filter((segment) => segment !== "");
|
|
1964
|
+
}
|
|
1965
|
+
function indexDocument(pages, exclude = /* @__PURE__ */ new Set()) {
|
|
1966
|
+
const refs = [];
|
|
1967
|
+
const fragments = [];
|
|
1968
|
+
const positions = [];
|
|
1969
|
+
let cursor = 0;
|
|
1970
|
+
pages.forEach((page, pageIndex) => {
|
|
1971
|
+
page.words.forEach((word, index) => {
|
|
1972
|
+
const key = `${pageIndex}:${index}`;
|
|
1973
|
+
const fragment = exclude.has(key) ? "" : normalizeForMatch(word.text);
|
|
1974
|
+
refs.push({ pageIndex, word: index });
|
|
1975
|
+
fragments.push(fragment);
|
|
1976
|
+
positions.push(cursor);
|
|
1977
|
+
cursor += fragment.length;
|
|
1978
|
+
});
|
|
1979
|
+
});
|
|
1980
|
+
return { pages, refs, fragments, positions, stream: fragments.join("") };
|
|
1981
|
+
}
|
|
1982
|
+
function fragmentAt(index, offset) {
|
|
1983
|
+
let low = 0;
|
|
1984
|
+
let high = index.positions.length - 1;
|
|
1985
|
+
while (low < high) {
|
|
1986
|
+
const mid = Math.ceil((low + high) / 2);
|
|
1987
|
+
if (index.positions[mid] <= offset) low = mid;
|
|
1988
|
+
else high = mid - 1;
|
|
1989
|
+
}
|
|
1990
|
+
while (low < index.fragments.length && index.fragments[low] === "") low++;
|
|
1991
|
+
return low;
|
|
1992
|
+
}
|
|
1993
|
+
function alignedToFragments(index, at, end) {
|
|
1994
|
+
const first = fragmentAt(index, at);
|
|
1995
|
+
if (first >= index.fragments.length || index.positions[first] !== at) {
|
|
1996
|
+
return false;
|
|
1997
|
+
}
|
|
1998
|
+
let i = first;
|
|
1999
|
+
while (i < index.fragments.length && index.positions[i] + index.fragments[i].length < end) {
|
|
2000
|
+
i++;
|
|
2001
|
+
}
|
|
2002
|
+
return i < index.fragments.length && index.positions[i] + index.fragments[i].length === end;
|
|
2003
|
+
}
|
|
2004
|
+
function partsOf(index, at, end) {
|
|
2005
|
+
const parts = [];
|
|
2006
|
+
for (let i = fragmentAt(index, at); i < index.fragments.length; i++) {
|
|
2007
|
+
const start = index.positions[i];
|
|
2008
|
+
if (start >= end) break;
|
|
2009
|
+
if (index.fragments[i] === "") continue;
|
|
2010
|
+
const ref = index.refs[i];
|
|
2011
|
+
const w = index.pages[ref.pageIndex].words[ref.word];
|
|
2012
|
+
let part = parts[parts.length - 1];
|
|
2013
|
+
if (!part || part.pageIndex !== ref.pageIndex) {
|
|
2014
|
+
part = {
|
|
2015
|
+
pageIndex: ref.pageIndex,
|
|
2016
|
+
words: [],
|
|
2017
|
+
xMin: Infinity,
|
|
2018
|
+
yMin: Infinity,
|
|
2019
|
+
xMax: -Infinity,
|
|
2020
|
+
yMax: -Infinity
|
|
2021
|
+
};
|
|
2022
|
+
parts.push(part);
|
|
2023
|
+
}
|
|
2024
|
+
part.words.push(ref.word);
|
|
2025
|
+
part.xMin = Math.min(part.xMin, w.xMin);
|
|
2026
|
+
part.yMin = Math.min(part.yMin, w.yMin);
|
|
2027
|
+
part.xMax = Math.max(part.xMax, w.xMax);
|
|
2028
|
+
part.yMax = Math.max(part.yMax, w.yMax);
|
|
2029
|
+
}
|
|
2030
|
+
return parts;
|
|
2031
|
+
}
|
|
2032
|
+
function occurrence(index, at, end) {
|
|
2033
|
+
const parts = partsOf(index, at, end);
|
|
2034
|
+
return {
|
|
2035
|
+
at,
|
|
2036
|
+
pageIndex: parts[0].pageIndex,
|
|
2037
|
+
endPageIndex: parts[parts.length - 1].pageIndex,
|
|
2038
|
+
parts
|
|
2039
|
+
};
|
|
2040
|
+
}
|
|
2041
|
+
var MAX_FIELD_GAP = 24;
|
|
2042
|
+
function findOccurrences(index, segments) {
|
|
2043
|
+
if (segments.length === 0 || segments[0] === "") return [];
|
|
2044
|
+
const hits = [];
|
|
2045
|
+
let searchFrom = 0;
|
|
2046
|
+
for (; ; ) {
|
|
2047
|
+
const at = index.stream.indexOf(segments[0], searchFrom);
|
|
2048
|
+
if (at === -1) break;
|
|
2049
|
+
searchFrom = at + 1;
|
|
2050
|
+
let end = at + segments[0].length;
|
|
2051
|
+
let complete = true;
|
|
2052
|
+
for (let s = 1; s < segments.length; s++) {
|
|
2053
|
+
const next = index.stream.indexOf(segments[s], end);
|
|
2054
|
+
if (next === -1 || next - end > MAX_FIELD_GAP) {
|
|
2055
|
+
complete = false;
|
|
2056
|
+
break;
|
|
2057
|
+
}
|
|
2058
|
+
end = next + segments[s].length;
|
|
2059
|
+
}
|
|
2060
|
+
if (!complete || !alignedToFragments(index, at, end)) continue;
|
|
2061
|
+
hits.push(occurrence(index, at, end));
|
|
2062
|
+
searchFrom = end;
|
|
2063
|
+
}
|
|
2064
|
+
return hits;
|
|
2065
|
+
}
|
|
2066
|
+
var MIN_NEEDLE_LENGTH = 3;
|
|
2067
|
+
var MIN_PARTIAL_PREFIX = 16;
|
|
2068
|
+
var CHROME_BAND = 0.2;
|
|
2069
|
+
function longestRenderedPrefix(index, needle) {
|
|
2070
|
+
if (needle.length < MIN_PARTIAL_PREFIX) return void 0;
|
|
2071
|
+
const anchor = needle.slice(0, MIN_PARTIAL_PREFIX);
|
|
2072
|
+
let best;
|
|
2073
|
+
let searchFrom = 0;
|
|
2074
|
+
for (; ; ) {
|
|
2075
|
+
const at = index.stream.indexOf(anchor, searchFrom);
|
|
2076
|
+
if (at === -1) break;
|
|
2077
|
+
searchFrom = at + 1;
|
|
2078
|
+
const first = fragmentAt(index, at);
|
|
2079
|
+
if (first >= index.fragments.length || index.positions[first] !== at) {
|
|
2080
|
+
continue;
|
|
2081
|
+
}
|
|
2082
|
+
let length = MIN_PARTIAL_PREFIX;
|
|
2083
|
+
while (length < needle.length && index.stream.charCodeAt(at + length) === needle.charCodeAt(length)) {
|
|
2084
|
+
length++;
|
|
2085
|
+
}
|
|
2086
|
+
if (!best || length > best.length) best = { length, at };
|
|
2087
|
+
}
|
|
2088
|
+
return best;
|
|
2089
|
+
}
|
|
2090
|
+
function sameRow(a, b) {
|
|
2091
|
+
const overlap = Math.min(a.yMax, b.yMax) - Math.max(a.yMin, b.yMin);
|
|
2092
|
+
const shorter = Math.min(a.yMax - a.yMin, b.yMax - b.yMin);
|
|
2093
|
+
return shorter > 0 && overlap > shorter / 2;
|
|
2094
|
+
}
|
|
2095
|
+
function inChromeBand(page, box) {
|
|
2096
|
+
return box.yMax <= page.heightPt * CHROME_BAND || box.yMin >= page.heightPt * (1 - CHROME_BAND);
|
|
2097
|
+
}
|
|
2098
|
+
var PAGE_NUMBER_ROW = /^[0-9\s./|-]*$/;
|
|
2099
|
+
function assignInventory(pages, inventory) {
|
|
2100
|
+
const results = /* @__PURE__ */ new Map();
|
|
2101
|
+
const chromeWords = /* @__PURE__ */ new Set();
|
|
2102
|
+
const claimRow = (pageIndex, box) => {
|
|
2103
|
+
pages[pageIndex].words.forEach((w, i) => {
|
|
2104
|
+
if (sameRow(w, box)) chromeWords.add(`${pageIndex}:${i}`);
|
|
2105
|
+
});
|
|
2106
|
+
};
|
|
2107
|
+
const repeating = inventory.filter((entry) => entry.repeats);
|
|
2108
|
+
if (repeating.length > 0) {
|
|
2109
|
+
const full = indexDocument(pages);
|
|
2110
|
+
for (const entry of repeating) {
|
|
2111
|
+
const segments = needleSegments(entry.text);
|
|
2112
|
+
const needle = segments.join("");
|
|
2113
|
+
if (needle.length < MIN_NEEDLE_LENGTH) {
|
|
2114
|
+
results.set(entry, {
|
|
2115
|
+
entry,
|
|
2116
|
+
needle,
|
|
2117
|
+
status: "skipped",
|
|
2118
|
+
occurrences: []
|
|
2119
|
+
});
|
|
2120
|
+
continue;
|
|
2121
|
+
}
|
|
2122
|
+
const occurrences = findOccurrences(full, segments).filter(
|
|
2123
|
+
(o) => o.pageIndex === o.endPageIndex && inChromeBand(pages[o.pageIndex], o.parts[0])
|
|
2124
|
+
);
|
|
2125
|
+
for (const o of occurrences) claimRow(o.pageIndex, o.parts[0]);
|
|
2126
|
+
results.set(entry, {
|
|
2127
|
+
entry,
|
|
2128
|
+
needle,
|
|
2129
|
+
status: occurrences.length > 0 ? "mapped" : "missing",
|
|
2130
|
+
occurrences
|
|
2131
|
+
});
|
|
2132
|
+
}
|
|
2133
|
+
}
|
|
2134
|
+
pages.forEach((page, pageIndex) => {
|
|
2135
|
+
page.words.forEach((word, i) => {
|
|
2136
|
+
if (chromeWords.has(`${pageIndex}:${i}`)) return;
|
|
2137
|
+
if (!inChromeBand(page, word)) return;
|
|
2138
|
+
const row = page.words.filter((w) => sameRow(w, word));
|
|
2139
|
+
if (row.every((w) => PAGE_NUMBER_ROW.test(w.text))) {
|
|
2140
|
+
claimRow(pageIndex, word);
|
|
2141
|
+
}
|
|
2142
|
+
});
|
|
2143
|
+
});
|
|
2144
|
+
const body = indexDocument(pages, chromeWords);
|
|
2145
|
+
const claimed = /* @__PURE__ */ new Set();
|
|
2146
|
+
let cursor = 0;
|
|
2147
|
+
for (const entry of inventory) {
|
|
2148
|
+
if (entry.repeats) continue;
|
|
2149
|
+
const segments = needleSegments(entry.text);
|
|
2150
|
+
const needle = segments.join("");
|
|
2151
|
+
if (needle.length < MIN_NEEDLE_LENGTH) {
|
|
2152
|
+
results.set(entry, { entry, needle, status: "skipped", occurrences: [] });
|
|
2153
|
+
continue;
|
|
2154
|
+
}
|
|
2155
|
+
let all = findOccurrences(body, segments);
|
|
2156
|
+
let partial;
|
|
2157
|
+
if (all.length === 0 && segments.length === 1) {
|
|
2158
|
+
const prefix = longestRenderedPrefix(body, needle);
|
|
2159
|
+
if (prefix) {
|
|
2160
|
+
all = [occurrence(body, prefix.at, prefix.at + prefix.length)];
|
|
2161
|
+
partial = { matchedChars: prefix.length, totalChars: needle.length };
|
|
2162
|
+
}
|
|
2163
|
+
}
|
|
2164
|
+
if (all.length === 0) {
|
|
2165
|
+
results.set(entry, { entry, needle, status: "missing", occurrences: [] });
|
|
2166
|
+
continue;
|
|
2167
|
+
}
|
|
2168
|
+
const free = all.filter((o) => !claimed.has(o.at));
|
|
2169
|
+
if (free.length === 0) {
|
|
2170
|
+
results.set(entry, {
|
|
2171
|
+
entry,
|
|
2172
|
+
needle,
|
|
2173
|
+
status: "ambiguous",
|
|
2174
|
+
occurrences: all
|
|
2175
|
+
});
|
|
2176
|
+
continue;
|
|
2177
|
+
}
|
|
2178
|
+
const chosen = free.find((o) => o.at >= cursor) ?? free[0];
|
|
2179
|
+
claimed.add(chosen.at);
|
|
2180
|
+
cursor = chosen.at;
|
|
2181
|
+
results.set(entry, {
|
|
2182
|
+
entry,
|
|
2183
|
+
needle,
|
|
2184
|
+
status: "mapped",
|
|
2185
|
+
occurrences: [chosen],
|
|
2186
|
+
...partial && { partial }
|
|
2187
|
+
});
|
|
2188
|
+
}
|
|
2189
|
+
return {
|
|
2190
|
+
matches: inventory.map((entry) => results.get(entry)),
|
|
2191
|
+
chromeWords
|
|
2192
|
+
};
|
|
2193
|
+
}
|
|
2194
|
+
|
|
2195
|
+
// src/rendered-analysis.ts
|
|
2196
|
+
var VISIBLE_SPILL_PT = 2;
|
|
2197
|
+
var OVERLAP_FRACTION = 0.3;
|
|
2198
|
+
function finding(draft) {
|
|
2199
|
+
const { ruleId, mapping, page, context, ...rest } = draft;
|
|
2200
|
+
return {
|
|
2201
|
+
source: "quality",
|
|
2202
|
+
ruleId,
|
|
2203
|
+
certainty: "rendered",
|
|
2204
|
+
blocking: false,
|
|
2205
|
+
...rest,
|
|
2206
|
+
context: {
|
|
2207
|
+
...context,
|
|
2208
|
+
mapping,
|
|
2209
|
+
...page !== void 0 && { page }
|
|
2210
|
+
}
|
|
2211
|
+
};
|
|
2212
|
+
}
|
|
2213
|
+
function round(value) {
|
|
2214
|
+
return Math.round(value * 10) / 10;
|
|
2215
|
+
}
|
|
2216
|
+
function intersects(a, b) {
|
|
2217
|
+
const w = Math.min(a.xMax, b.xMax) - Math.max(a.xMin, b.xMin);
|
|
2218
|
+
const h = Math.min(a.yMax, b.yMax) - Math.max(a.yMin, b.yMin);
|
|
2219
|
+
return w > 0 && h > 0 ? w * h : 0;
|
|
2220
|
+
}
|
|
2221
|
+
function area(w) {
|
|
2222
|
+
return Math.max(0, w.xMax - w.xMin) * Math.max(0, w.yMax - w.yMin);
|
|
2223
|
+
}
|
|
2224
|
+
function excerpt(text, max = 60) {
|
|
2225
|
+
const flat = text.replace(/\s+/g, " ").trim();
|
|
2226
|
+
return flat.length > max ? `${flat.slice(0, max - 1)}\u2026` : flat;
|
|
2227
|
+
}
|
|
2228
|
+
function wordOwners(matches) {
|
|
2229
|
+
const owners = /* @__PURE__ */ new Map();
|
|
2230
|
+
for (const match of matches) {
|
|
2231
|
+
if (match.status !== "mapped") continue;
|
|
2232
|
+
for (const o of match.occurrences) {
|
|
2233
|
+
for (const part of o.parts) {
|
|
2234
|
+
for (const w of part.words) owners.set(`${part.pageIndex}:${w}`, match);
|
|
2235
|
+
}
|
|
2236
|
+
}
|
|
2237
|
+
}
|
|
2238
|
+
return owners;
|
|
2239
|
+
}
|
|
2240
|
+
var lineMaps = /* @__PURE__ */ new WeakMap();
|
|
2241
|
+
function lineOfWord(page) {
|
|
2242
|
+
const cached2 = lineMaps.get(page);
|
|
2243
|
+
if (cached2) return cached2;
|
|
2244
|
+
const lines = /* @__PURE__ */ new Map();
|
|
2245
|
+
page.lines.forEach((line, index) => {
|
|
2246
|
+
for (const w of line.words) lines.set(w, index);
|
|
2247
|
+
});
|
|
2248
|
+
lineMaps.set(page, lines);
|
|
2249
|
+
return lines;
|
|
2250
|
+
}
|
|
2251
|
+
function unionBox(parts) {
|
|
2252
|
+
return parts.reduce(
|
|
2253
|
+
(box, part) => ({
|
|
2254
|
+
xMin: Math.min(box.xMin, part.xMin),
|
|
2255
|
+
yMin: Math.min(box.yMin, part.yMin),
|
|
2256
|
+
xMax: Math.max(box.xMax, part.xMax),
|
|
2257
|
+
yMax: Math.max(box.yMax, part.yMax)
|
|
2258
|
+
}),
|
|
2259
|
+
{ xMin: Infinity, yMin: Infinity, xMax: -Infinity, yMax: -Infinity }
|
|
2260
|
+
);
|
|
2261
|
+
}
|
|
2262
|
+
function analyzeRenderedDocument(input) {
|
|
2263
|
+
const { pages, format } = input;
|
|
2264
|
+
const { matches, chromeWords } = assignInventory(pages, input.inventory);
|
|
2265
|
+
const owners = wordOwners(matches);
|
|
2266
|
+
const findings = [];
|
|
2267
|
+
const ownerOf = (pageIndex, word) => owners.get(`${pageIndex}:${word}`);
|
|
2268
|
+
const clipped = /* @__PURE__ */ new Map();
|
|
2269
|
+
pages.forEach((page, pageIndex) => {
|
|
2270
|
+
page.words.forEach((word, index) => {
|
|
2271
|
+
const beyond = Math.max(
|
|
2272
|
+
-word.xMin,
|
|
2273
|
+
word.xMax - page.widthPt,
|
|
2274
|
+
-word.yMin,
|
|
2275
|
+
word.yMax - page.heightPt
|
|
2276
|
+
);
|
|
2277
|
+
if (beyond <= VISIBLE_SPILL_PT) return;
|
|
2278
|
+
const owner = ownerOf(pageIndex, index);
|
|
2279
|
+
const key = owner ? `entry:${owner.entry.path}` : `page:${pageIndex}`;
|
|
2280
|
+
const bucket = clipped.get(key) ?? {
|
|
2281
|
+
match: owner,
|
|
2282
|
+
words: [],
|
|
2283
|
+
page: pageIndex + 1,
|
|
2284
|
+
maxPt: 0
|
|
2285
|
+
};
|
|
2286
|
+
bucket.words.push(word);
|
|
2287
|
+
bucket.maxPt = Math.max(bucket.maxPt, beyond);
|
|
2288
|
+
clipped.set(key, bucket);
|
|
2289
|
+
});
|
|
2290
|
+
});
|
|
2291
|
+
for (const bucket of clipped.values()) {
|
|
2292
|
+
const sample = excerpt(bucket.words.map((w) => w.text).join(" "));
|
|
2293
|
+
findings.push(
|
|
2294
|
+
finding({
|
|
2295
|
+
ruleId: "rendered/clip",
|
|
2296
|
+
code: QUALITY_CODES.RENDERED_CLIP,
|
|
2297
|
+
category: "integrity",
|
|
2298
|
+
severity: "warning",
|
|
2299
|
+
mapping: bucket.match ? "mapped" : "unmapped",
|
|
2300
|
+
page: bucket.page,
|
|
2301
|
+
path: bucket.match?.entry.path ?? "",
|
|
2302
|
+
message: `${bucket.words.length} word${bucket.words.length === 1 ? "" : "s"} rendered past the page edge on page ${bucket.page} by up to ${round(bucket.maxPt)} pt: "${sample}".`,
|
|
2303
|
+
suggestion: "Shorten the text, allow it to wrap, or widen the column or frame that holds it; text outside the page is cut off in print and in Word.",
|
|
2304
|
+
evidence: {
|
|
2305
|
+
summary: "Word boxes beyond the page bounds",
|
|
2306
|
+
actual: round(bucket.maxPt),
|
|
2307
|
+
expected: 0,
|
|
2308
|
+
unit: "pt",
|
|
2309
|
+
values: { words: bucket.words.length }
|
|
2310
|
+
}
|
|
2311
|
+
})
|
|
2312
|
+
);
|
|
2313
|
+
}
|
|
2314
|
+
for (const match of matches) {
|
|
2315
|
+
if (!match.partial || match.status !== "mapped") continue;
|
|
2316
|
+
const o = match.occurrences[0];
|
|
2317
|
+
const shown = Math.round(
|
|
2318
|
+
100 * match.partial.matchedChars / match.partial.totalChars
|
|
2319
|
+
);
|
|
2320
|
+
findings.push(
|
|
2321
|
+
finding({
|
|
2322
|
+
ruleId: "rendered/clip",
|
|
2323
|
+
code: QUALITY_CODES.RENDERED_CLIP,
|
|
2324
|
+
category: "integrity",
|
|
2325
|
+
severity: "warning",
|
|
2326
|
+
mapping: "mapped",
|
|
2327
|
+
page: o.endPageIndex + 1,
|
|
2328
|
+
path: match.entry.path,
|
|
2329
|
+
message: `"${excerpt(match.entry.text)}" is cut off: about ${shown}% of it rendered on page ${o.endPageIndex + 1} and the rest is nowhere on the page.`,
|
|
2330
|
+
suggestion: "Shorten the text or enlarge the frame, box or cell that holds it; what did not render here will not print either.",
|
|
2331
|
+
evidence: {
|
|
2332
|
+
summary: "Share of the authored text found in the PDF",
|
|
2333
|
+
actual: shown,
|
|
2334
|
+
expected: 100,
|
|
2335
|
+
unit: "%",
|
|
2336
|
+
values: match.partial
|
|
2337
|
+
},
|
|
2338
|
+
context: { kind: "truncated" }
|
|
2339
|
+
})
|
|
2340
|
+
);
|
|
2341
|
+
}
|
|
2342
|
+
for (const match of matches) {
|
|
2343
|
+
const box = match.entry.box;
|
|
2344
|
+
if (!box || match.status !== "mapped") continue;
|
|
2345
|
+
for (const o of match.occurrences) {
|
|
2346
|
+
const extent = unionBox(o.parts);
|
|
2347
|
+
const rendered = {
|
|
2348
|
+
heightPt: extent.yMax - extent.yMin,
|
|
2349
|
+
widthPt: extent.xMax - extent.xMin
|
|
2350
|
+
};
|
|
2351
|
+
const worst = ["heightPt", "widthPt"].filter((axis) => box[axis] !== void 0).map((axis) => ({
|
|
2352
|
+
axis,
|
|
2353
|
+
declared: box[axis],
|
|
2354
|
+
rendered: rendered[axis],
|
|
2355
|
+
over: rendered[axis] - box[axis]
|
|
2356
|
+
})).sort((a, b) => b.over - a.over)[0];
|
|
2357
|
+
if (!worst || worst.over <= VISIBLE_SPILL_PT) continue;
|
|
2358
|
+
const word = worst.axis === "heightPt" ? "taller" : "wider";
|
|
2359
|
+
findings.push(
|
|
2360
|
+
finding({
|
|
2361
|
+
ruleId: "rendered/spill",
|
|
2362
|
+
code: QUALITY_CODES.RENDERED_SPILL,
|
|
2363
|
+
category: "integrity",
|
|
2364
|
+
severity: "warning",
|
|
2365
|
+
mapping: "mapped",
|
|
2366
|
+
page: o.pageIndex + 1,
|
|
2367
|
+
path: match.entry.path,
|
|
2368
|
+
message: `"${excerpt(match.entry.text)}" rendered ${round(worst.over)} pt ${word} than its ${round(worst.declared)} pt box on page ${o.pageIndex + 1}.`,
|
|
2369
|
+
suggestion: "Shorten the text, reduce its size, or enlarge the box; the renderer let it spill past the edge the author drew.",
|
|
2370
|
+
evidence: {
|
|
2371
|
+
summary: "Rendered extent against the declared box",
|
|
2372
|
+
actual: round(worst.rendered),
|
|
2373
|
+
expected: round(worst.declared),
|
|
2374
|
+
unit: "pt",
|
|
2375
|
+
values: { marginPt: round(-worst.over) }
|
|
2376
|
+
}
|
|
2377
|
+
})
|
|
2378
|
+
);
|
|
2379
|
+
}
|
|
2380
|
+
}
|
|
2381
|
+
pages.forEach((page, pageIndex) => {
|
|
2382
|
+
const lines = lineOfWord(page);
|
|
2383
|
+
const seen = /* @__PURE__ */ new Set();
|
|
2384
|
+
const words = page.words;
|
|
2385
|
+
const byTop = words.map((_, i) => i).sort((a, b) => words[a].yMin - words[b].yMin);
|
|
2386
|
+
for (let p = 0; p < byTop.length; p++) {
|
|
2387
|
+
const i = byTop[p];
|
|
2388
|
+
const a = words[i];
|
|
2389
|
+
for (let q = p + 1; q < byTop.length; q++) {
|
|
2390
|
+
const j = byTop[q];
|
|
2391
|
+
const b = words[j];
|
|
2392
|
+
if (b.yMin >= a.yMax) break;
|
|
2393
|
+
const shared = intersects(a, b);
|
|
2394
|
+
if (shared <= 0) continue;
|
|
2395
|
+
const lineA = lines.get(i);
|
|
2396
|
+
if (lineA !== void 0 && lineA === lines.get(j)) continue;
|
|
2397
|
+
const smaller = Math.min(area(a), area(b));
|
|
2398
|
+
if (smaller <= 0 || shared / smaller < OVERLAP_FRACTION) continue;
|
|
2399
|
+
const ownerA = ownerOf(pageIndex, i);
|
|
2400
|
+
const ownerB = ownerOf(pageIndex, j);
|
|
2401
|
+
const pathA = ownerA?.entry.path ?? "";
|
|
2402
|
+
const pathB = ownerB?.entry.path ?? "";
|
|
2403
|
+
const key = pathA && pathB ? [pathA, pathB].sort().join("|") : `${Math.min(i, j)}:${Math.max(i, j)}`;
|
|
2404
|
+
if (seen.has(key)) continue;
|
|
2405
|
+
seen.add(key);
|
|
2406
|
+
const mapping = ownerA && ownerB ? "mapped" : ownerA || ownerB ? "ambiguous" : "unmapped";
|
|
2407
|
+
findings.push(
|
|
2408
|
+
finding({
|
|
2409
|
+
ruleId: "rendered/overlap",
|
|
2410
|
+
code: QUALITY_CODES.RENDERED_OVERLAP,
|
|
2411
|
+
category: "integrity",
|
|
2412
|
+
severity: "warning",
|
|
2413
|
+
mapping,
|
|
2414
|
+
page: pageIndex + 1,
|
|
2415
|
+
path: pathA || pathB,
|
|
2416
|
+
...pathA && pathB && pathA !== pathB && { relatedPaths: [pathB] },
|
|
2417
|
+
message: `"${a.text}" and "${b.text}" are drawn over each other on page ${pageIndex + 1}.`,
|
|
2418
|
+
suggestion: "Give the two elements separate room: move one, reduce the text, or increase the spacing between them.",
|
|
2419
|
+
evidence: {
|
|
2420
|
+
summary: "Word boxes intersect",
|
|
2421
|
+
actual: round(100 * shared / smaller),
|
|
2422
|
+
expected: 0,
|
|
2423
|
+
unit: "% of the smaller word"
|
|
2424
|
+
}
|
|
2425
|
+
})
|
|
2426
|
+
);
|
|
2427
|
+
}
|
|
2428
|
+
}
|
|
2429
|
+
});
|
|
2430
|
+
for (const match of matches) {
|
|
2431
|
+
if (match.status !== "missing") continue;
|
|
2432
|
+
findings.push(
|
|
2433
|
+
finding({
|
|
2434
|
+
ruleId: "rendered/text-missing",
|
|
2435
|
+
code: QUALITY_CODES.RENDERED_TEXT_MISSING,
|
|
2436
|
+
category: "integrity",
|
|
2437
|
+
severity: "warning",
|
|
2438
|
+
mapping: "mapped",
|
|
2439
|
+
path: match.entry.path,
|
|
2440
|
+
message: `"${excerpt(match.entry.text)}" does not appear anywhere in the rendered pages.`,
|
|
2441
|
+
suggestion: "The text was fully clipped, hidden behind another element, or set somewhere the renderer could not place it; preview the page it belongs to.",
|
|
2442
|
+
evidence: { summary: "No rendered occurrence of the authored text" },
|
|
2443
|
+
context: { role: match.entry.role }
|
|
2444
|
+
})
|
|
2445
|
+
);
|
|
2446
|
+
}
|
|
2447
|
+
let substituted = 0;
|
|
2448
|
+
if (input.fonts && input.fonts.length > 0 && input.requestedFonts) {
|
|
2449
|
+
const seen = /* @__PURE__ */ new Set();
|
|
2450
|
+
for (const requested of input.requestedFonts) {
|
|
2451
|
+
const family = requested.family.trim();
|
|
2452
|
+
if (family === "" || seen.has(family.toLowerCase())) continue;
|
|
2453
|
+
seen.add(family.toLowerCase());
|
|
2454
|
+
if (familyRendered(family, input.fonts)) continue;
|
|
2455
|
+
substituted += 1;
|
|
2456
|
+
findings.push(
|
|
2457
|
+
finding({
|
|
2458
|
+
ruleId: "rendered/font-substituted",
|
|
2459
|
+
code: QUALITY_CODES.RENDERED_FONT_SUBSTITUTED,
|
|
2460
|
+
category: "brand",
|
|
2461
|
+
severity: requested.declared ? "warning" : "info",
|
|
2462
|
+
mapping: requested.path ? "mapped" : "unmapped",
|
|
2463
|
+
path: requested.path ?? "",
|
|
2464
|
+
message: requested.declared ? `"${family}" was declared with a source but the rendered PDF embeds no face of it; the renderer substituted another family.` : `"${family}" is not installed on this host, so the preview substituted another family; a recipient with the font sees the intended face.`,
|
|
2465
|
+
suggestion: requested.declared ? "Check the font source: the file or URL must resolve to a face LibreOffice can load, or the recipient will see the same substitution." : "Install the family on the host for a faithful preview, or declare a source for it in the document fonts so every renderer has it.",
|
|
2466
|
+
context: { declared: requested.declared === true },
|
|
2467
|
+
evidence: {
|
|
2468
|
+
summary: "Requested family against embedded PDF fonts",
|
|
2469
|
+
expected: family,
|
|
2470
|
+
actual: [...new Set(input.fonts.map((f) => f.baseName))].join(", ")
|
|
2471
|
+
}
|
|
2472
|
+
})
|
|
2473
|
+
);
|
|
2474
|
+
}
|
|
2475
|
+
}
|
|
2476
|
+
pages.forEach((page, pageIndex) => {
|
|
2477
|
+
if (page.words.length > 0) return;
|
|
2478
|
+
findings.push(
|
|
2479
|
+
finding({
|
|
2480
|
+
ruleId: "rendered/empty-page",
|
|
2481
|
+
code: QUALITY_CODES.RENDERED_EMPTY_PAGE,
|
|
2482
|
+
category: "integrity",
|
|
2483
|
+
severity: "info",
|
|
2484
|
+
mapping: "unmapped",
|
|
2485
|
+
page: pageIndex + 1,
|
|
2486
|
+
path: "",
|
|
2487
|
+
message: `Page ${pageIndex + 1} carries no text. A full-page figure is fine; a blank page from a stray break is not.`,
|
|
2488
|
+
suggestion: "Preview the page; remove the page break or the empty section if nothing was meant to be there."
|
|
2489
|
+
})
|
|
2490
|
+
);
|
|
2491
|
+
});
|
|
2492
|
+
if (format === "docx") {
|
|
2493
|
+
for (const match of matches) {
|
|
2494
|
+
if (match.entry.role !== "heading" || match.status !== "mapped") continue;
|
|
2495
|
+
const o = match.occurrences[0];
|
|
2496
|
+
const part = o.parts[o.parts.length - 1];
|
|
2497
|
+
if (part.pageIndex >= pages.length - 1) continue;
|
|
2498
|
+
const page = pages[part.pageIndex];
|
|
2499
|
+
const below = page.words.some(
|
|
2500
|
+
(w, i) => !chromeWords.has(`${part.pageIndex}:${i}`) && !part.words.includes(i) && w.yMin >= part.yMax - 1
|
|
2501
|
+
);
|
|
2502
|
+
if (below) continue;
|
|
2503
|
+
findings.push(
|
|
2504
|
+
finding({
|
|
2505
|
+
ruleId: "rendered/heading-stranded",
|
|
2506
|
+
code: QUALITY_CODES.RENDERED_HEADING_STRANDED,
|
|
2507
|
+
category: "composition",
|
|
2508
|
+
severity: "warning",
|
|
2509
|
+
mapping: "mapped",
|
|
2510
|
+
page: part.pageIndex + 1,
|
|
2511
|
+
path: match.entry.path,
|
|
2512
|
+
message: `Heading "${excerpt(match.entry.text)}" is the last line on page ${part.pageIndex + 1}; its content starts on the next page.`,
|
|
2513
|
+
suggestion: "Keep the heading with its first paragraph (keep-with-next), or move a page break so they land together."
|
|
2514
|
+
})
|
|
2515
|
+
);
|
|
2516
|
+
}
|
|
2517
|
+
for (const match of matches) {
|
|
2518
|
+
const role = match.entry.role;
|
|
2519
|
+
if (role !== "body" && role !== "list-item" || match.status !== "mapped")
|
|
2520
|
+
continue;
|
|
2521
|
+
const o = match.occurrences[0];
|
|
2522
|
+
if (o.pageIndex === o.endPageIndex) continue;
|
|
2523
|
+
const first = o.parts[0];
|
|
2524
|
+
const last = o.parts[o.parts.length - 1];
|
|
2525
|
+
const linesOn = (part) => {
|
|
2526
|
+
const lines = lineOfWord(pages[part.pageIndex]);
|
|
2527
|
+
const set = /* @__PURE__ */ new Set();
|
|
2528
|
+
for (const w of part.words) {
|
|
2529
|
+
const line = lines.get(w);
|
|
2530
|
+
if (line !== void 0) set.add(line);
|
|
2531
|
+
}
|
|
2532
|
+
return set.size;
|
|
2533
|
+
};
|
|
2534
|
+
const before = linesOn(first);
|
|
2535
|
+
const after = linesOn(last);
|
|
2536
|
+
if (before === 0 || after === 0) continue;
|
|
2537
|
+
if (before > 1 && after > 1) continue;
|
|
2538
|
+
const which = before === 1 ? "orphan" : "widow";
|
|
2539
|
+
findings.push(
|
|
2540
|
+
finding({
|
|
2541
|
+
ruleId: "rendered/paragraph-split",
|
|
2542
|
+
code: QUALITY_CODES.RENDERED_PARAGRAPH_SPLIT,
|
|
2543
|
+
category: "composition",
|
|
2544
|
+
severity: "info",
|
|
2545
|
+
mapping: "mapped",
|
|
2546
|
+
page: (which === "orphan" ? first.pageIndex : last.pageIndex) + 1,
|
|
2547
|
+
path: match.entry.path,
|
|
2548
|
+
message: which === "orphan" ? `"${excerpt(match.entry.text)}" leaves one line alone at the foot of page ${first.pageIndex + 1}.` : `"${excerpt(match.entry.text)}" leaves one line alone at the top of page ${last.pageIndex + 1}.`,
|
|
2549
|
+
suggestion: "Edit the paragraph a few words shorter or longer, or set widow/orphan control on the style.",
|
|
2550
|
+
evidence: {
|
|
2551
|
+
summary: "Lines on each side of the page break",
|
|
2552
|
+
values: { linesBefore: before, linesAfter: after }
|
|
2553
|
+
},
|
|
2554
|
+
context: { kind: which }
|
|
2555
|
+
})
|
|
2556
|
+
);
|
|
2557
|
+
}
|
|
2558
|
+
}
|
|
2559
|
+
const inventory = {
|
|
2560
|
+
mapped: 0,
|
|
2561
|
+
ambiguous: 0,
|
|
2562
|
+
missing: 0,
|
|
2563
|
+
skipped: 0
|
|
2564
|
+
};
|
|
2565
|
+
for (const match of matches) inventory[match.status] += 1;
|
|
2566
|
+
const byMapping = {
|
|
2567
|
+
mapped: 0,
|
|
2568
|
+
ambiguous: 0,
|
|
2569
|
+
unmapped: 0
|
|
2570
|
+
};
|
|
2571
|
+
for (const f of findings) {
|
|
2572
|
+
byMapping[f.context?.mapping ?? "unmapped"] += 1;
|
|
2573
|
+
}
|
|
2574
|
+
return {
|
|
2575
|
+
findings,
|
|
2576
|
+
summary: {
|
|
2577
|
+
pages: pages.length,
|
|
2578
|
+
words: pages.reduce((n, p) => n + p.words.length, 0),
|
|
2579
|
+
inventory,
|
|
2580
|
+
findings: byMapping,
|
|
2581
|
+
// The same gate the check itself runs under: an empty font list
|
|
2582
|
+
// verified nothing, and must not read as "nothing was substituted".
|
|
2583
|
+
...input.fonts && input.fonts.length > 0 && input.requestedFonts && {
|
|
2584
|
+
fonts: { requested: input.requestedFonts.length, substituted }
|
|
2585
|
+
}
|
|
2586
|
+
}
|
|
2587
|
+
};
|
|
2588
|
+
}
|
|
1835
2589
|
export {
|
|
1836
2590
|
DocxFormatAdapter,
|
|
1837
2591
|
FontconfigStager,
|
|
1838
2592
|
MacOSCoreTextStager,
|
|
1839
2593
|
NoopFontStager,
|
|
1840
2594
|
PptxFormatAdapter,
|
|
2595
|
+
VISIBLE_SPILL_PT,
|
|
1841
2596
|
WindowsFontStager,
|
|
2597
|
+
analyzeRenderedDocument,
|
|
2598
|
+
assignInventory,
|
|
2599
|
+
authoredTextForMatch,
|
|
1842
2600
|
clearRasterizerCache,
|
|
1843
2601
|
createAdapter,
|
|
1844
2602
|
createLibreOfficePptxBatchRasterizer,
|
|
1845
2603
|
createLibreOfficePptxRasterizer,
|
|
1846
2604
|
emitDiagnostic,
|
|
2605
|
+
extractPdfFonts,
|
|
1847
2606
|
extractPdfTextGeometry,
|
|
2607
|
+
familyRendered,
|
|
1848
2608
|
getFontStager,
|
|
1849
2609
|
getRasterizerCacheStats,
|
|
2610
|
+
normalizeForMatch,
|
|
2611
|
+
parsePdfFonts,
|
|
1850
2612
|
parsePdfTextBbox,
|
|
2613
|
+
pdffontsAvailable,
|
|
1851
2614
|
pdftotextAvailable,
|
|
1852
2615
|
runWithDiagnosticSink,
|
|
1853
2616
|
stderrDiagnosticSink
|