pdf-codec.js 5.0.1 → 5.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -0
- package/dist/codec.d.cts +1 -0
- package/dist/codec.d.ts +1 -0
- package/dist/font-read.cjs +94 -12
- package/dist/font-read.d.cts +7 -1
- package/dist/font-read.d.ts +7 -1
- package/dist/font-read.js +94 -12
- package/dist/index.cjs +7 -0
- package/dist/index.d.cts +2 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/interpret.cjs +23 -7
- package/dist/interpret.d.cts +9 -1
- package/dist/interpret.d.ts +9 -1
- package/dist/interpret.js +23 -7
- package/dist/layout.cjs +1 -0
- package/dist/layout.d.cts +4 -0
- package/dist/layout.d.ts +4 -0
- package/dist/layout.js +1 -0
- package/dist/raster.cjs +14 -8
- package/dist/raster.js +14 -8
- package/dist/read.cjs +1 -0
- package/dist/read.js +1 -0
- package/dist/text-group.cjs +227 -0
- package/dist/text-group.d.cts +75 -0
- package/dist/text-group.d.ts +75 -0
- package/dist/text-group.js +221 -0
- package/package.json +1 -1
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
//#region src/text-group.d.ts
|
|
2
|
+
interface PdfTextRunGeometry {
|
|
3
|
+
readonly text: string;
|
|
4
|
+
readonly xPt: number;
|
|
5
|
+
readonly yPt: number;
|
|
6
|
+
readonly sizePt: number;
|
|
7
|
+
readonly widthPt?: number;
|
|
8
|
+
readonly rotationDeg?: number;
|
|
9
|
+
readonly writingMode?: "vertical";
|
|
10
|
+
}
|
|
11
|
+
interface PdfTextBox {
|
|
12
|
+
readonly xPt: number;
|
|
13
|
+
readonly yPt: number;
|
|
14
|
+
readonly widthPt: number;
|
|
15
|
+
readonly heightPt: number;
|
|
16
|
+
}
|
|
17
|
+
type PdfWordSeparator = "none" | "space" | "column";
|
|
18
|
+
interface PdfTextWord<TRun extends PdfTextRunGeometry> {
|
|
19
|
+
readonly text: string;
|
|
20
|
+
readonly separatorBefore: PdfWordSeparator;
|
|
21
|
+
readonly runs: readonly TRun[];
|
|
22
|
+
readonly bounds?: PdfTextBox;
|
|
23
|
+
}
|
|
24
|
+
interface PdfTextLine<TRun extends PdfTextRunGeometry> {
|
|
25
|
+
readonly text: string;
|
|
26
|
+
readonly words: readonly PdfTextWord<TRun>[];
|
|
27
|
+
readonly baselineYPt: number;
|
|
28
|
+
readonly rotationDeg: number;
|
|
29
|
+
readonly writingMode?: "vertical";
|
|
30
|
+
readonly bounds?: PdfTextBox;
|
|
31
|
+
}
|
|
32
|
+
interface PdfTextGroupingOptions {
|
|
33
|
+
readonly baselineToleranceEm?: number;
|
|
34
|
+
readonly wordGapEm?: number;
|
|
35
|
+
readonly columnGapEm?: number;
|
|
36
|
+
}
|
|
37
|
+
declare const DEFAULT_BASELINE_TOLERANCE_EM: number;
|
|
38
|
+
declare const DEFAULT_WORD_GAP_EM: number;
|
|
39
|
+
declare const DEFAULT_COLUMN_GAP_EM = 1;
|
|
40
|
+
/**
|
|
41
|
+
* Whether two positioned runs sit on one visual line.
|
|
42
|
+
*
|
|
43
|
+
* The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
|
|
44
|
+
* @param a - one run
|
|
45
|
+
* @param b - the other run
|
|
46
|
+
* @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
|
|
47
|
+
* @returns true when the two baselines are within tolerance of each other
|
|
48
|
+
*/
|
|
49
|
+
declare function runsShareBaseline(a: PdfTextRunGeometry, b: PdfTextRunGeometry, options?: PdfTextGroupingOptions): boolean;
|
|
50
|
+
/**
|
|
51
|
+
* The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
|
|
52
|
+
*
|
|
53
|
+
* Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
|
|
54
|
+
* @param previous - the run to the left, in reading order
|
|
55
|
+
* @param next - the run that follows it
|
|
56
|
+
* @returns the gap in points, or undefined when it cannot be derived
|
|
57
|
+
*/
|
|
58
|
+
declare function runGapPt(previous: PdfTextRunGeometry, next: PdfTextRunGeometry): number | undefined;
|
|
59
|
+
/**
|
|
60
|
+
* Groups positioned PDF text runs into lines of words.
|
|
61
|
+
*
|
|
62
|
+
* Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
|
|
63
|
+
*
|
|
64
|
+
* Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
|
|
65
|
+
*
|
|
66
|
+
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
67
|
+
*
|
|
68
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
69
|
+
* @param runs - positioned text runs, in any order
|
|
70
|
+
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
71
|
+
* @returns the grouped lines, in reading order
|
|
72
|
+
*/
|
|
73
|
+
declare function groupPdfTextRuns<TRun extends PdfTextRunGeometry>(runs: readonly TRun[], options?: PdfTextGroupingOptions): PdfTextLine<TRun>[];
|
|
74
|
+
//#endregion
|
|
75
|
+
export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, PdfTextBox, PdfTextGroupingOptions, PdfTextLine, PdfTextRunGeometry, PdfTextWord, PdfWordSeparator, groupPdfTextRuns, runGapPt, runsShareBaseline };
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
//#region src/text-group.ts
|
|
2
|
+
const DEFAULT_BASELINE_TOLERANCE_EM = 2 / 3;
|
|
3
|
+
const DEFAULT_WORD_GAP_EM = 1 / 8;
|
|
4
|
+
const DEFAULT_COLUMN_GAP_EM = 1;
|
|
5
|
+
const ROTATION_NOISE_DEG = 1e-6;
|
|
6
|
+
const DEGREES_PER_TURN = 360;
|
|
7
|
+
const VERTICAL_ADVANCE_OFFSET_DEG = -90;
|
|
8
|
+
function advanceAngleDeg(run) {
|
|
9
|
+
const raw = (run.rotationDeg ?? 0) + (run.writingMode === "vertical" ? VERTICAL_ADVANCE_OFFSET_DEG : 0);
|
|
10
|
+
return (Math.round(raw / ROTATION_NOISE_DEG) * ROTATION_NOISE_DEG % DEGREES_PER_TURN + DEGREES_PER_TURN) % DEGREES_PER_TURN;
|
|
11
|
+
}
|
|
12
|
+
const RADIANS_PER_TURN = 2 * Math.PI;
|
|
13
|
+
function baselineAxis(rotationDeg) {
|
|
14
|
+
const radians = rotationDeg * RADIANS_PER_TURN / DEGREES_PER_TURN;
|
|
15
|
+
return {
|
|
16
|
+
cos: Math.cos(radians),
|
|
17
|
+
sin: Math.sin(radians)
|
|
18
|
+
};
|
|
19
|
+
}
|
|
20
|
+
function project(run, rotationDeg) {
|
|
21
|
+
const { cos, sin } = baselineAxis(rotationDeg);
|
|
22
|
+
return {
|
|
23
|
+
run,
|
|
24
|
+
alongPt: run.xPt * cos + run.yPt * sin,
|
|
25
|
+
acrossPt: run.yPt * cos - run.xPt * sin,
|
|
26
|
+
sizePt: run.sizePt
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
function unproject(alongPt, acrossPt, rotationDeg) {
|
|
30
|
+
const { cos, sin } = baselineAxis(rotationDeg);
|
|
31
|
+
return {
|
|
32
|
+
xPt: alongPt * cos - acrossPt * sin,
|
|
33
|
+
yPt: alongPt * sin + acrossPt * cos
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* Whether two positioned runs sit on one visual line.
|
|
38
|
+
*
|
|
39
|
+
* The tolerance is a fraction of an em of the smaller of the two runs, never of one nominated run, so a large heading can never claim the small line beneath it. Runs set at different angles never share a line.
|
|
40
|
+
* @param a - one run
|
|
41
|
+
* @param b - the other run
|
|
42
|
+
* @param options - only `baselineToleranceEm` is read; it defaults to {@link DEFAULT_BASELINE_TOLERANCE_EM}
|
|
43
|
+
* @returns true when the two baselines are within tolerance of each other
|
|
44
|
+
*/
|
|
45
|
+
function runsShareBaseline(a, b, options = {}) {
|
|
46
|
+
const rotationDeg = advanceAngleDeg(a);
|
|
47
|
+
if (rotationDeg !== advanceAngleDeg(b) || a.writingMode !== b.writingMode) return false;
|
|
48
|
+
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
49
|
+
return Math.abs(project(a, rotationDeg).acrossPt - project(b, rotationDeg).acrossPt) <= toleranceEm * Math.min(a.sizePt, b.sizePt);
|
|
50
|
+
}
|
|
51
|
+
/**
|
|
52
|
+
* The horizontal gap along the baseline between the end of `previous` and the start of `next`, or undefined when the geometry does not state one.
|
|
53
|
+
*
|
|
54
|
+
* Undefined has exactly two causes, and a caller must treat both as "no evidence of a space" rather than substituting zero: `previous` stated no advance width, so where it ends is unknown; or the two runs are set at different angles, so they share no axis to measure along. A negative result is real and means the two runs overlap.
|
|
55
|
+
* @param previous - the run to the left, in reading order
|
|
56
|
+
* @param next - the run that follows it
|
|
57
|
+
* @returns the gap in points, or undefined when it cannot be derived
|
|
58
|
+
*/
|
|
59
|
+
function runGapPt(previous, next) {
|
|
60
|
+
const rotationDeg = advanceAngleDeg(previous);
|
|
61
|
+
if (rotationDeg !== advanceAngleDeg(next) || previous.writingMode !== next.writingMode) return;
|
|
62
|
+
const previousWidthPt = previous.widthPt;
|
|
63
|
+
if (previousWidthPt === void 0) return;
|
|
64
|
+
return project(next, rotationDeg).alongPt - (project(previous, rotationDeg).alongPt + previousWidthPt);
|
|
65
|
+
}
|
|
66
|
+
function clusterIntoLines(entries, toleranceEm) {
|
|
67
|
+
const lines = [];
|
|
68
|
+
for (const entry of entries) {
|
|
69
|
+
let nearest;
|
|
70
|
+
for (const line of lines) if (runsShareBaseline(line.anchor.run, entry.run, { baselineToleranceEm: toleranceEm })) nearest = line;
|
|
71
|
+
if (nearest === void 0) lines.push({
|
|
72
|
+
anchor: entry,
|
|
73
|
+
entries: [entry]
|
|
74
|
+
});
|
|
75
|
+
else nearest.entries.push(entry);
|
|
76
|
+
}
|
|
77
|
+
return lines;
|
|
78
|
+
}
|
|
79
|
+
function tokeniseRun(text) {
|
|
80
|
+
const words = text.match(/\S+/g) ?? [];
|
|
81
|
+
const leadingSpace = /^\s/.test(text);
|
|
82
|
+
const trailingSpace = /\s$/.test(text);
|
|
83
|
+
return {
|
|
84
|
+
words,
|
|
85
|
+
leadingSpace,
|
|
86
|
+
trailingSpace,
|
|
87
|
+
whole: words.length === 1 && !leadingSpace && !trailingSpace
|
|
88
|
+
};
|
|
89
|
+
}
|
|
90
|
+
const SEPARATOR_TEXT = {
|
|
91
|
+
none: "",
|
|
92
|
+
space: " ",
|
|
93
|
+
column: " "
|
|
94
|
+
};
|
|
95
|
+
function separatorForGap(gapPt, smallerSizePt, wordGapEm, columnGapEm) {
|
|
96
|
+
if (gapPt === void 0) return "none";
|
|
97
|
+
if (gapPt >= columnGapEm * smallerSizePt) return "column";
|
|
98
|
+
if (gapPt >= wordGapEm * smallerSizePt) return "space";
|
|
99
|
+
return "none";
|
|
100
|
+
}
|
|
101
|
+
function boundsOf(minAlongPt, maxAlongEndPt, acrossPt, maxSizePt, rotationDeg) {
|
|
102
|
+
const origin = unproject(minAlongPt, acrossPt, rotationDeg);
|
|
103
|
+
return {
|
|
104
|
+
xPt: origin.xPt,
|
|
105
|
+
yPt: origin.yPt,
|
|
106
|
+
widthPt: maxAlongEndPt - minAlongPt,
|
|
107
|
+
heightPt: maxSizePt
|
|
108
|
+
};
|
|
109
|
+
}
|
|
110
|
+
function buildWords(line, rotationDeg, wordGapEm, columnGapEm) {
|
|
111
|
+
const working = [];
|
|
112
|
+
let previous;
|
|
113
|
+
for (const entry of line.entries) {
|
|
114
|
+
const tokens = tokeniseRun(entry.run.text);
|
|
115
|
+
const gapSeparator = previous === void 0 ? "none" : separatorForGap(runGapPt(previous.entry.run, entry.run), Math.min(previous.entry.sizePt, entry.sizePt), wordGapEm, columnGapEm);
|
|
116
|
+
const pendingSpace = previous?.trailingSpace === true;
|
|
117
|
+
let separator = "none";
|
|
118
|
+
if (working.length > 0) separator = gapSeparator === "none" && (pendingSpace || tokens.leadingSpace) ? "space" : gapSeparator;
|
|
119
|
+
const widthPt = entry.run.widthPt;
|
|
120
|
+
const alongEndPt = entry.alongPt + (widthPt ?? 0);
|
|
121
|
+
tokens.words.forEach((word, index) => {
|
|
122
|
+
const last = working[working.length - 1];
|
|
123
|
+
if (index === 0 && separator === "none" && last !== void 0) {
|
|
124
|
+
last.text += word;
|
|
125
|
+
last.runs.push(entry.run);
|
|
126
|
+
last.measurable = last.measurable && tokens.whole && widthPt !== void 0;
|
|
127
|
+
last.minAlongPt = Math.min(last.minAlongPt, entry.alongPt);
|
|
128
|
+
last.maxAlongEndPt = Math.max(last.maxAlongEndPt, alongEndPt);
|
|
129
|
+
last.maxSizePt = Math.max(last.maxSizePt, entry.sizePt);
|
|
130
|
+
return;
|
|
131
|
+
}
|
|
132
|
+
working.push({
|
|
133
|
+
text: word,
|
|
134
|
+
separatorBefore: index === 0 ? separator : "space",
|
|
135
|
+
runs: [entry.run],
|
|
136
|
+
measurable: tokens.whole && widthPt !== void 0,
|
|
137
|
+
minAlongPt: entry.alongPt,
|
|
138
|
+
maxAlongEndPt: alongEndPt,
|
|
139
|
+
maxSizePt: entry.sizePt
|
|
140
|
+
});
|
|
141
|
+
});
|
|
142
|
+
previous = {
|
|
143
|
+
entry,
|
|
144
|
+
trailingSpace: tokens.trailingSpace
|
|
145
|
+
};
|
|
146
|
+
}
|
|
147
|
+
return working.map((word) => ({
|
|
148
|
+
text: word.text,
|
|
149
|
+
separatorBefore: word.separatorBefore,
|
|
150
|
+
runs: word.runs,
|
|
151
|
+
...word.measurable ? { bounds: boundsOf(word.minAlongPt, word.maxAlongEndPt, line.anchor.acrossPt, word.maxSizePt, rotationDeg) } : {}
|
|
152
|
+
}));
|
|
153
|
+
}
|
|
154
|
+
function buildLine(line, rotationDeg, vertical, wordGapEm, columnGapEm) {
|
|
155
|
+
const words = buildWords(line, rotationDeg, wordGapEm, columnGapEm);
|
|
156
|
+
const text = words.map((word) => SEPARATOR_TEXT[word.separatorBefore] + word.text).join("");
|
|
157
|
+
let minAlongPt = Number.POSITIVE_INFINITY;
|
|
158
|
+
let maxAlongEndPt = Number.NEGATIVE_INFINITY;
|
|
159
|
+
let maxSizePt = 0;
|
|
160
|
+
let measurable = true;
|
|
161
|
+
for (const entry of line.entries) {
|
|
162
|
+
const widthPt = entry.run.widthPt;
|
|
163
|
+
if (widthPt === void 0) measurable = false;
|
|
164
|
+
minAlongPt = Math.min(minAlongPt, entry.alongPt);
|
|
165
|
+
maxAlongEndPt = Math.max(maxAlongEndPt, entry.alongPt + (widthPt ?? 0));
|
|
166
|
+
maxSizePt = Math.max(maxSizePt, entry.sizePt);
|
|
167
|
+
}
|
|
168
|
+
return {
|
|
169
|
+
text,
|
|
170
|
+
words,
|
|
171
|
+
baselineYPt: line.anchor.acrossPt,
|
|
172
|
+
rotationDeg,
|
|
173
|
+
...vertical ? { writingMode: "vertical" } : {},
|
|
174
|
+
...measurable ? { bounds: boundsOf(minAlongPt, maxAlongEndPt, line.anchor.acrossPt, maxSizePt, rotationDeg) } : {}
|
|
175
|
+
};
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
* Groups positioned PDF text runs into lines of words.
|
|
179
|
+
*
|
|
180
|
+
* Runs are grouped by baseline into lines, ordered down the page, and each line's runs are ordered along the baseline and split into words wherever the geometry between them, or the whitespace their own text carries, says a word ends. A gap wider than a full em of the smaller of the two runs is reported as a column boundary rather than a word space, so a table's columns stay distinct instead of reading as one sentence.
|
|
181
|
+
*
|
|
182
|
+
* Runs set at different angles are never grouped together: each angle is grouped in its own frame and reported as its own block of lines, unrotated text first and the remaining angles in ascending order. No attempt is made to interleave rotated text into the reading order of the unrotated text around it, because a page's geometry alone does not say where a rotated block belongs in that order.
|
|
183
|
+
*
|
|
184
|
+
* Grouping is by baseline alone, with no page segmentation: on a multi-column page whose columns are set at different vertical offsets, a line of one column can fall within tolerance of a line of the next and the two are reported as one line, because from geometry alone at this level that is what they are. The gutter between them still reads as a column boundary, so the separator says where to cut. A caller that needs the columns apart segments the page first, which is what document-outline.js's `segmentPdfRegions` is for, and groups each region's own runs.
|
|
185
|
+
*
|
|
186
|
+
* Known limitations, all of them inherent to what a PDF states rather than to this implementation. Text is reported in visual order, left to right along the baseline: a right-to-left script arrives from the content stream already laid out visually and carries no direction of its own, so a consumer needing logical order applies the Unicode Bidi Algorithm to the result. A vertically set run is grouped into columns rather than lines, from the writingMode its own reader reported, and its column is reported with a rotationDeg of 270 and its own writingMode; columns order right to left, the direction vertical setting is read in. What is NOT recognised is a predefined non-Identity CMap's own byte decoding, so a vertically set font using one (90ms-RKSJ-V and its siblings) still has its codes read as two-byte CIDs, which is this package's own long-standing composite-font limit rather than anything about writing mode. A word hyphenated across a line end is left split, with its hyphen intact, because rejoining it needs to know the two lines belong to one paragraph, which is a semantic judgement this package deliberately leaves to its consumers. Runs with empty text are dropped, and so are the empty words a whitespace-only run would otherwise produce, but no other normalisation of the decoded text is attempted: a ligature and a zero-width character each reach the output exactly as the font's ToUnicode mapping spelled them.
|
|
187
|
+
* @param runs - positioned text runs, in any order
|
|
188
|
+
* @param options - tolerance overrides; each defaults to the exported constant of the same name
|
|
189
|
+
* @returns the grouped lines, in reading order
|
|
190
|
+
*/
|
|
191
|
+
function groupPdfTextRuns(runs, options = {}) {
|
|
192
|
+
const toleranceEm = options.baselineToleranceEm ?? .6666666666666666;
|
|
193
|
+
const wordGapEm = options.wordGapEm ?? .125;
|
|
194
|
+
const columnGapEm = options.columnGapEm ?? 1;
|
|
195
|
+
const buckets = /* @__PURE__ */ new Map();
|
|
196
|
+
for (const run of runs) {
|
|
197
|
+
if (run.text.length === 0) continue;
|
|
198
|
+
const rotationDeg = advanceAngleDeg(run);
|
|
199
|
+
const vertical = run.writingMode === "vertical";
|
|
200
|
+
const key = `${String(rotationDeg)}|${String(vertical)}`;
|
|
201
|
+
const bucket = buckets.get(key);
|
|
202
|
+
if (bucket === void 0) buckets.set(key, {
|
|
203
|
+
rotationDeg,
|
|
204
|
+
vertical,
|
|
205
|
+
runs: [run]
|
|
206
|
+
});
|
|
207
|
+
else bucket.runs.push(run);
|
|
208
|
+
}
|
|
209
|
+
const result = [];
|
|
210
|
+
const ordered = [...buckets.values()].sort((a, b) => a.rotationDeg - b.rotationDeg || Number(a.vertical) - Number(b.vertical));
|
|
211
|
+
for (const bucket of ordered) {
|
|
212
|
+
const entries = bucket.runs.map((run) => project(run, bucket.rotationDeg)).sort((a, b) => b.acrossPt - a.acrossPt || a.alongPt - b.alongPt);
|
|
213
|
+
for (const line of clusterIntoLines(entries, toleranceEm)) {
|
|
214
|
+
line.entries.sort((a, b) => a.alongPt - b.alongPt);
|
|
215
|
+
result.push(buildLine(line, bucket.rotationDeg, bucket.vertical, wordGapEm, columnGapEm));
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
return result;
|
|
219
|
+
}
|
|
220
|
+
//#endregion
|
|
221
|
+
export { DEFAULT_BASELINE_TOLERANCE_EM, DEFAULT_COLUMN_GAP_EM, DEFAULT_WORD_GAP_EM, groupPdfTextRuns, runGapPt, runsShareBaseline };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pdf-codec.js",
|
|
3
|
-
"version": "5.0
|
|
3
|
+
"version": "5.2.0",
|
|
4
4
|
"description": "Hand-written, dependency-minimal PDF codec: parses arbitrary real-world PDFs and generates new ones, built on its own codec-owned LayoutDocument item model and Zod 4 codecs.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"repository": {
|