reamkit 1.30.0 → 1.31.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -24
- package/dist/esm/core/converter/project.js +3 -1
- package/dist/esm/core/crypto/offcrypto.js +1 -1
- package/dist/esm/core/document-model/index.d.ts +1 -1
- package/dist/esm/core/document-model/types.d.ts +63 -0
- package/dist/esm/core/font/index.d.ts +2 -0
- package/dist/esm/core/font/ttf-build.d.ts +113 -0
- package/dist/esm/core/font/ttf-build.js +1224 -0
- package/dist/esm/core/font/ttf-subset.d.ts +19 -0
- package/dist/esm/core/font/ttf-subset.js +14 -1
- package/dist/esm/core/fonts/provider.d.ts +17 -0
- package/dist/esm/core/fonts/provider.js +27 -2
- package/dist/esm/core/fonts/remote-fonts.js +8 -1
- package/dist/esm/core/ir/flow.d.ts +71 -1
- package/dist/esm/core/numbering/index.d.ts +1 -1
- package/dist/esm/core/numbering/state.d.ts +11 -1
- package/dist/esm/core/numbering/state.js +10 -1
- package/dist/esm/core/style-cascade/resolver.js +35 -4
- package/dist/esm/core/style-cascade/types.d.ts +19 -1
- package/dist/esm/core/style-cascade/types.js +3 -0
- package/dist/esm/index.d.ts +2 -2
- package/dist/esm/layout/page-doc.d.ts +10 -3
- package/dist/esm/layout/styled-layout.d.ts +48 -7
- package/dist/esm/layout/styled-layout.js +881 -119
- package/dist/esm/layout/turned-section.d.ts +38 -0
- package/dist/esm/layout/turned-section.js +193 -0
- package/dist/esm/pdf/styled-page-emitter.js +78 -2
- package/dist/esm/pdf-reader/cff-outline.d.ts +27 -0
- package/dist/esm/pdf-reader/cff-outline.js +169 -21
- package/dist/esm/pdf-reader/content.d.ts +53 -0
- package/dist/esm/pdf-reader/content.js +9 -1
- package/dist/esm/pdf-reader/display.d.ts +36 -0
- package/dist/esm/pdf-reader/display.js +66 -1
- package/dist/esm/pdf-reader/document.js +5 -1
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +2 -1
- package/dist/esm/pdf-reader/embedded-fonts.js +32 -2
- package/dist/esm/pdf-reader/encodings.d.ts +8 -0
- package/dist/esm/pdf-reader/encodings.js +25 -3
- package/dist/esm/pdf-reader/face-outlines.d.ts +78 -0
- package/dist/esm/pdf-reader/face-outlines.js +362 -0
- package/dist/esm/pdf-reader/figures.d.ts +52 -0
- package/dist/esm/pdf-reader/figures.js +433 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +131 -5
- package/dist/esm/pdf-reader/flow-build.js +345 -26
- package/dist/esm/pdf-reader/font.js +321 -36
- package/dist/esm/pdf-reader/glyf-outline.d.ts +33 -0
- package/dist/esm/pdf-reader/glyf-outline.js +135 -1
- package/dist/esm/pdf-reader/glyph-names.js +154 -1
- package/dist/esm/pdf-reader/layout.d.ts +93 -0
- package/dist/esm/pdf-reader/layout.js +1703 -214
- package/dist/esm/pdf-reader/page-numbers.d.ts +53 -0
- package/dist/esm/pdf-reader/page-numbers.js +167 -0
- package/dist/esm/pdf-reader/tagged.js +182 -23
- package/dist/esm/pdf-reader/text.d.ts +6 -3
- package/dist/esm/pdf-reader/text.js +98 -9
- package/dist/esm/pdf-reader/type1-outline.d.ts +11 -0
- package/dist/esm/pdf-reader/type1-outline.js +63 -8
- package/dist/esm/pdf-reader/vector.js +71 -1
- package/dist/esm/word/doc/doc-reader.js +6 -2
- package/dist/esm/word/doc/doc-text.d.ts +6 -0
- package/dist/esm/word/doc/doc-text.js +19 -1
- package/dist/esm/word/document-parser.d.ts +2 -2
- package/dist/esm/word/document-parser.js +11 -1
- package/dist/esm/word/docx-reader.js +5 -3
- package/dist/esm/word/docx-writer.js +207 -21
- package/dist/esm/word/drawing-parser.d.ts +5 -3
- package/dist/esm/word/drawing-parser.js +49 -8
- package/dist/esm/word/font-embed.d.ts +30 -0
- package/dist/esm/word/font-embed.js +173 -0
- package/dist/esm/word/font-table.d.ts +10 -0
- package/dist/esm/word/font-table.js +13 -1
- package/dist/esm/word/index.js +1 -1
- package/dist/esm/word/numbering-parser.d.ts +3 -1
- package/dist/esm/word/numbering-parser.js +2 -1
- package/dist/esm/word/paragraph-properties.d.ts +7 -6
- package/dist/esm/word/paragraph-properties.js +14 -2
- package/dist/esm/word/run-properties.js +26 -0
- package/package.json +8 -3
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import { NumberingFormat } from '../core/document-model/index.js';
|
|
2
|
+
/** A page's number as its band prints it. */
|
|
3
|
+
export interface PageNumber {
|
|
4
|
+
/** The numeral as the page shows it: "iv", "47". */
|
|
5
|
+
readonly text: string;
|
|
6
|
+
readonly value: number;
|
|
7
|
+
readonly format: NumberingFormat;
|
|
8
|
+
}
|
|
9
|
+
/** A stretch of pages counted in one sequence (§17.6.12). */
|
|
10
|
+
export interface NumberingRun {
|
|
11
|
+
/** The first page of the stretch. */
|
|
12
|
+
readonly from: number;
|
|
13
|
+
readonly format: NumberingFormat;
|
|
14
|
+
/** The number the stretch's first page carries. */
|
|
15
|
+
readonly start: number;
|
|
16
|
+
}
|
|
17
|
+
/** What the pages' bands say about their numbering. */
|
|
18
|
+
export interface PageNumbering {
|
|
19
|
+
/** Each page's number, where its band shows one. */
|
|
20
|
+
readonly numbers: ReadonlyArray<PageNumber | undefined>;
|
|
21
|
+
/** The sequences the pages are counted in, first page first. */
|
|
22
|
+
readonly runs: ReadonlyArray<NumberingRun>;
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* The numerals a band's text holds, in order.
|
|
26
|
+
*
|
|
27
|
+
* @param text The band's text on one page.
|
|
28
|
+
* @returns Its numerals; a roman one only where it is a numeral as written.
|
|
29
|
+
*/
|
|
30
|
+
export declare function numeralsIn(text: string): Array<PageNumber>;
|
|
31
|
+
/**
|
|
32
|
+
* The page numbers a document's running band prints, and the sequences they
|
|
33
|
+
* run in.
|
|
34
|
+
*
|
|
35
|
+
* The number is the numeral that CHANGES from page to page: "Page 3 of 10" is
|
|
36
|
+
* counted by its first, "Chapter I — 47" by its last. A sequence goes on while
|
|
37
|
+
* each page's number is one more than the last in the same numerals, and
|
|
38
|
+
* starts again where it does not — i, ii, iii and then 1 are two sequences.
|
|
39
|
+
* A page whose band shows no number is counted on.
|
|
40
|
+
*
|
|
41
|
+
* @param bands Each page's band text, or undefined where the page has none.
|
|
42
|
+
* @returns The numbering, or undefined where the bands show no page number
|
|
43
|
+
* worth reading as one.
|
|
44
|
+
*/
|
|
45
|
+
export declare function pageNumberingOf(bands: ReadonlyArray<string | undefined>): PageNumbering | undefined;
|
|
46
|
+
/**
|
|
47
|
+
* The sequence a page is counted in.
|
|
48
|
+
*
|
|
49
|
+
* @param runs The document's sequences, first page first.
|
|
50
|
+
* @param page The page.
|
|
51
|
+
* @returns The run that page belongs to.
|
|
52
|
+
*/
|
|
53
|
+
export declare function runOf(runs: ReadonlyArray<NumberingRun>, page: number): NumberingRun | undefined;
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
//#region src/pdf-reader/page-numbers.ts
|
|
2
|
+
/** A numeral standing as a word of its own: digits, or roman in one case. */
|
|
3
|
+
var NUMERAL = /(?<=^|\s)(\d{1,4}|[ivxlcdm]{1,9}|[IVXLCDM]{1,9})(?=\s|$)/gu;
|
|
4
|
+
/**
|
|
5
|
+
* More sequences than this and the "numbers" are something else that changes
|
|
6
|
+
* from page to page — a date, a figure count — and the pages are left counted
|
|
7
|
+
* the ordinary way.
|
|
8
|
+
*/
|
|
9
|
+
var MOST_RUNS = 4;
|
|
10
|
+
/**
|
|
11
|
+
* …unless the sequences run long: a book leaves its blank pages out and skips
|
|
12
|
+
* their numbers, and freeculture.pdf's count jumps two at five of its chapters
|
|
13
|
+
* — seven sequences over three hundred and thirty-five numbered pages, and
|
|
14
|
+
* read as none its footer printed no number at all. A date or a figure count
|
|
15
|
+
* starts a sequence of its own on nearly every page.
|
|
16
|
+
*/
|
|
17
|
+
var LEAST_RUN_PAGES = 10;
|
|
18
|
+
/**
|
|
19
|
+
* The numerals a band's text holds, in order.
|
|
20
|
+
*
|
|
21
|
+
* @param text The band's text on one page.
|
|
22
|
+
* @returns Its numerals; a roman one only where it is a numeral as written.
|
|
23
|
+
*/
|
|
24
|
+
function numeralsIn(text) {
|
|
25
|
+
const out = [];
|
|
26
|
+
for (const m of text.matchAll(NUMERAL)) {
|
|
27
|
+
const token = m[1];
|
|
28
|
+
if (/^\d+$/u.test(token)) {
|
|
29
|
+
out.push({
|
|
30
|
+
text: token,
|
|
31
|
+
value: Number(token),
|
|
32
|
+
format: "decimal"
|
|
33
|
+
});
|
|
34
|
+
continue;
|
|
35
|
+
}
|
|
36
|
+
const value = romanValue(token);
|
|
37
|
+
if (value === void 0) continue;
|
|
38
|
+
out.push({
|
|
39
|
+
text: token,
|
|
40
|
+
value,
|
|
41
|
+
format: token === token.toLowerCase() ? "lowerRoman" : "upperRoman"
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
return out;
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* A roman numeral's value, where the letters ARE one as a numeral is written
|
|
48
|
+
* — "iv", "xiv", "MCMXC" — and not merely letters a numeral uses: "ic", "vx"
|
|
49
|
+
* and "mix" are words or nothing.
|
|
50
|
+
*/
|
|
51
|
+
function romanValue(token) {
|
|
52
|
+
const upper = token.toUpperCase();
|
|
53
|
+
const worth = {
|
|
54
|
+
I: 1,
|
|
55
|
+
V: 5,
|
|
56
|
+
X: 10,
|
|
57
|
+
L: 50,
|
|
58
|
+
C: 100,
|
|
59
|
+
D: 500,
|
|
60
|
+
M: 1e3
|
|
61
|
+
};
|
|
62
|
+
let value = 0;
|
|
63
|
+
for (let i = 0; i < upper.length; i++) {
|
|
64
|
+
const here = worth[upper[i]];
|
|
65
|
+
const next = worth[upper[i + 1] ?? ""] ?? 0;
|
|
66
|
+
value += here < next ? -here : here;
|
|
67
|
+
}
|
|
68
|
+
return value > 0 && value < 4e3 && toRoman(value) === upper ? value : void 0;
|
|
69
|
+
}
|
|
70
|
+
/** The canonical roman numeral for a value, which is how a valid one reads. */
|
|
71
|
+
function toRoman(value) {
|
|
72
|
+
const steps = [
|
|
73
|
+
[1e3, "M"],
|
|
74
|
+
[900, "CM"],
|
|
75
|
+
[500, "D"],
|
|
76
|
+
[400, "CD"],
|
|
77
|
+
[100, "C"],
|
|
78
|
+
[90, "XC"],
|
|
79
|
+
[50, "L"],
|
|
80
|
+
[40, "XL"],
|
|
81
|
+
[10, "X"],
|
|
82
|
+
[9, "IX"],
|
|
83
|
+
[5, "V"],
|
|
84
|
+
[4, "IV"],
|
|
85
|
+
[1, "I"]
|
|
86
|
+
];
|
|
87
|
+
let out = "";
|
|
88
|
+
let left = value;
|
|
89
|
+
for (const [n, s] of steps) while (left >= n) {
|
|
90
|
+
out += s;
|
|
91
|
+
left -= n;
|
|
92
|
+
}
|
|
93
|
+
return out;
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* The page numbers a document's running band prints, and the sequences they
|
|
97
|
+
* run in.
|
|
98
|
+
*
|
|
99
|
+
* The number is the numeral that CHANGES from page to page: "Page 3 of 10" is
|
|
100
|
+
* counted by its first, "Chapter I — 47" by its last. A sequence goes on while
|
|
101
|
+
* each page's number is one more than the last in the same numerals, and
|
|
102
|
+
* starts again where it does not — i, ii, iii and then 1 are two sequences.
|
|
103
|
+
* A page whose band shows no number is counted on.
|
|
104
|
+
*
|
|
105
|
+
* @param bands Each page's band text, or undefined where the page has none.
|
|
106
|
+
* @returns The numbering, or undefined where the bands show no page number
|
|
107
|
+
* worth reading as one.
|
|
108
|
+
*/
|
|
109
|
+
function pageNumberingOf(bands) {
|
|
110
|
+
const found = bands.map((text) => text === void 0 ? [] : numeralsIn(text));
|
|
111
|
+
const counted = found.filter((n) => n.length > 0);
|
|
112
|
+
if (counted.length < 2) return void 0;
|
|
113
|
+
const width = Math.min(...counted.map((n) => n.length));
|
|
114
|
+
let at = -1;
|
|
115
|
+
for (let k = 0; k < width && at < 0; k++) if (new Set(counted.map((n) => n[k].text)).size > 1) at = k;
|
|
116
|
+
if (at < 0) return void 0;
|
|
117
|
+
const numbers = found.map((n) => n[at]);
|
|
118
|
+
const runs = [];
|
|
119
|
+
numbers.forEach((number, page) => {
|
|
120
|
+
if (number === void 0) return;
|
|
121
|
+
const run = runs[runs.length - 1];
|
|
122
|
+
if (run !== void 0) {
|
|
123
|
+
const expected = run.start + (page - run.from);
|
|
124
|
+
if (number.format === run.format && number.value === expected) return;
|
|
125
|
+
runs.push({
|
|
126
|
+
from: page,
|
|
127
|
+
format: number.format,
|
|
128
|
+
start: number.value
|
|
129
|
+
});
|
|
130
|
+
return;
|
|
131
|
+
}
|
|
132
|
+
const start = number.value - page;
|
|
133
|
+
if (start >= 1) runs.push({
|
|
134
|
+
from: 0,
|
|
135
|
+
format: number.format,
|
|
136
|
+
start
|
|
137
|
+
});
|
|
138
|
+
else runs.push({
|
|
139
|
+
from: 0,
|
|
140
|
+
format: "decimal",
|
|
141
|
+
start: 1
|
|
142
|
+
}, {
|
|
143
|
+
from: page,
|
|
144
|
+
format: number.format,
|
|
145
|
+
start: number.value
|
|
146
|
+
});
|
|
147
|
+
});
|
|
148
|
+
if (runs.length === 0 || runs.length > Math.max(MOST_RUNS, counted.length / LEAST_RUN_PAGES)) return;
|
|
149
|
+
return {
|
|
150
|
+
numbers,
|
|
151
|
+
runs
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
/**
|
|
155
|
+
* The sequence a page is counted in.
|
|
156
|
+
*
|
|
157
|
+
* @param runs The document's sequences, first page first.
|
|
158
|
+
* @param page The page.
|
|
159
|
+
* @returns The run that page belongs to.
|
|
160
|
+
*/
|
|
161
|
+
function runOf(runs, page) {
|
|
162
|
+
let found;
|
|
163
|
+
for (const run of runs) if (run.from <= page) found = run;
|
|
164
|
+
return found;
|
|
165
|
+
}
|
|
166
|
+
//#endregion
|
|
167
|
+
export { pageNumberingOf, runOf };
|
|
@@ -3,13 +3,14 @@ import { ResourceStore } from "../core/ir/resources.js";
|
|
|
3
3
|
import { FEATURES } from "../core/ir/features.js";
|
|
4
4
|
import { collectEmbeddedFonts } from "./embedded-fonts.js";
|
|
5
5
|
import { collectFaceFamilies } from "./font.js";
|
|
6
|
+
import { displayOf, placeRuns, placeVectors, textFrameOf, wordsTurnOf } from "./display.js";
|
|
7
|
+
import { ASCENDER, CARRIER_LINE_PT, buildFlowDoc, dedupeLosses, floatOntoSheet, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, sectionOnSheet, shapeBlock, spaceAfter, withMeasuredMargins } from "./flow-build.js";
|
|
8
|
+
import { faceOutlinesOf, kernedFaces, pageSpacing } from "./face-outlines.js";
|
|
6
9
|
import { extractPageText } from "./text.js";
|
|
7
10
|
import { collectPageVectors } from "./vector.js";
|
|
8
|
-
import { displayOf, placeRuns, placeVectors } from "./display.js";
|
|
9
|
-
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock, spaceAfter, withMeasuredMargins } from "./flow-build.js";
|
|
10
11
|
import { collectPageImages } from "./images.js";
|
|
11
12
|
import { markDrawnRules } from "./text-rules.js";
|
|
12
|
-
import { endedParagraph } from "./layout.js";
|
|
13
|
+
import { FOOTER_PART, HEADER_PART, endedParagraph, footerBand, numberedFrom, numberingOf, pageTextEdges, runningFoot, stepsBetweenWords } from "./layout.js";
|
|
13
14
|
import { readStructTree } from "./struct-tree.js";
|
|
14
15
|
//#region src/pdf-reader/tagged.ts
|
|
15
16
|
var ASSUMED_CONTENT_WIDTH_PT = 468;
|
|
@@ -31,8 +32,15 @@ function reconstructTaggedPdf(file) {
|
|
|
31
32
|
const root = readStructTree(file);
|
|
32
33
|
if (!root) return void 0;
|
|
33
34
|
const pages = file.pages();
|
|
34
|
-
const
|
|
35
|
-
const
|
|
35
|
+
const sheets = pages.map((page) => displayOf(page));
|
|
36
|
+
const painted = /* @__PURE__ */ new Map();
|
|
37
|
+
const extracted = pages.map((page) => extractPageText(file, page, painted));
|
|
38
|
+
const spacing = pageSpacing(extracted);
|
|
39
|
+
const onSheets = extracted.map((runs, i) => placeRuns(runs, sheets[i]));
|
|
40
|
+
const worded = onSheets.filter((runs) => runs.some((r) => r.text.trim() !== ""));
|
|
41
|
+
const turned = worded.length > 0 && worded.every((runs) => wordsTurnOf(runs) === 270);
|
|
42
|
+
const shown = turned ? sheets.map((sheet) => textFrameOf(sheet)) : sheets;
|
|
43
|
+
const placedRuns = turned ? extracted.map((runs, i) => placeRuns(runs, shown[i])) : onSheets;
|
|
36
44
|
const vectorLosses = [];
|
|
37
45
|
const pageVectors = pages.map((page, i) => {
|
|
38
46
|
const lifted = collectPageVectors(file, page, collectPageImages(file, page).images.map((img) => ({
|
|
@@ -106,6 +114,23 @@ function reconstructTaggedPdf(file) {
|
|
|
106
114
|
});
|
|
107
115
|
const collectText = (node) => squash([textOf(node), ...node.children.map(collectText)].join(" "));
|
|
108
116
|
const setting = /* @__PURE__ */ new Map();
|
|
117
|
+
const pageOf = /* @__PURE__ */ new Map();
|
|
118
|
+
/** The highest baseline a page's own paragraphs stand on, page space (y up). */
|
|
119
|
+
const highestOf = (own) => {
|
|
120
|
+
const tops = own.flatMap((el) => {
|
|
121
|
+
const set = setting.get(el);
|
|
122
|
+
return set ? [set.top] : [];
|
|
123
|
+
});
|
|
124
|
+
return tops.length > 0 ? Math.max(...tops) : void 0;
|
|
125
|
+
};
|
|
126
|
+
/** The lowest baseline a page's own paragraphs reach, page space (y up). */
|
|
127
|
+
const lowestOf = (own) => {
|
|
128
|
+
const bottoms = own.flatMap((el) => {
|
|
129
|
+
const set = setting.get(el);
|
|
130
|
+
return set ? [set.bottom] : [];
|
|
131
|
+
});
|
|
132
|
+
return bottoms.length > 0 ? Math.min(...bottoms) : void 0;
|
|
133
|
+
};
|
|
109
134
|
/** The topmost and bottommost baseline under a node, and its largest face. */
|
|
110
135
|
function baselinesOf(node) {
|
|
111
136
|
let top;
|
|
@@ -205,33 +230,48 @@ function reconstructTaggedPdf(file) {
|
|
|
205
230
|
}).filter((l) => l.spans.some((sp) => sp.text.trim().length > 0));
|
|
206
231
|
}
|
|
207
232
|
function emit(node, out) {
|
|
233
|
+
const on = (el) => {
|
|
234
|
+
const page = firstPageOf(node);
|
|
235
|
+
if (page !== void 0) pageOf.set(el, page);
|
|
236
|
+
return el;
|
|
237
|
+
};
|
|
208
238
|
if (node.type === "Table") {
|
|
209
239
|
const table = buildTable(node);
|
|
210
|
-
if (table) out.push(table);
|
|
240
|
+
if (table) out.push(on(table));
|
|
211
241
|
return;
|
|
212
242
|
}
|
|
213
243
|
if (node.type === "Figure") {
|
|
214
244
|
for (const img of imagesForNode(node)) {
|
|
215
245
|
emitted.add(img);
|
|
216
|
-
out.push(imageBlock(img, resources, node.alt));
|
|
246
|
+
out.push(on(imageBlock(img, resources, node.alt)));
|
|
217
247
|
}
|
|
218
248
|
return;
|
|
219
249
|
}
|
|
220
250
|
if (node.type === "LI") {
|
|
221
251
|
const text = collectText(node);
|
|
222
|
-
if (text.length > 0) out.push(paragraphBlock(text, void 0));
|
|
252
|
+
if (text.length > 0) out.push(on(paragraphBlock(text, void 0)));
|
|
223
253
|
return;
|
|
224
254
|
}
|
|
225
255
|
if (node.children.length === 0) {
|
|
226
256
|
if (textOf(node).length > 0) for (const part of settingsOf(node)) {
|
|
227
257
|
const el = paragraphFromRuns(part.spans, headingLevel(node.type));
|
|
228
258
|
setting.set(el, part.set);
|
|
229
|
-
out.push(el);
|
|
259
|
+
out.push(on(el));
|
|
230
260
|
}
|
|
231
261
|
return;
|
|
232
262
|
}
|
|
233
263
|
for (const child of node.children) emit(child, out);
|
|
234
264
|
}
|
|
265
|
+
/** The first page any marked content under a node stands on. */
|
|
266
|
+
function firstPageOf(node) {
|
|
267
|
+
let first;
|
|
268
|
+
const visit = (n) => {
|
|
269
|
+
for (const { page } of n.mcids) if (first === void 0 || page < first) first = page;
|
|
270
|
+
for (const child of n.children) visit(child);
|
|
271
|
+
};
|
|
272
|
+
visit(node);
|
|
273
|
+
return first;
|
|
274
|
+
}
|
|
235
275
|
function buildTable(tableNode) {
|
|
236
276
|
const raw = [];
|
|
237
277
|
const collectRows = (n) => {
|
|
@@ -295,9 +335,14 @@ function reconstructTaggedPdf(file) {
|
|
|
295
335
|
cells
|
|
296
336
|
};
|
|
297
337
|
}
|
|
298
|
-
const
|
|
299
|
-
emit(root,
|
|
300
|
-
|
|
338
|
+
const named = [];
|
|
339
|
+
emit(root, named);
|
|
340
|
+
const byPage = pages.map(() => []);
|
|
341
|
+
let current = 0;
|
|
342
|
+
for (const el of named) {
|
|
343
|
+
current = Math.max(current, pageOf.get(el) ?? current);
|
|
344
|
+
byPage[current].push(el);
|
|
345
|
+
}
|
|
301
346
|
let zOrder = -1e6;
|
|
302
347
|
imageLosses.push(...vectorLosses);
|
|
303
348
|
pages.forEach((_page, index) => {
|
|
@@ -306,31 +351,104 @@ function reconstructTaggedPdf(file) {
|
|
|
306
351
|
top: shown[index].height
|
|
307
352
|
};
|
|
308
353
|
const taken = ruled[index]?.consumed;
|
|
354
|
+
const drawn = [];
|
|
309
355
|
for (const v of pageVectors[index] ?? []) {
|
|
310
356
|
if (taken?.has(v) === true) continue;
|
|
311
|
-
|
|
357
|
+
const shape = shapeBlock(v, frame, zOrder++, true);
|
|
358
|
+
drawn.push(turned ? floatOntoSheet(shape, shown[index].height) : shape);
|
|
312
359
|
}
|
|
360
|
+
byPage[index].unshift(...drawn);
|
|
313
361
|
});
|
|
314
|
-
const orphans = [];
|
|
315
362
|
pageImages.forEach((p, page) => {
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
363
|
+
const left = p.images.filter((img) => !emitted.has(img)).sort((a, b) => b.y - a.y);
|
|
364
|
+
byPage[page].push(...left.map((img) => imageBlock(img, resources)));
|
|
365
|
+
});
|
|
366
|
+
if (byPage.every((own) => own.length === 0)) return void 0;
|
|
367
|
+
const loose = placedRuns.map((runs, i) => shown[i].sheet ? [] : runs.filter((r) => !claimed.has(r)));
|
|
368
|
+
const foot = runningFoot(loose, shown, "foot");
|
|
369
|
+
const head = runningFoot(loose, shown, "head");
|
|
370
|
+
const numberedBand = foot?.numbered === true ? foot : head?.numbered === true ? head : void 0;
|
|
371
|
+
const numbering = numberingOf(numberedBand);
|
|
372
|
+
const lifted = (r, i) => foot?.lift[i]?.has(r) === true || head?.lift[i]?.has(r) === true;
|
|
373
|
+
const measured = withMeasuredMargins(sectionFromPdfPages(pages, shown[0]), shown, placedRuns.map((runs, i) => runs.filter((r) => !lifted(r, i))), pageImages.map((p) => p.images), foot?.band);
|
|
374
|
+
const body = [];
|
|
375
|
+
const top = measured?.margins?.top ?? 0;
|
|
376
|
+
const bottom = measured?.margins?.bottom ?? 0;
|
|
377
|
+
const ends = byPage.map((own) => lowestOf(own));
|
|
378
|
+
const tops = byPage.map((own) => highestOf(own));
|
|
379
|
+
const sectionsFrom = (numbering?.runs ?? []).map((run) => run.from).filter((from) => from > 0);
|
|
380
|
+
const endsAt = [];
|
|
381
|
+
let sectionFrom = 0;
|
|
382
|
+
byPage.forEach((own, index) => {
|
|
383
|
+
if (sectionsFrom.includes(index)) {
|
|
384
|
+
endsAt.push({
|
|
385
|
+
at: body.length,
|
|
386
|
+
from: sectionFrom
|
|
387
|
+
});
|
|
388
|
+
sectionFrom = index;
|
|
389
|
+
}
|
|
390
|
+
const height = shown[index]?.height ?? 0;
|
|
391
|
+
spaceParagraphs(own, setting, height);
|
|
392
|
+
const ended = ends[index - 1];
|
|
393
|
+
const opens = index > 0 && (own.length === 0 || ended === void 0 || ended - bottom > (height - top - bottom) * SHORT_PAGE_SHARE);
|
|
394
|
+
const lead = own.findIndex((el) => !floats(el));
|
|
395
|
+
const set = lead >= 0 ? setting.get(own[lead]) : void 0;
|
|
396
|
+
const fresh = index === 0 || opens || sectionsFrom.includes(index);
|
|
397
|
+
const highest = tops[index];
|
|
398
|
+
const first = set !== void 0 && highest !== void 0 && set.top >= highest - set.size;
|
|
399
|
+
if (fresh && first && measured?.margins) {
|
|
400
|
+
const gap = height - top - (set.top + set.size * ASCENDER);
|
|
401
|
+
if (gap > 1) own[lead] = spacedBefore(own[lead], gap);
|
|
402
|
+
}
|
|
403
|
+
if (opens && !sectionsFrom.includes(index)) body.push(pageBreak(true));
|
|
404
|
+
else if (index === 0 && own.length === 0 && pages.length > 1) body.push(pageBreak(false));
|
|
405
|
+
body.push(...own);
|
|
320
406
|
});
|
|
321
|
-
orphans.sort((a, b) => a.page - b.page || b.img.y - a.img.y);
|
|
322
|
-
for (const { img } of orphans) body.push(imageBlock(img, resources));
|
|
323
|
-
if (body.length === 0) return void 0;
|
|
324
407
|
if (placedRuns.some((page) => page.some((r) => r.text.includes("�")))) imageLosses.push({
|
|
325
408
|
severity: "dropped",
|
|
326
409
|
feature: FEATURES.text,
|
|
327
410
|
detail: "some glyphs map to no character — the font states no /ToUnicode and its program says nothing either, so that text is unrecoverable"
|
|
328
411
|
});
|
|
329
|
-
const onPage = placedRuns.
|
|
412
|
+
const onPage = placedRuns.flatMap((runs, i) => runs.filter((r) => !lifted(r, i))).reduce((n, r) => n + r.text.length, 0);
|
|
330
413
|
const reached = [...claimed].reduce((n, r) => n + r.text.length, 0);
|
|
331
414
|
if (onPage > 0 && reached * 2 < onPage) return void 0;
|
|
415
|
+
endsAt.push({
|
|
416
|
+
at: body.length,
|
|
417
|
+
from: sectionFrom
|
|
418
|
+
});
|
|
419
|
+
const stepped0 = stepsBetweenWords(placedRuns[0] ?? []);
|
|
420
|
+
const edges0 = pageTextEdges(placedRuns[0] ?? []);
|
|
421
|
+
const numeralOf = (of) => {
|
|
422
|
+
if (of === void 0 || of !== numberedBand || numbering === void 0) return void 0;
|
|
423
|
+
const first = of.lift.findIndex((set) => set.size > 0);
|
|
424
|
+
return first >= 0 ? numbering.numbers[first]?.text : void 0;
|
|
425
|
+
};
|
|
426
|
+
const footBand = foot ? footerBand(foot.band, stepped0, edges0, foot.numbered, numeralOf(foot)) : [];
|
|
427
|
+
const headBand = head ? footerBand(head.band, stepped0, edges0, head.numbered, numeralOf(head)) : [];
|
|
428
|
+
const sectionAt = (from) => {
|
|
429
|
+
const base = sectionOnSheet(measured, shown[0]);
|
|
430
|
+
return base ? {
|
|
431
|
+
...base,
|
|
432
|
+
...numberedFrom(numbering, from),
|
|
433
|
+
...footBand.length > 0 ? { footers: [{
|
|
434
|
+
type: "default",
|
|
435
|
+
relationshipId: FOOTER_PART
|
|
436
|
+
}] } : {},
|
|
437
|
+
...headBand.length > 0 ? { headers: [{
|
|
438
|
+
type: "default",
|
|
439
|
+
relationshipId: HEADER_PART
|
|
440
|
+
}] } : {}
|
|
441
|
+
} : base;
|
|
442
|
+
};
|
|
443
|
+
const sections = endsAt.length > 1 ? endsAt.flatMap((end) => {
|
|
444
|
+
const properties = sectionAt(end.from);
|
|
445
|
+
return properties ? [{
|
|
446
|
+
properties,
|
|
447
|
+
endIndex: end.at
|
|
448
|
+
}] : [];
|
|
449
|
+
}) : [];
|
|
332
450
|
return {
|
|
333
|
-
doc: buildFlowDoc(body, resources,
|
|
451
|
+
doc: buildFlowDoc(body, resources, sectionAt(0), collectEmbeddedFonts(file, pages, imageLosses), sections, footBand.length > 0 || headBand.length > 0 ? new Map([...footBand.length > 0 ? [[FOOTER_PART, footBand]] : [], ...headBand.length > 0 ? [[HEADER_PART, headBand]] : []]) : void 0, collectFaceFamilies(file, pages), faceOutlinesOf(painted, spacing), kernedFaces(spacing)),
|
|
334
452
|
losses: imageLosses
|
|
335
453
|
};
|
|
336
454
|
}
|
|
@@ -437,6 +555,47 @@ var MCID_SPACE_EM = .15;
|
|
|
437
555
|
function squash(text) {
|
|
438
556
|
return text.replace(/\s+/g, " ").trim();
|
|
439
557
|
}
|
|
558
|
+
/**
|
|
559
|
+
* How much of a page's text area the page may leave empty below its last line
|
|
560
|
+
* and still be FULL: a page that ends higher than this ended on purpose.
|
|
561
|
+
*/
|
|
562
|
+
var SHORT_PAGE_SHARE = .25;
|
|
563
|
+
/** Whether an element FLOATS — a drawing anchored to its page, which takes no room. */
|
|
564
|
+
function floats(el) {
|
|
565
|
+
return el.kind === "image" && el.image.float !== void 0 || el.kind === "shape" && el.shape.float !== void 0;
|
|
566
|
+
}
|
|
567
|
+
/** A paragraph given the space the page left above it; anything else as it is. */
|
|
568
|
+
function spacedBefore(el, before) {
|
|
569
|
+
if (el.kind !== "paragraph") return el;
|
|
570
|
+
const properties = {
|
|
571
|
+
...el.paragraph.properties,
|
|
572
|
+
spacingBefore: pt(before)
|
|
573
|
+
};
|
|
574
|
+
return {
|
|
575
|
+
...el,
|
|
576
|
+
paragraph: {
|
|
577
|
+
...el.paragraph,
|
|
578
|
+
properties
|
|
579
|
+
}
|
|
580
|
+
};
|
|
581
|
+
}
|
|
582
|
+
/**
|
|
583
|
+
* An empty paragraph that takes no room: the carrier of a page break, or of
|
|
584
|
+
* nothing at all on a blank first sheet the next page's break has to follow.
|
|
585
|
+
*/
|
|
586
|
+
function pageBreak(breaks) {
|
|
587
|
+
return {
|
|
588
|
+
kind: "paragraph",
|
|
589
|
+
paragraph: {
|
|
590
|
+
properties: {
|
|
591
|
+
...breaks ? { pageBreakBefore: true } : {},
|
|
592
|
+
spacingLine: CARRIER_LINE_PT,
|
|
593
|
+
spacingLineRule: "exact"
|
|
594
|
+
},
|
|
595
|
+
runs: []
|
|
596
|
+
}
|
|
597
|
+
};
|
|
598
|
+
}
|
|
440
599
|
function headingLevel(type) {
|
|
441
600
|
const m = /^H([1-6])$/.exec(type);
|
|
442
601
|
return m ? Number(m[1]) - 1 : void 0;
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { ContentFont, TextRun } from './content.js';
|
|
2
|
+
import { ShownCodes } from './face-outlines.js';
|
|
2
3
|
import { PdfDict } from '../pdf/objects.js';
|
|
3
4
|
import { PdfFile, PdfPage } from './document.js';
|
|
4
5
|
/**
|
|
@@ -8,11 +9,13 @@ import { PdfFile, PdfPage } from './document.js';
|
|
|
8
9
|
* origin falls inside a `/Link` annotation's `/Rect` with that link's URI (EP8)
|
|
9
10
|
* so hyperlinks survive.
|
|
10
11
|
*
|
|
11
|
-
* @param file
|
|
12
|
-
* @param page
|
|
12
|
+
* @param file The owning {@link PdfFile}.
|
|
13
|
+
* @param page The page to extract.
|
|
14
|
+
* @param painted Where to gather the codes each font painted, for a caller
|
|
15
|
+
* that embeds the faces (see `./face-outlines`).
|
|
13
16
|
* @returns The page's runs, each carrying an `href` when it sits under a link.
|
|
14
17
|
*/
|
|
15
|
-
export declare function extractPageText(file: PdfFile, page: PdfPage): Array<TextRun>;
|
|
18
|
+
export declare function extractPageText(file: PdfFile, page: PdfPage, painted?: ShownCodes): Array<TextRun>;
|
|
16
19
|
/**
|
|
17
20
|
* The `/Font` resources of one dictionary, built into interpreter fonts. Shared
|
|
18
21
|
* with the path and picture walks, which need them for one thing only: a Type 3
|