extract-pdf 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -10
- package/dist/models/annotation.d.ts +20 -0
- package/dist/models/block-type.d.ts +10 -0
- package/dist/models/headline-finder.d.ts +11 -0
- package/dist/models/line-converter.d.ts +10 -0
- package/dist/models/line-item-block.d.ts +14 -0
- package/dist/models/line-item.d.ts +25 -0
- package/dist/models/metadata.d.ts +23 -0
- package/dist/models/page-item.d.ts +21 -0
- package/dist/models/page.d.ts +14 -0
- package/dist/models/parse-result.d.ts +24 -0
- package/dist/models/parsed-elements.d.ts +20 -0
- package/dist/models/stashing-stream.d.ts +23 -0
- package/dist/models/text-item-line-grouper.d.ts +8 -0
- package/dist/models/text-item.d.ts +29 -0
- package/dist/models/word.d.ts +27 -0
- package/dist/pdf-to-html.cjs.js +1 -1
- package/dist/pdf-to-html.d.ts +36 -41
- package/dist/pdf-to-html.es.js +1 -1
- package/dist/transforms/base/to-line-item-block-transform.d.ts +6 -0
- package/dist/transforms/base/to-line-item-transform.d.ts +6 -0
- package/dist/transforms/base/to-text-item-transform.d.ts +6 -0
- package/dist/transforms/base/transformation.d.ts +8 -0
- package/dist/transforms/block/detect-code-quote-blocks.d.ts +6 -0
- package/dist/transforms/block/detect-list-levels.d.ts +6 -0
- package/dist/transforms/block/gather-blocks.d.ts +6 -0
- package/dist/transforms/calculate-global-stats.d.ts +11 -0
- package/dist/transforms/line-item/compact-lines.d.ts +6 -0
- package/dist/transforms/line-item/detect-headers.d.ts +6 -0
- package/dist/transforms/line-item/detect-list-items.d.ts +6 -0
- package/dist/transforms/line-item/detect-toc.d.ts +6 -0
- package/dist/transforms/line-item/remove-repetitive-elements.d.ts +6 -0
- package/dist/transforms/line-item/vertical-to-horizontal.d.ts +6 -0
- package/dist/transforms/to-html.d.ts +6 -0
- package/dist/transforms/to-text-blocks.d.ts +6 -0
- package/dist/utils/is-url-pdf.d.ts +1 -0
- package/dist/utils/page-item-functions.d.ts +8 -0
- package/dist/utils/page-number-functions.d.ts +14 -0
- package/dist/utils/string-functions.d.ts +14 -0
- package/package.json +8 -11
- package/src/models/annotation.ts +41 -0
- package/src/models/block-type.ts +203 -0
- package/src/models/headline-finder.ts +53 -0
- package/src/models/line-converter.ts +224 -0
- package/src/models/line-item-block.ts +51 -0
- package/src/models/line-item.ts +59 -0
- package/src/models/metadata.ts +29 -0
- package/src/models/page-item.ts +36 -0
- package/src/models/page.ts +16 -0
- package/src/models/parse-result.ts +32 -0
- package/src/models/parsed-elements.ts +29 -0
- package/src/models/stashing-stream.ts +86 -0
- package/src/models/text-item-line-grouper.ts +41 -0
- package/src/models/text-item.ts +50 -0
- package/src/models/word.ts +31 -0
- package/src/pdf-to-html.ts +225 -0
- package/src/transforms/base/to-line-item-block-transform.ts +29 -0
- package/src/transforms/base/to-line-item-transform.ts +29 -0
- package/src/transforms/base/to-text-item-transform.ts +28 -0
- package/src/transforms/base/transformation.ts +35 -0
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -0
- package/src/transforms/block/detect-list-levels.ts +64 -0
- package/src/transforms/block/gather-blocks.ts +113 -0
- package/src/transforms/calculate-global-stats.ts +132 -0
- package/src/transforms/line-item/compact-lines.ts +92 -0
- package/src/transforms/line-item/detect-headers.ts +173 -0
- package/src/transforms/line-item/detect-list-items.ts +68 -0
- package/src/transforms/line-item/detect-toc.ts +459 -0
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -0
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -0
- package/src/transforms/to-html.ts +46 -0
- package/src/transforms/to-text-blocks.ts +38 -0
- package/src/utils/is-url-pdf.ts +33 -0
- package/src/utils/page-item-functions.ts +35 -0
- package/src/utils/page-number-functions.ts +109 -0
- package/src/utils/string-functions.ts +124 -0
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description A single raw text span extracted from a PDF page, carrying spatial
|
|
3
|
+
* coordinates (x, y, width, height), the text string, the font name, and optional
|
|
4
|
+
* inline format markers (lineFormat, unopenedFormat, unclosedFormat) used by the
|
|
5
|
+
* line-compaction pipeline to track bold/italic spans that cross item boundaries.
|
|
6
|
+
*/
|
|
7
|
+
import PageItem, { BlockTypeEntry } from "./page-item";
|
|
8
|
+
import Annotation from "./annotation";
|
|
9
|
+
import ParsedElements from "./parsed-elements";
|
|
10
|
+
import { WordFormatEntry } from "./word";
|
|
11
|
+
|
|
12
|
+
export default class TextItem extends PageItem {
|
|
13
|
+
x: number;
|
|
14
|
+
y: number;
|
|
15
|
+
width: number;
|
|
16
|
+
height: number;
|
|
17
|
+
text: string;
|
|
18
|
+
font: string;
|
|
19
|
+
lineFormat: WordFormatEntry | null;
|
|
20
|
+
unopenedFormat: WordFormatEntry | null;
|
|
21
|
+
unclosedFormat: WordFormatEntry | null;
|
|
22
|
+
|
|
23
|
+
constructor(options: {
|
|
24
|
+
x: number;
|
|
25
|
+
y: number;
|
|
26
|
+
width: number;
|
|
27
|
+
height: number;
|
|
28
|
+
text: string;
|
|
29
|
+
font: string;
|
|
30
|
+
lineFormat?: WordFormatEntry | null;
|
|
31
|
+
unopenedFormat?: WordFormatEntry | null;
|
|
32
|
+
unclosedFormat?: WordFormatEntry | null;
|
|
33
|
+
type?: BlockTypeEntry | null;
|
|
34
|
+
annotation?: Annotation | null;
|
|
35
|
+
parsedElements?: ParsedElements | null;
|
|
36
|
+
}) {
|
|
37
|
+
super(options);
|
|
38
|
+
this.x = options.x;
|
|
39
|
+
this.y = options.y;
|
|
40
|
+
this.width = options.width;
|
|
41
|
+
this.height = options.height;
|
|
42
|
+
this.text = options.text;
|
|
43
|
+
this.font = options.font;
|
|
44
|
+
|
|
45
|
+
this.lineFormat = options.lineFormat ?? null;
|
|
46
|
+
this.unopenedFormat = options.unopenedFormat ?? null;
|
|
47
|
+
this.unclosedFormat = options.unclosedFormat ?? null;
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Minimal token model for a single word within a `LineItem`.
|
|
3
|
+
* Carries the text `string`, an optional `type` (`WordType`: LINK,
|
|
4
|
+
* FOOTNOTE_LINK, FOOTNOTE) for semantic rendering, and an optional `format`
|
|
5
|
+
* (`WordFormat`: BOLD, OBLIQUE, BOLD_OBLIQUE) for inline HTML styling.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
export interface WordFormatEntry {
|
|
9
|
+
name: string;
|
|
10
|
+
startSymbol: string;
|
|
11
|
+
endSymbol: string;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export interface WordTypeEntry {
|
|
15
|
+
name: string;
|
|
16
|
+
attachWithoutWhitespace?: boolean;
|
|
17
|
+
plainTextFormat?: boolean;
|
|
18
|
+
toText(string: string): string;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export default class Word {
|
|
22
|
+
string: string;
|
|
23
|
+
type: WordTypeEntry | null;
|
|
24
|
+
format: WordFormatEntry | null;
|
|
25
|
+
|
|
26
|
+
constructor(options: { string: string; type?: WordTypeEntry | null; format?: WordFormatEntry | null }) {
|
|
27
|
+
this.string = options.string;
|
|
28
|
+
this.type = options.type ?? null;
|
|
29
|
+
this.format = options.format ?? null;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview High-fidelity PDF-to-HTML conversion pipeline.
|
|
3
|
+
* Extracts structural elements (headers, lists, code blocks) and handles page-level metadata.
|
|
4
|
+
*/
|
|
5
|
+
import {
|
|
6
|
+
findPageNumbers,
|
|
7
|
+
findFirstPage,
|
|
8
|
+
removePageNumber,
|
|
9
|
+
} from "./utils/page-number-functions";
|
|
10
|
+
import TextItem from "./models/text-item";
|
|
11
|
+
import Page from "./models/page";
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Fetch wrapper for grabbing binary content
|
|
15
|
+
*/
|
|
16
|
+
async function grab(url: string, options: { responseType?: string; timeout?: number } = {}) {
|
|
17
|
+
const timeout = options.timeout ? options.timeout * 1000 : 10000;
|
|
18
|
+
const controller = new AbortController();
|
|
19
|
+
const timeoutId = setTimeout(() => controller.abort(), timeout);
|
|
20
|
+
|
|
21
|
+
try {
|
|
22
|
+
const response = await fetch(url, {
|
|
23
|
+
signal: controller.signal,
|
|
24
|
+
});
|
|
25
|
+
clearTimeout(timeoutId);
|
|
26
|
+
|
|
27
|
+
if (!response.ok) {
|
|
28
|
+
throw new Error(`HTTP ${response.status}`);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
if (options.responseType === "arraybuffer") {
|
|
32
|
+
return await response.arrayBuffer();
|
|
33
|
+
}
|
|
34
|
+
return await response.text();
|
|
35
|
+
} catch (error) {
|
|
36
|
+
clearTimeout(timeoutId);
|
|
37
|
+
throw error;
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
import CalculateGlobalStats from "./transforms/calculate-global-stats";
|
|
42
|
+
import CompactLines from "./transforms/line-item/compact-lines";
|
|
43
|
+
import RemoveRepetitiveElements from "./transforms/line-item/remove-repetitive-elements";
|
|
44
|
+
import VerticalToHorizontal from "./transforms/line-item/vertical-to-horizontal";
|
|
45
|
+
import DetectTOC from "./transforms/line-item/detect-toc";
|
|
46
|
+
import DetectListItems from "./transforms/line-item/detect-list-items";
|
|
47
|
+
import DetectHeaders from "./transforms/line-item/detect-headers";
|
|
48
|
+
|
|
49
|
+
import GatherBlocks from "./transforms/block/gather-blocks";
|
|
50
|
+
import DetectCodeQuoteBlocks from "./transforms/block/detect-code-quote-blocks";
|
|
51
|
+
import DetectListLevels from "./transforms/block/detect-list-levels";
|
|
52
|
+
import ToTextBlocks from "./transforms/to-text-blocks";
|
|
53
|
+
import ToHTML from "./transforms/to-html";
|
|
54
|
+
import ParseResult from "./models/parse-result";
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Extracts formatted text from PDF with parsing of linebreaks ,
|
|
58
|
+
* page headers, footnotes, and section headings. Supports fonts, links, bold,
|
|
59
|
+
* italics, lists, headings, headers, footnotes, and Table of Contents,
|
|
60
|
+
* Quotes, and Code Blocks, . Removes repeated headers, links footnote anchors to the footnote,
|
|
61
|
+
* and preserves number of the PDF page with invisible I element.
|
|
62
|
+
*
|
|
63
|
+
* This function uses [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless)
|
|
64
|
+
* to work in more environments than PDF.js-based tools:
|
|
65
|
+
* Cloudflare workers, serverless, node.js, and front-end only.
|
|
66
|
+
* @param {string} pdfURLOrBuffer - URL to a PDF file or buffer from fs.readFile
|
|
67
|
+
* @param {Object} [options]
|
|
68
|
+
* @param {boolean} options.addPageNumbers default=false - Adds # to end of each page
|
|
69
|
+
* @param {boolean} options.removePageHeaders default=true - Removes repeated headers found on each page
|
|
70
|
+
* @returns {string|Object} HTML formatted text
|
|
71
|
+
* @category Extract
|
|
72
|
+
* @author [vtempest (2025)](https://github.com/vtempest),
|
|
73
|
+
* [pdf-to-markdown (2017)](https://github.com/jzillmann/pdf-to-markdown/tree/master),
|
|
74
|
+
* [pdf.js (2012-)](https://github.com/mozilla/pdf.js/releases),
|
|
75
|
+
*/
|
|
76
|
+
export async function convertPDFToHTML(
|
|
77
|
+
pdfURLOrBuffer: any,
|
|
78
|
+
options: { addPageNumbers?: boolean; addCitation?: boolean } = {},
|
|
79
|
+
) {
|
|
80
|
+
// try {
|
|
81
|
+
var { addPageNumbers = false, addCitation = true } = options;
|
|
82
|
+
|
|
83
|
+
// pass in databuffer or download all pdf data
|
|
84
|
+
// and convert to array buffer
|
|
85
|
+
var buffer =
|
|
86
|
+
typeof pdfURLOrBuffer === "string"
|
|
87
|
+
? await grab(pdfURLOrBuffer, {
|
|
88
|
+
responseType: "arraybuffer",
|
|
89
|
+
timeout: 10,
|
|
90
|
+
})
|
|
91
|
+
: pdfURLOrBuffer;
|
|
92
|
+
|
|
93
|
+
let pdfDocument;
|
|
94
|
+
try {
|
|
95
|
+
let { resolvePDFJS } = await import("https://cdn.jsdelivr.net/npm/pdfjs-serverless@1.1.0/+esm" as any);
|
|
96
|
+
|
|
97
|
+
const { getDocument } = await resolvePDFJS();
|
|
98
|
+
pdfDocument = await getDocument({
|
|
99
|
+
data: new Uint8Array(buffer),
|
|
100
|
+
useSystemFonts: true,
|
|
101
|
+
verbosity: 0,
|
|
102
|
+
}).promise;
|
|
103
|
+
} catch (e: any) {
|
|
104
|
+
return { error: e.message };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const pages = [...Array(pdfDocument.numPages).keys()].map(
|
|
108
|
+
(index) => new Page({ index }),
|
|
109
|
+
);
|
|
110
|
+
|
|
111
|
+
let pageIndexNumMap = {};
|
|
112
|
+
let firstPage;
|
|
113
|
+
for (let j = 1; j <= pdfDocument.numPages; j++) {
|
|
114
|
+
const page = await pdfDocument.getPage(j);
|
|
115
|
+
const textContent = await page.getTextContent();
|
|
116
|
+
|
|
117
|
+
if (Object.keys(pageIndexNumMap).length < 10) {
|
|
118
|
+
pageIndexNumMap = findPageNumbers(
|
|
119
|
+
pageIndexNumMap,
|
|
120
|
+
page.pageNumber - 1,
|
|
121
|
+
textContent.items,
|
|
122
|
+
);
|
|
123
|
+
} else {
|
|
124
|
+
firstPage = findFirstPage(pageIndexNumMap);
|
|
125
|
+
break;
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
let pageNum = firstPage ? firstPage.pageNum : 0;
|
|
130
|
+
for (let j = 1; j <= pdfDocument.numPages; j++) {
|
|
131
|
+
const page = await pdfDocument.getPage(j);
|
|
132
|
+
|
|
133
|
+
// Trigger the font retrieval for the page
|
|
134
|
+
await page.getOperatorList();
|
|
135
|
+
|
|
136
|
+
const scale = 1.0;
|
|
137
|
+
const viewport = page.getViewport({ scale });
|
|
138
|
+
let textContent = await page.getTextContent();
|
|
139
|
+
if (firstPage && (page as any).pageIndex >= firstPage.pageIndex) {
|
|
140
|
+
textContent = removePageNumber(textContent as any, pageNum) as any;
|
|
141
|
+
pageNum++;
|
|
142
|
+
}
|
|
143
|
+
const textItems = (textContent.items as any[]).map((item: any) => {
|
|
144
|
+
const tx = [1, 0, 0, 1, 0, 0];
|
|
145
|
+
for (let i = 0; i < 6; i++) {
|
|
146
|
+
tx[i] += item.transform[i] * viewport.transform[i % 2 ? 3 : 0];
|
|
147
|
+
if (i % 2) {
|
|
148
|
+
tx[i + 1] += item.transform[i] * viewport.transform[1];
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
const fontHeight = Math.sqrt(tx[2] * tx[2] + tx[3] * tx[3]);
|
|
153
|
+
const dividedHeight = item.height / fontHeight;
|
|
154
|
+
return new TextItem({
|
|
155
|
+
x: Math.round(item.transform[4]),
|
|
156
|
+
y: Math.round(item.transform[5]),
|
|
157
|
+
width: Math.round(item.width),
|
|
158
|
+
height: Math.round(dividedHeight <= 1 ? item.height : dividedHeight),
|
|
159
|
+
text: item.str,
|
|
160
|
+
font: item.fontName,
|
|
161
|
+
});
|
|
162
|
+
});
|
|
163
|
+
pages[page.pageNumber - 1].items = textItems;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
var parseResult = new ParseResult({ pages });
|
|
167
|
+
|
|
168
|
+
let lastTransformation: (typeof transformations)[number] | undefined,
|
|
169
|
+
transformations = [
|
|
170
|
+
new CalculateGlobalStats(),
|
|
171
|
+
new CompactLines(),
|
|
172
|
+
new RemoveRepetitiveElements(),
|
|
173
|
+
new VerticalToHorizontal(),
|
|
174
|
+
new DetectTOC(),
|
|
175
|
+
new DetectHeaders(),
|
|
176
|
+
new DetectListItems(),
|
|
177
|
+
|
|
178
|
+
new GatherBlocks(),
|
|
179
|
+
new DetectCodeQuoteBlocks(),
|
|
180
|
+
new DetectListLevels(),
|
|
181
|
+
|
|
182
|
+
new ToTextBlocks(),
|
|
183
|
+
new ToHTML(),
|
|
184
|
+
];
|
|
185
|
+
|
|
186
|
+
transformations?.forEach((transformation) => {
|
|
187
|
+
if (lastTransformation) {
|
|
188
|
+
parseResult = lastTransformation.completeTransform(parseResult);
|
|
189
|
+
}
|
|
190
|
+
parseResult = transformation.transform(parseResult);
|
|
191
|
+
lastTransformation = transformation;
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
var html = parseResult.pages.reduce((acc, page, pageNumber) => {
|
|
195
|
+
return (
|
|
196
|
+
acc +
|
|
197
|
+
`<p id="page-${pageNumber + 1}">${
|
|
198
|
+
addPageNumbers ? ` [${pageNumber + 1}] ` : ""
|
|
199
|
+
}${page.items.join('</p><p id="page-' + pageNumber + '">')}</p>`
|
|
200
|
+
);
|
|
201
|
+
}, "");
|
|
202
|
+
|
|
203
|
+
if (addCitation) {
|
|
204
|
+
// Get metadata
|
|
205
|
+
// avoid using date as it is unreliable sand generally file mod date
|
|
206
|
+
var metadata = await pdfDocument.getMetadata();
|
|
207
|
+
var { Author: author, Title: title } = metadata.info as any;
|
|
208
|
+
// date =
|
|
209
|
+
// date.slice(2, 6) + "-" + date.slice(6, 8) + "-" + date.slice(8, 10);
|
|
210
|
+
// date = date ? new Date(date)?.toISOString().split("T")[0] : null;
|
|
211
|
+
|
|
212
|
+
//look for date in first page
|
|
213
|
+
// date = chrono
|
|
214
|
+
// .parseDate(content.slice(0, 400))
|
|
215
|
+
// ?.toISOString()
|
|
216
|
+
// .split("T")[0];
|
|
217
|
+
// // || date;
|
|
218
|
+
|
|
219
|
+
title = html.slice(0, 400).match(/<h[0-9]>(.*?)<\/h[0-9]>/)?.[1] || title;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
return { author, title, html, format: "pdf" };
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Abstract base for pipeline stages that produce `LineItemBlock`-typed
|
|
3
|
+
* page items (GatherBlocks, DetectCodeQuoteBlocks, DetectListLevels). Provides
|
|
4
|
+
* `completeTransform` which strips `REMOVED_ANNOTATION` items and clears all
|
|
5
|
+
* remaining annotations before passing the result to the next stage.
|
|
6
|
+
*/
|
|
7
|
+
import Transformation from './transformation'
|
|
8
|
+
import LineItemBlock from '../../models/line-item-block'
|
|
9
|
+
import ParseResult from '../../models/parse-result'
|
|
10
|
+
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
+
|
|
12
|
+
// Abstract class for transformations producing LineItemBlock(s) to be shown in the LineItemBlockPageView
|
|
13
|
+
export default class ToLineItemBlockTransformation extends Transformation {
|
|
14
|
+
constructor(name: string) {
|
|
15
|
+
super(name, LineItemBlock.name)
|
|
16
|
+
if (this.constructor === ToLineItemBlockTransformation) {
|
|
17
|
+
throw new TypeError('Can not construct abstract class.')
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
completeTransform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
parseResult.messages = []
|
|
23
|
+
parseResult.pages.forEach(page => {
|
|
24
|
+
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
25
|
+
page.items.forEach(item => (item.annotation = null))
|
|
26
|
+
})
|
|
27
|
+
return parseResult
|
|
28
|
+
}
|
|
29
|
+
}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Abstract base for pipeline stages that produce `LineItem`-typed
|
|
3
|
+
* page items (CompactLines, RemoveRepetitiveElements, VerticalToHorizontal,
|
|
4
|
+
* DetectTOC, DetectHeaders, DetectListItems). Provides `completeTransform` which
|
|
5
|
+
* strips `REMOVED_ANNOTATION` items and clears annotations before the next stage.
|
|
6
|
+
*/
|
|
7
|
+
import Transformation from './transformation'
|
|
8
|
+
import LineItem from '../../models/line-item'
|
|
9
|
+
import ParseResult from '../../models/parse-result'
|
|
10
|
+
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
+
|
|
12
|
+
// Abstract class for transformations producing LineItem(s) to be shown in the LineItemPageView
|
|
13
|
+
export default class ToLineItemTransformation extends Transformation {
|
|
14
|
+
constructor(name: string) {
|
|
15
|
+
super(name, LineItem.name)
|
|
16
|
+
if (this.constructor === ToLineItemTransformation) {
|
|
17
|
+
throw new TypeError('Can not construct abstract class.')
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
completeTransform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
parseResult.messages = []
|
|
23
|
+
parseResult.pages.forEach(page => {
|
|
24
|
+
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
25
|
+
page.items.forEach(item => (item.annotation = null))
|
|
26
|
+
})
|
|
27
|
+
return parseResult
|
|
28
|
+
}
|
|
29
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Abstract base for pipeline stages that operate on raw `TextItem`
|
|
3
|
+
* page items (the earliest stages, currently only `CalculateGlobalStats`). Provides
|
|
4
|
+
* the standard `completeTransform` cleanup that removes `REMOVED_ANNOTATION` items
|
|
5
|
+
* and clears annotations before passing control to the next transformation.
|
|
6
|
+
*/
|
|
7
|
+
import Transformation from './transformation'
|
|
8
|
+
import TextItem from '../../models/text-item'
|
|
9
|
+
import ParseResult from '../../models/parse-result'
|
|
10
|
+
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
+
|
|
12
|
+
export default class ToTextItemTransformation extends Transformation {
|
|
13
|
+
constructor(name: string) {
|
|
14
|
+
super(name, TextItem.name)
|
|
15
|
+
if (this.constructor === ToTextItemTransformation) {
|
|
16
|
+
throw new TypeError('Can not construct abstract class.')
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
completeTransform(parseResult: ParseResult): ParseResult {
|
|
21
|
+
parseResult.messages = []
|
|
22
|
+
parseResult.pages.forEach(page => {
|
|
23
|
+
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
24
|
+
page.items.forEach(item => (item.annotation = null))
|
|
25
|
+
})
|
|
26
|
+
return parseResult
|
|
27
|
+
}
|
|
28
|
+
}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Abstract root class for all PDF pipeline stages. Enforces that
|
|
3
|
+
* subclasses implement `transform(parseResult)` and stores `name` (for debug
|
|
4
|
+
* display) and `itemType` (the class name of the items each stage produces,
|
|
5
|
+
* used for view routing). `completeTransform` is a no-op by default; subclasses
|
|
6
|
+
* override it to flush deferred mutations after the stage's debug view is rendered.
|
|
7
|
+
*/
|
|
8
|
+
import ParseResult from "../../models/parse-result";
|
|
9
|
+
|
|
10
|
+
// A transformation from an PdfPage to an PdfPage
|
|
11
|
+
export default class Transformation {
|
|
12
|
+
name: string;
|
|
13
|
+
itemType: string;
|
|
14
|
+
|
|
15
|
+
constructor(name: string, itemType: string) {
|
|
16
|
+
if (this.constructor === Transformation) {
|
|
17
|
+
throw new TypeError("Can not construct abstract class.");
|
|
18
|
+
}
|
|
19
|
+
if (this.transform === Transformation.prototype.transform) {
|
|
20
|
+
throw new TypeError("Please implement abstract method 'transform()'.");
|
|
21
|
+
}
|
|
22
|
+
this.name = name;
|
|
23
|
+
this.itemType = itemType;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
27
|
+
throw new TypeError("Do not call abstract method foo from child.");
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
completeTransform(parseResult: ParseResult): ParseResult {
|
|
31
|
+
parseResult.messages = [];
|
|
32
|
+
return parseResult;
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Block-level transformation that marks blocks as `BlockType.CODE`
|
|
3
|
+
* when all constituent lines are indented beyond the page's leftmost x position
|
|
4
|
+
* and the text height matches body-text height. Heuristically distinguishes
|
|
5
|
+
* indented code/blockquote sections from regular paragraphs without relying on
|
|
6
|
+
* font metadata.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import ToLineItemBlockTransformation from '../base/to-line-item-block-transform'
|
|
10
|
+
import ParseResult from '../../models/parse-result'
|
|
11
|
+
import { DETECTED_ANNOTATION } from '../../models/annotation'
|
|
12
|
+
import BlockType from '../../models/block-type'
|
|
13
|
+
import { minXFromBlocks } from '../../utils/page-item-functions'
|
|
14
|
+
|
|
15
|
+
// Detect items which are code/quote blocks
|
|
16
|
+
export default class DetectCodeQuoteBlocks extends ToLineItemBlockTransformation {
|
|
17
|
+
constructor () {
|
|
18
|
+
super('$1')
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
const mostUsedHeight = parseResult.globals.mostUsedHeight ?? 0
|
|
23
|
+
var foundCodeItems = 0
|
|
24
|
+
parseResult.pages.forEach(page => {
|
|
25
|
+
var minX = minXFromBlocks(page.items)
|
|
26
|
+
page.items.forEach(block => {
|
|
27
|
+
if (!block.type && looksLikeCodeBlock(minX, block.items, mostUsedHeight)) {
|
|
28
|
+
block.annotation = DETECTED_ANNOTATION
|
|
29
|
+
block.type = BlockType.CODE
|
|
30
|
+
foundCodeItems++
|
|
31
|
+
}
|
|
32
|
+
})
|
|
33
|
+
})
|
|
34
|
+
|
|
35
|
+
return new ParseResult({
|
|
36
|
+
...parseResult,
|
|
37
|
+
messages: [
|
|
38
|
+
'Detected ' + foundCodeItems + ' code/quote items.',
|
|
39
|
+
],
|
|
40
|
+
})
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function looksLikeCodeBlock(minX: number, items: any[], mostUsedHeight: number): boolean {
|
|
45
|
+
if (items.length === 0) {
|
|
46
|
+
return false
|
|
47
|
+
}
|
|
48
|
+
if (items.length === 1) {
|
|
49
|
+
return items[0].x > minX && items[0].height <= mostUsedHeight + 1
|
|
50
|
+
}
|
|
51
|
+
for (var item of items) {
|
|
52
|
+
if (item.x === minX) {
|
|
53
|
+
return false
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return true
|
|
57
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Block-level transformation that adds hierarchical indentation to
|
|
3
|
+
* nested list items within `BlockType.LIST` blocks. Tracks nesting level by
|
|
4
|
+
* comparing each item's x position to the previous item's — deeper x means a
|
|
5
|
+
* new sub-level — and prepends spaces proportional to the level so the HTML
|
|
6
|
+
* output reflects the original visual hierarchy.
|
|
7
|
+
*/
|
|
8
|
+
import ToLineItemBlockTransformation from '../base/to-line-item-block-transform'
|
|
9
|
+
import ParseResult from '../../models/parse-result'
|
|
10
|
+
import Word from '../../models/word'
|
|
11
|
+
import { MODIFIED_ANNOTATION, UNCHANGED_ANNOTATION } from '../../models/annotation'
|
|
12
|
+
import BlockType from '../../models/block-type'
|
|
13
|
+
|
|
14
|
+
// Cares for proper sub-item spacing/leveling
|
|
15
|
+
export default class DetectListLevels extends ToLineItemBlockTransformation {
|
|
16
|
+
constructor () {
|
|
17
|
+
super('Level Lists')
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
21
|
+
var listBlocks = 0
|
|
22
|
+
var modifiedBlocks = 0
|
|
23
|
+
parseResult.pages.forEach(page => {
|
|
24
|
+
page.items.filter(block => block.type === BlockType.LIST).forEach(listBlock => {
|
|
25
|
+
var lastItemX: number | undefined
|
|
26
|
+
var currentLevel = 0
|
|
27
|
+
const xByLevel: Record<number, number> = {}
|
|
28
|
+
var modifiedBlock = false
|
|
29
|
+
listBlock.items.forEach(item => {
|
|
30
|
+
const isListItem = true
|
|
31
|
+
if (lastItemX && isListItem) {
|
|
32
|
+
if (item.x > lastItemX) {
|
|
33
|
+
currentLevel++
|
|
34
|
+
xByLevel[item.x] = currentLevel
|
|
35
|
+
} else if (item.x < lastItemX) {
|
|
36
|
+
currentLevel = xByLevel[item.x]
|
|
37
|
+
}
|
|
38
|
+
} else {
|
|
39
|
+
xByLevel[item.x] = 0
|
|
40
|
+
}
|
|
41
|
+
if (currentLevel > 0) {
|
|
42
|
+
item.words = [
|
|
43
|
+
new Word({ string: ' '.repeat(currentLevel * 3) }),
|
|
44
|
+
].concat(item.words)
|
|
45
|
+
modifiedBlock = true
|
|
46
|
+
}
|
|
47
|
+
lastItemX = item.x
|
|
48
|
+
})
|
|
49
|
+
listBlocks++
|
|
50
|
+
if (modifiedBlock) {
|
|
51
|
+
modifiedBlocks++
|
|
52
|
+
listBlock.annotation = MODIFIED_ANNOTATION
|
|
53
|
+
} else {
|
|
54
|
+
listBlock.annotation = UNCHANGED_ANNOTATION
|
|
55
|
+
}
|
|
56
|
+
})
|
|
57
|
+
})
|
|
58
|
+
|
|
59
|
+
return new ParseResult({
|
|
60
|
+
...parseResult,
|
|
61
|
+
messages: ['Modified ' + modifiedBlocks + ' / ' + listBlocks + ' list blocks.'],
|
|
62
|
+
})
|
|
63
|
+
}
|
|
64
|
+
}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Block-level transformation that merges consecutive `LineItem`s into
|
|
3
|
+
* `LineItemBlock`s. Flushes the current block when the item type changes or the
|
|
4
|
+
* vertical gap exceeds `mostUsedDistance`, while respecting per-type merge flags
|
|
5
|
+
* (`mergeToBlock`, `mergeFollowingNonTypedItems`,
|
|
6
|
+
* `mergeFollowingNonTypedItemsWithSmallDistance`) to handle lists and footnotes correctly.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import ToLineItemBlockTransformation from "../base/to-line-item-block-transform";
|
|
10
|
+
import ParseResult from "../../models/parse-result";
|
|
11
|
+
import LineItemBlock from "../../models/line-item-block";
|
|
12
|
+
import { DETECTED_ANNOTATION } from "../../models/annotation";
|
|
13
|
+
import { minXFromPageItems } from "../../utils/page-item-functions";
|
|
14
|
+
|
|
15
|
+
// Gathers lines to blocks
|
|
16
|
+
export default class GatherBlocks extends ToLineItemBlockTransformation {
|
|
17
|
+
constructor() {
|
|
18
|
+
super("Gather Blocks");
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
const { mostUsedDistance } = parseResult.globals;
|
|
23
|
+
var createdBlocks = 0;
|
|
24
|
+
var lineItemCount = 0;
|
|
25
|
+
parseResult.pages.map((page) => {
|
|
26
|
+
lineItemCount += page.items.length;
|
|
27
|
+
const blocks = [];
|
|
28
|
+
var stashedBlock = new LineItemBlock({});
|
|
29
|
+
const flushStashedItems = () => {
|
|
30
|
+
if (stashedBlock.items.length > 1) {
|
|
31
|
+
stashedBlock.annotation = DETECTED_ANNOTATION;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
blocks.push(stashedBlock);
|
|
35
|
+
stashedBlock = new LineItemBlock({});
|
|
36
|
+
createdBlocks++;
|
|
37
|
+
};
|
|
38
|
+
|
|
39
|
+
var minX = minXFromPageItems(page.items);
|
|
40
|
+
page.items.forEach((item) => {
|
|
41
|
+
if (
|
|
42
|
+
stashedBlock.items.length > 0 &&
|
|
43
|
+
shouldFlushBlock(stashedBlock, item, minX, mostUsedDistance)
|
|
44
|
+
) {
|
|
45
|
+
flushStashedItems();
|
|
46
|
+
}
|
|
47
|
+
stashedBlock.addItem(item);
|
|
48
|
+
});
|
|
49
|
+
if (stashedBlock.items.length > 0) {
|
|
50
|
+
flushStashedItems();
|
|
51
|
+
}
|
|
52
|
+
page.items = blocks;
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
return new ParseResult({
|
|
56
|
+
...parseResult,
|
|
57
|
+
messages: [
|
|
58
|
+
"Gathered " +
|
|
59
|
+
createdBlocks +
|
|
60
|
+
" blocks out of " +
|
|
61
|
+
lineItemCount +
|
|
62
|
+
" line items",
|
|
63
|
+
],
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function shouldFlushBlock(stashedBlock: any, item: any, minX: number, mostUsedDistance: number): boolean {
|
|
69
|
+
if (
|
|
70
|
+
stashedBlock.type &&
|
|
71
|
+
stashedBlock.type.mergeFollowingNonTypedItems &&
|
|
72
|
+
!item.type
|
|
73
|
+
) {
|
|
74
|
+
return false;
|
|
75
|
+
}
|
|
76
|
+
const lastItem = stashedBlock.items[stashedBlock.items.length - 1];
|
|
77
|
+
const hasBigDistance = bigDistance(lastItem, item, minX, mostUsedDistance);
|
|
78
|
+
if (
|
|
79
|
+
stashedBlock.type &&
|
|
80
|
+
stashedBlock.type.mergeFollowingNonTypedItemsWithSmallDistance &&
|
|
81
|
+
!item.type &&
|
|
82
|
+
!hasBigDistance
|
|
83
|
+
) {
|
|
84
|
+
return false;
|
|
85
|
+
}
|
|
86
|
+
if (item.type !== stashedBlock.type) {
|
|
87
|
+
return true;
|
|
88
|
+
}
|
|
89
|
+
if (item.type) {
|
|
90
|
+
return !item.type.mergeToBlock;
|
|
91
|
+
} else {
|
|
92
|
+
return hasBigDistance;
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
function bigDistance(lastItem: any, item: any, minX: number, mostUsedDistance: number): boolean {
|
|
97
|
+
const distance = lastItem.y - item.y;
|
|
98
|
+
if (distance < 0 - mostUsedDistance / 2) {
|
|
99
|
+
// distance is negative - and not only a bit
|
|
100
|
+
return true;
|
|
101
|
+
}
|
|
102
|
+
var allowedDisctance = mostUsedDistance + 1;
|
|
103
|
+
if (lastItem.x > minX && item.x > minX) {
|
|
104
|
+
// intended elements like lists often have greater spacing
|
|
105
|
+
allowedDisctance = mostUsedDistance + mostUsedDistance / 2;
|
|
106
|
+
}
|
|
107
|
+
if (distance > allowedDisctance) {
|
|
108
|
+
return true;
|
|
109
|
+
}
|
|
110
|
+
return false;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
|