extract-pdf 0.1.21 → 0.1.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -74
- package/package.json +1 -1
- package/src/models/annotation.ts +41 -41
- package/src/models/block-type.ts +203 -203
- package/src/models/line-converter.ts +224 -224
- package/src/models/metadata.ts +29 -29
- package/src/models/page.ts +16 -16
- package/src/models/parse-result.ts +32 -32
- package/src/models/parsed-elements.ts +29 -29
- package/src/models/stashing-stream.ts +86 -86
- package/src/models/text-item-line-grouper.ts +41 -41
- package/src/models/word.ts +31 -31
- package/src/pdf-to-html.ts +225 -225
- package/src/transforms/base/to-line-item-block-transform.ts +29 -29
- package/src/transforms/base/to-line-item-transform.ts +29 -29
- package/src/transforms/base/to-text-item-transform.ts +28 -28
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
- package/src/transforms/block/detect-list-levels.ts +64 -64
- package/src/transforms/block/gather-blocks.ts +113 -113
- package/src/transforms/calculate-global-stats.ts +132 -132
- package/src/transforms/line-item/compact-lines.ts +92 -92
- package/src/transforms/line-item/detect-headers.ts +173 -173
- package/src/transforms/line-item/detect-list-items.ts +68 -68
- package/src/transforms/line-item/detect-toc.ts +459 -459
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
- package/src/transforms/to-html.ts +46 -46
- package/src/utils/is-url-pdf.ts +33 -33
- package/src/utils/page-item-functions.ts +35 -35
- package/src/utils/string-functions.ts +124 -124
package/src/pdf-to-html.ts
CHANGED
|
@@ -1,225 +1,225 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @fileoverview High-fidelity PDF-to-HTML conversion pipeline.
|
|
3
|
-
* Extracts structural elements (headers, lists, code blocks) and handles page-level metadata.
|
|
4
|
-
*/
|
|
5
|
-
import {
|
|
6
|
-
findPageNumbers,
|
|
7
|
-
findFirstPage,
|
|
8
|
-
removePageNumber,
|
|
9
|
-
} from "./utils/page-number-functions";
|
|
10
|
-
import TextItem from "./models/text-item";
|
|
11
|
-
import Page from "./models/page";
|
|
12
|
-
|
|
13
|
-
/**
|
|
14
|
-
* Fetch wrapper for grabbing binary content
|
|
15
|
-
*/
|
|
16
|
-
async function grab(url: string, options: { responseType?: string; timeout?: number } = {}) {
|
|
17
|
-
const timeout = options.timeout ? options.timeout * 1000 : 10000;
|
|
18
|
-
const controller = new AbortController();
|
|
19
|
-
const timeoutId = setTimeout(() => controller.abort(), timeout);
|
|
20
|
-
|
|
21
|
-
try {
|
|
22
|
-
const response = await fetch(url, {
|
|
23
|
-
signal: controller.signal,
|
|
24
|
-
});
|
|
25
|
-
clearTimeout(timeoutId);
|
|
26
|
-
|
|
27
|
-
if (!response.ok) {
|
|
28
|
-
throw new Error(`HTTP ${response.status}`);
|
|
29
|
-
}
|
|
30
|
-
|
|
31
|
-
if (options.responseType === "arraybuffer") {
|
|
32
|
-
return await response.arrayBuffer();
|
|
33
|
-
}
|
|
34
|
-
return await response.text();
|
|
35
|
-
} catch (error) {
|
|
36
|
-
clearTimeout(timeoutId);
|
|
37
|
-
throw error;
|
|
38
|
-
}
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
import CalculateGlobalStats from "./transforms/calculate-global-stats";
|
|
42
|
-
import CompactLines from "./transforms/line-item/compact-lines";
|
|
43
|
-
import RemoveRepetitiveElements from "./transforms/line-item/remove-repetitive-elements";
|
|
44
|
-
import VerticalToHorizontal from "./transforms/line-item/vertical-to-horizontal";
|
|
45
|
-
import DetectTOC from "./transforms/line-item/detect-toc";
|
|
46
|
-
import DetectListItems from "./transforms/line-item/detect-list-items";
|
|
47
|
-
import DetectHeaders from "./transforms/line-item/detect-headers";
|
|
48
|
-
|
|
49
|
-
import GatherBlocks from "./transforms/block/gather-blocks";
|
|
50
|
-
import DetectCodeQuoteBlocks from "./transforms/block/detect-code-quote-blocks";
|
|
51
|
-
import DetectListLevels from "./transforms/block/detect-list-levels";
|
|
52
|
-
import ToTextBlocks from "./transforms/to-text-blocks";
|
|
53
|
-
import ToHTML from "./transforms/to-html";
|
|
54
|
-
import ParseResult from "./models/parse-result";
|
|
55
|
-
|
|
56
|
-
/**
|
|
57
|
-
* Extracts formatted text from PDF with parsing of linebreaks ,
|
|
58
|
-
* page headers, footnotes, and section headings. Supports fonts, links, bold,
|
|
59
|
-
* italics, lists, headings, headers, footnotes, and Table of Contents,
|
|
60
|
-
* Quotes, and Code Blocks, . Removes repeated headers, links footnote anchors to the footnote,
|
|
61
|
-
* and preserves number of the PDF page with invisible I element.
|
|
62
|
-
*
|
|
63
|
-
* This function uses [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless)
|
|
64
|
-
* to work in more environments than PDF.js-based tools:
|
|
65
|
-
* Cloudflare workers, serverless, node.js, and front-end only.
|
|
66
|
-
* @param {string} pdfURLOrBuffer - URL to a PDF file or buffer from fs.readFile
|
|
67
|
-
* @param {Object} [options]
|
|
68
|
-
* @param {boolean} options.addPageNumbers default=false - Adds # to end of each page
|
|
69
|
-
* @param {boolean} options.removePageHeaders default=true - Removes repeated headers found on each page
|
|
70
|
-
* @returns {string|Object} HTML formatted text
|
|
71
|
-
* @category Extract
|
|
72
|
-
* @author [vtempest (2025)](https://github.com/vtempest),
|
|
73
|
-
* [pdf-to-markdown (2017)](https://github.com/jzillmann/pdf-to-markdown/tree/master),
|
|
74
|
-
* [pdf.js (2012-)](https://github.com/mozilla/pdf.js/releases),
|
|
75
|
-
*/
|
|
76
|
-
export async function convertPDFToHTML(
|
|
77
|
-
pdfURLOrBuffer: any,
|
|
78
|
-
options: { addPageNumbers?: boolean; addCitation?: boolean } = {},
|
|
79
|
-
) {
|
|
80
|
-
// try {
|
|
81
|
-
var { addPageNumbers = false, addCitation = true } = options;
|
|
82
|
-
|
|
83
|
-
// pass in databuffer or download all pdf data
|
|
84
|
-
// and convert to array buffer
|
|
85
|
-
var buffer =
|
|
86
|
-
typeof pdfURLOrBuffer === "string"
|
|
87
|
-
? await grab(pdfURLOrBuffer, {
|
|
88
|
-
responseType: "arraybuffer",
|
|
89
|
-
timeout: 10,
|
|
90
|
-
})
|
|
91
|
-
: pdfURLOrBuffer;
|
|
92
|
-
|
|
93
|
-
let pdfDocument;
|
|
94
|
-
try {
|
|
95
|
-
let { resolvePDFJS } = await import("https://cdn.jsdelivr.net/npm/pdfjs-serverless@1.1.0/+esm" as any);
|
|
96
|
-
|
|
97
|
-
const { getDocument } = await resolvePDFJS();
|
|
98
|
-
pdfDocument = await getDocument({
|
|
99
|
-
data: new Uint8Array(buffer),
|
|
100
|
-
useSystemFonts: true,
|
|
101
|
-
verbosity: 0,
|
|
102
|
-
}).promise;
|
|
103
|
-
} catch (e: any) {
|
|
104
|
-
return { error: e.message };
|
|
105
|
-
}
|
|
106
|
-
|
|
107
|
-
const pages = [...Array(pdfDocument.numPages).keys()].map(
|
|
108
|
-
(index) => new Page({ index }),
|
|
109
|
-
);
|
|
110
|
-
|
|
111
|
-
let pageIndexNumMap = {};
|
|
112
|
-
let firstPage;
|
|
113
|
-
for (let j = 1; j <= pdfDocument.numPages; j++) {
|
|
114
|
-
const page = await pdfDocument.getPage(j);
|
|
115
|
-
const textContent = await page.getTextContent();
|
|
116
|
-
|
|
117
|
-
if (Object.keys(pageIndexNumMap).length < 10) {
|
|
118
|
-
pageIndexNumMap = findPageNumbers(
|
|
119
|
-
pageIndexNumMap,
|
|
120
|
-
page.pageNumber - 1,
|
|
121
|
-
textContent.items,
|
|
122
|
-
);
|
|
123
|
-
} else {
|
|
124
|
-
firstPage = findFirstPage(pageIndexNumMap);
|
|
125
|
-
break;
|
|
126
|
-
}
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
let pageNum = firstPage ? firstPage.pageNum : 0;
|
|
130
|
-
for (let j = 1; j <= pdfDocument.numPages; j++) {
|
|
131
|
-
const page = await pdfDocument.getPage(j);
|
|
132
|
-
|
|
133
|
-
// Trigger the font retrieval for the page
|
|
134
|
-
await page.getOperatorList();
|
|
135
|
-
|
|
136
|
-
const scale = 1.0;
|
|
137
|
-
const viewport = page.getViewport({ scale });
|
|
138
|
-
let textContent = await page.getTextContent();
|
|
139
|
-
if (firstPage && (page as any).pageIndex >= firstPage.pageIndex) {
|
|
140
|
-
textContent = removePageNumber(textContent as any, pageNum) as any;
|
|
141
|
-
pageNum++;
|
|
142
|
-
}
|
|
143
|
-
const textItems = (textContent.items as any[]).map((item: any) => {
|
|
144
|
-
const tx = [1, 0, 0, 1, 0, 0];
|
|
145
|
-
for (let i = 0; i < 6; i++) {
|
|
146
|
-
tx[i] += item.transform[i] * viewport.transform[i % 2 ? 3 : 0];
|
|
147
|
-
if (i % 2) {
|
|
148
|
-
tx[i + 1] += item.transform[i] * viewport.transform[1];
|
|
149
|
-
}
|
|
150
|
-
}
|
|
151
|
-
|
|
152
|
-
const fontHeight = Math.sqrt(tx[2] * tx[2] + tx[3] * tx[3]);
|
|
153
|
-
const dividedHeight = item.height / fontHeight;
|
|
154
|
-
return new TextItem({
|
|
155
|
-
x: Math.round(item.transform[4]),
|
|
156
|
-
y: Math.round(item.transform[5]),
|
|
157
|
-
width: Math.round(item.width),
|
|
158
|
-
height: Math.round(dividedHeight <= 1 ? item.height : dividedHeight),
|
|
159
|
-
text: item.str,
|
|
160
|
-
font: item.fontName,
|
|
161
|
-
});
|
|
162
|
-
});
|
|
163
|
-
pages[page.pageNumber - 1].items = textItems;
|
|
164
|
-
}
|
|
165
|
-
|
|
166
|
-
var parseResult = new ParseResult({ pages });
|
|
167
|
-
|
|
168
|
-
let lastTransformation: (typeof transformations)[number] | undefined,
|
|
169
|
-
transformations = [
|
|
170
|
-
new CalculateGlobalStats(),
|
|
171
|
-
new CompactLines(),
|
|
172
|
-
new RemoveRepetitiveElements(),
|
|
173
|
-
new VerticalToHorizontal(),
|
|
174
|
-
new DetectTOC(),
|
|
175
|
-
new DetectHeaders(),
|
|
176
|
-
new DetectListItems(),
|
|
177
|
-
|
|
178
|
-
new GatherBlocks(),
|
|
179
|
-
new DetectCodeQuoteBlocks(),
|
|
180
|
-
new DetectListLevels(),
|
|
181
|
-
|
|
182
|
-
new ToTextBlocks(),
|
|
183
|
-
new ToHTML(),
|
|
184
|
-
];
|
|
185
|
-
|
|
186
|
-
transformations?.forEach((transformation) => {
|
|
187
|
-
if (lastTransformation) {
|
|
188
|
-
parseResult = lastTransformation.completeTransform(parseResult);
|
|
189
|
-
}
|
|
190
|
-
parseResult = transformation.transform(parseResult);
|
|
191
|
-
lastTransformation = transformation;
|
|
192
|
-
});
|
|
193
|
-
|
|
194
|
-
var html = parseResult.pages.reduce((acc, page, pageNumber) => {
|
|
195
|
-
return (
|
|
196
|
-
acc +
|
|
197
|
-
`<p id="page-${pageNumber + 1}">${
|
|
198
|
-
addPageNumbers ? ` [${pageNumber + 1}] ` : ""
|
|
199
|
-
}${page.items.join('</p><p id="page-' + pageNumber + '">')}</p>`
|
|
200
|
-
);
|
|
201
|
-
}, "");
|
|
202
|
-
|
|
203
|
-
if (addCitation) {
|
|
204
|
-
// Get metadata
|
|
205
|
-
// avoid using date as it is unreliable sand generally file mod date
|
|
206
|
-
var metadata = await pdfDocument.getMetadata();
|
|
207
|
-
var { Author: author, Title: title } = metadata.info as any;
|
|
208
|
-
// date =
|
|
209
|
-
// date.slice(2, 6) + "-" + date.slice(6, 8) + "-" + date.slice(8, 10);
|
|
210
|
-
// date = date ? new Date(date)?.toISOString().split("T")[0] : null;
|
|
211
|
-
|
|
212
|
-
//look for date in first page
|
|
213
|
-
// date = chrono
|
|
214
|
-
// .parseDate(content.slice(0, 400))
|
|
215
|
-
// ?.toISOString()
|
|
216
|
-
// .split("T")[0];
|
|
217
|
-
// // || date;
|
|
218
|
-
|
|
219
|
-
title = html.slice(0, 400).match(/<h[0-9]>(.*?)<\/h[0-9]>/)?.[1] || title;
|
|
220
|
-
}
|
|
221
|
-
|
|
222
|
-
return { author, title, html, format: "pdf" };
|
|
223
|
-
}
|
|
224
|
-
|
|
225
|
-
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview High-fidelity PDF-to-HTML conversion pipeline.
|
|
3
|
+
* Extracts structural elements (headers, lists, code blocks) and handles page-level metadata.
|
|
4
|
+
*/
|
|
5
|
+
import {
|
|
6
|
+
findPageNumbers,
|
|
7
|
+
findFirstPage,
|
|
8
|
+
removePageNumber,
|
|
9
|
+
} from "./utils/page-number-functions";
|
|
10
|
+
import TextItem from "./models/text-item";
|
|
11
|
+
import Page from "./models/page";
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Fetch wrapper for grabbing binary content
|
|
15
|
+
*/
|
|
16
|
+
async function grab(url: string, options: { responseType?: string; timeout?: number } = {}) {
|
|
17
|
+
const timeout = options.timeout ? options.timeout * 1000 : 10000;
|
|
18
|
+
const controller = new AbortController();
|
|
19
|
+
const timeoutId = setTimeout(() => controller.abort(), timeout);
|
|
20
|
+
|
|
21
|
+
try {
|
|
22
|
+
const response = await fetch(url, {
|
|
23
|
+
signal: controller.signal,
|
|
24
|
+
});
|
|
25
|
+
clearTimeout(timeoutId);
|
|
26
|
+
|
|
27
|
+
if (!response.ok) {
|
|
28
|
+
throw new Error(`HTTP ${response.status}`);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
if (options.responseType === "arraybuffer") {
|
|
32
|
+
return await response.arrayBuffer();
|
|
33
|
+
}
|
|
34
|
+
return await response.text();
|
|
35
|
+
} catch (error) {
|
|
36
|
+
clearTimeout(timeoutId);
|
|
37
|
+
throw error;
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
import CalculateGlobalStats from "./transforms/calculate-global-stats";
|
|
42
|
+
import CompactLines from "./transforms/line-item/compact-lines";
|
|
43
|
+
import RemoveRepetitiveElements from "./transforms/line-item/remove-repetitive-elements";
|
|
44
|
+
import VerticalToHorizontal from "./transforms/line-item/vertical-to-horizontal";
|
|
45
|
+
import DetectTOC from "./transforms/line-item/detect-toc";
|
|
46
|
+
import DetectListItems from "./transforms/line-item/detect-list-items";
|
|
47
|
+
import DetectHeaders from "./transforms/line-item/detect-headers";
|
|
48
|
+
|
|
49
|
+
import GatherBlocks from "./transforms/block/gather-blocks";
|
|
50
|
+
import DetectCodeQuoteBlocks from "./transforms/block/detect-code-quote-blocks";
|
|
51
|
+
import DetectListLevels from "./transforms/block/detect-list-levels";
|
|
52
|
+
import ToTextBlocks from "./transforms/to-text-blocks";
|
|
53
|
+
import ToHTML from "./transforms/to-html";
|
|
54
|
+
import ParseResult from "./models/parse-result";
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Extracts formatted text from PDF with parsing of linebreaks ,
|
|
58
|
+
* page headers, footnotes, and section headings. Supports fonts, links, bold,
|
|
59
|
+
* italics, lists, headings, headers, footnotes, and Table of Contents,
|
|
60
|
+
* Quotes, and Code Blocks, . Removes repeated headers, links footnote anchors to the footnote,
|
|
61
|
+
* and preserves number of the PDF page with invisible I element.
|
|
62
|
+
*
|
|
63
|
+
* This function uses [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless)
|
|
64
|
+
* to work in more environments than PDF.js-based tools:
|
|
65
|
+
* Cloudflare workers, serverless, node.js, and front-end only.
|
|
66
|
+
* @param {string} pdfURLOrBuffer - URL to a PDF file or buffer from fs.readFile
|
|
67
|
+
* @param {Object} [options]
|
|
68
|
+
* @param {boolean} options.addPageNumbers default=false - Adds # to end of each page
|
|
69
|
+
* @param {boolean} options.removePageHeaders default=true - Removes repeated headers found on each page
|
|
70
|
+
* @returns {string|Object} HTML formatted text
|
|
71
|
+
* @category Extract
|
|
72
|
+
* @author [vtempest (2025)](https://github.com/vtempest),
|
|
73
|
+
* [pdf-to-markdown (2017)](https://github.com/jzillmann/pdf-to-markdown/tree/master),
|
|
74
|
+
* [pdf.js (2012-)](https://github.com/mozilla/pdf.js/releases),
|
|
75
|
+
*/
|
|
76
|
+
export async function convertPDFToHTML(
|
|
77
|
+
pdfURLOrBuffer: any,
|
|
78
|
+
options: { addPageNumbers?: boolean; addCitation?: boolean } = {},
|
|
79
|
+
) {
|
|
80
|
+
// try {
|
|
81
|
+
var { addPageNumbers = false, addCitation = true } = options;
|
|
82
|
+
|
|
83
|
+
// pass in databuffer or download all pdf data
|
|
84
|
+
// and convert to array buffer
|
|
85
|
+
var buffer =
|
|
86
|
+
typeof pdfURLOrBuffer === "string"
|
|
87
|
+
? await grab(pdfURLOrBuffer, {
|
|
88
|
+
responseType: "arraybuffer",
|
|
89
|
+
timeout: 10,
|
|
90
|
+
})
|
|
91
|
+
: pdfURLOrBuffer;
|
|
92
|
+
|
|
93
|
+
let pdfDocument;
|
|
94
|
+
try {
|
|
95
|
+
let { resolvePDFJS } = await import("https://cdn.jsdelivr.net/npm/pdfjs-serverless@1.1.0/+esm" as any);
|
|
96
|
+
|
|
97
|
+
const { getDocument } = await resolvePDFJS();
|
|
98
|
+
pdfDocument = await getDocument({
|
|
99
|
+
data: new Uint8Array(buffer),
|
|
100
|
+
useSystemFonts: true,
|
|
101
|
+
verbosity: 0,
|
|
102
|
+
}).promise;
|
|
103
|
+
} catch (e: any) {
|
|
104
|
+
return { error: e.message };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const pages = [...Array(pdfDocument.numPages).keys()].map(
|
|
108
|
+
(index) => new Page({ index }),
|
|
109
|
+
);
|
|
110
|
+
|
|
111
|
+
let pageIndexNumMap = {};
|
|
112
|
+
let firstPage;
|
|
113
|
+
for (let j = 1; j <= pdfDocument.numPages; j++) {
|
|
114
|
+
const page = await pdfDocument.getPage(j);
|
|
115
|
+
const textContent = await page.getTextContent();
|
|
116
|
+
|
|
117
|
+
if (Object.keys(pageIndexNumMap).length < 10) {
|
|
118
|
+
pageIndexNumMap = findPageNumbers(
|
|
119
|
+
pageIndexNumMap,
|
|
120
|
+
page.pageNumber - 1,
|
|
121
|
+
textContent.items,
|
|
122
|
+
);
|
|
123
|
+
} else {
|
|
124
|
+
firstPage = findFirstPage(pageIndexNumMap);
|
|
125
|
+
break;
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
let pageNum = firstPage ? firstPage.pageNum : 0;
|
|
130
|
+
for (let j = 1; j <= pdfDocument.numPages; j++) {
|
|
131
|
+
const page = await pdfDocument.getPage(j);
|
|
132
|
+
|
|
133
|
+
// Trigger the font retrieval for the page
|
|
134
|
+
await page.getOperatorList();
|
|
135
|
+
|
|
136
|
+
const scale = 1.0;
|
|
137
|
+
const viewport = page.getViewport({ scale });
|
|
138
|
+
let textContent = await page.getTextContent();
|
|
139
|
+
if (firstPage && (page as any).pageIndex >= firstPage.pageIndex) {
|
|
140
|
+
textContent = removePageNumber(textContent as any, pageNum) as any;
|
|
141
|
+
pageNum++;
|
|
142
|
+
}
|
|
143
|
+
const textItems = (textContent.items as any[]).map((item: any) => {
|
|
144
|
+
const tx = [1, 0, 0, 1, 0, 0];
|
|
145
|
+
for (let i = 0; i < 6; i++) {
|
|
146
|
+
tx[i] += item.transform[i] * viewport.transform[i % 2 ? 3 : 0];
|
|
147
|
+
if (i % 2) {
|
|
148
|
+
tx[i + 1] += item.transform[i] * viewport.transform[1];
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
const fontHeight = Math.sqrt(tx[2] * tx[2] + tx[3] * tx[3]);
|
|
153
|
+
const dividedHeight = item.height / fontHeight;
|
|
154
|
+
return new TextItem({
|
|
155
|
+
x: Math.round(item.transform[4]),
|
|
156
|
+
y: Math.round(item.transform[5]),
|
|
157
|
+
width: Math.round(item.width),
|
|
158
|
+
height: Math.round(dividedHeight <= 1 ? item.height : dividedHeight),
|
|
159
|
+
text: item.str,
|
|
160
|
+
font: item.fontName,
|
|
161
|
+
});
|
|
162
|
+
});
|
|
163
|
+
pages[page.pageNumber - 1].items = textItems;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
var parseResult = new ParseResult({ pages });
|
|
167
|
+
|
|
168
|
+
let lastTransformation: (typeof transformations)[number] | undefined,
|
|
169
|
+
transformations = [
|
|
170
|
+
new CalculateGlobalStats(),
|
|
171
|
+
new CompactLines(),
|
|
172
|
+
new RemoveRepetitiveElements(),
|
|
173
|
+
new VerticalToHorizontal(),
|
|
174
|
+
new DetectTOC(),
|
|
175
|
+
new DetectHeaders(),
|
|
176
|
+
new DetectListItems(),
|
|
177
|
+
|
|
178
|
+
new GatherBlocks(),
|
|
179
|
+
new DetectCodeQuoteBlocks(),
|
|
180
|
+
new DetectListLevels(),
|
|
181
|
+
|
|
182
|
+
new ToTextBlocks(),
|
|
183
|
+
new ToHTML(),
|
|
184
|
+
];
|
|
185
|
+
|
|
186
|
+
transformations?.forEach((transformation) => {
|
|
187
|
+
if (lastTransformation) {
|
|
188
|
+
parseResult = lastTransformation.completeTransform(parseResult);
|
|
189
|
+
}
|
|
190
|
+
parseResult = transformation.transform(parseResult);
|
|
191
|
+
lastTransformation = transformation;
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
var html = parseResult.pages.reduce((acc, page, pageNumber) => {
|
|
195
|
+
return (
|
|
196
|
+
acc +
|
|
197
|
+
`<p id="page-${pageNumber + 1}">${
|
|
198
|
+
addPageNumbers ? ` [${pageNumber + 1}] ` : ""
|
|
199
|
+
}${page.items.join('</p><p id="page-' + pageNumber + '">')}</p>`
|
|
200
|
+
);
|
|
201
|
+
}, "");
|
|
202
|
+
|
|
203
|
+
if (addCitation) {
|
|
204
|
+
// Get metadata
|
|
205
|
+
// avoid using date as it is unreliable sand generally file mod date
|
|
206
|
+
var metadata = await pdfDocument.getMetadata();
|
|
207
|
+
var { Author: author, Title: title } = metadata.info as any;
|
|
208
|
+
// date =
|
|
209
|
+
// date.slice(2, 6) + "-" + date.slice(6, 8) + "-" + date.slice(8, 10);
|
|
210
|
+
// date = date ? new Date(date)?.toISOString().split("T")[0] : null;
|
|
211
|
+
|
|
212
|
+
//look for date in first page
|
|
213
|
+
// date = chrono
|
|
214
|
+
// .parseDate(content.slice(0, 400))
|
|
215
|
+
// ?.toISOString()
|
|
216
|
+
// .split("T")[0];
|
|
217
|
+
// // || date;
|
|
218
|
+
|
|
219
|
+
title = html.slice(0, 400).match(/<h[0-9]>(.*?)<\/h[0-9]>/)?.[1] || title;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
return { author, title, html, format: "pdf" };
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
|
|
@@ -1,29 +1,29 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Abstract base for pipeline stages that produce `LineItemBlock`-typed
|
|
3
|
-
* page items (GatherBlocks, DetectCodeQuoteBlocks, DetectListLevels). Provides
|
|
4
|
-
* `completeTransform` which strips `REMOVED_ANNOTATION` items and clears all
|
|
5
|
-
* remaining annotations before passing the result to the next stage.
|
|
6
|
-
*/
|
|
7
|
-
import Transformation from './transformation'
|
|
8
|
-
import LineItemBlock from '../../models/line-item-block'
|
|
9
|
-
import ParseResult from '../../models/parse-result'
|
|
10
|
-
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
-
|
|
12
|
-
// Abstract class for transformations producing LineItemBlock(s) to be shown in the LineItemBlockPageView
|
|
13
|
-
export default class ToLineItemBlockTransformation extends Transformation {
|
|
14
|
-
constructor(name: string) {
|
|
15
|
-
super(name, LineItemBlock.name)
|
|
16
|
-
if (this.constructor === ToLineItemBlockTransformation) {
|
|
17
|
-
throw new TypeError('Can not construct abstract class.')
|
|
18
|
-
}
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
completeTransform(parseResult: ParseResult): ParseResult {
|
|
22
|
-
parseResult.messages = []
|
|
23
|
-
parseResult.pages.forEach(page => {
|
|
24
|
-
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
25
|
-
page.items.forEach(item => (item.annotation = null))
|
|
26
|
-
})
|
|
27
|
-
return parseResult
|
|
28
|
-
}
|
|
29
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Abstract base for pipeline stages that produce `LineItemBlock`-typed
|
|
3
|
+
* page items (GatherBlocks, DetectCodeQuoteBlocks, DetectListLevels). Provides
|
|
4
|
+
* `completeTransform` which strips `REMOVED_ANNOTATION` items and clears all
|
|
5
|
+
* remaining annotations before passing the result to the next stage.
|
|
6
|
+
*/
|
|
7
|
+
import Transformation from './transformation'
|
|
8
|
+
import LineItemBlock from '../../models/line-item-block'
|
|
9
|
+
import ParseResult from '../../models/parse-result'
|
|
10
|
+
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
+
|
|
12
|
+
// Abstract class for transformations producing LineItemBlock(s) to be shown in the LineItemBlockPageView
|
|
13
|
+
export default class ToLineItemBlockTransformation extends Transformation {
|
|
14
|
+
constructor(name: string) {
|
|
15
|
+
super(name, LineItemBlock.name)
|
|
16
|
+
if (this.constructor === ToLineItemBlockTransformation) {
|
|
17
|
+
throw new TypeError('Can not construct abstract class.')
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
completeTransform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
parseResult.messages = []
|
|
23
|
+
parseResult.pages.forEach(page => {
|
|
24
|
+
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
25
|
+
page.items.forEach(item => (item.annotation = null))
|
|
26
|
+
})
|
|
27
|
+
return parseResult
|
|
28
|
+
}
|
|
29
|
+
}
|
|
@@ -1,29 +1,29 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Abstract base for pipeline stages that produce `LineItem`-typed
|
|
3
|
-
* page items (CompactLines, RemoveRepetitiveElements, VerticalToHorizontal,
|
|
4
|
-
* DetectTOC, DetectHeaders, DetectListItems). Provides `completeTransform` which
|
|
5
|
-
* strips `REMOVED_ANNOTATION` items and clears annotations before the next stage.
|
|
6
|
-
*/
|
|
7
|
-
import Transformation from './transformation'
|
|
8
|
-
import LineItem from '../../models/line-item'
|
|
9
|
-
import ParseResult from '../../models/parse-result'
|
|
10
|
-
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
-
|
|
12
|
-
// Abstract class for transformations producing LineItem(s) to be shown in the LineItemPageView
|
|
13
|
-
export default class ToLineItemTransformation extends Transformation {
|
|
14
|
-
constructor(name: string) {
|
|
15
|
-
super(name, LineItem.name)
|
|
16
|
-
if (this.constructor === ToLineItemTransformation) {
|
|
17
|
-
throw new TypeError('Can not construct abstract class.')
|
|
18
|
-
}
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
completeTransform(parseResult: ParseResult): ParseResult {
|
|
22
|
-
parseResult.messages = []
|
|
23
|
-
parseResult.pages.forEach(page => {
|
|
24
|
-
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
25
|
-
page.items.forEach(item => (item.annotation = null))
|
|
26
|
-
})
|
|
27
|
-
return parseResult
|
|
28
|
-
}
|
|
29
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Abstract base for pipeline stages that produce `LineItem`-typed
|
|
3
|
+
* page items (CompactLines, RemoveRepetitiveElements, VerticalToHorizontal,
|
|
4
|
+
* DetectTOC, DetectHeaders, DetectListItems). Provides `completeTransform` which
|
|
5
|
+
* strips `REMOVED_ANNOTATION` items and clears annotations before the next stage.
|
|
6
|
+
*/
|
|
7
|
+
import Transformation from './transformation'
|
|
8
|
+
import LineItem from '../../models/line-item'
|
|
9
|
+
import ParseResult from '../../models/parse-result'
|
|
10
|
+
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
+
|
|
12
|
+
// Abstract class for transformations producing LineItem(s) to be shown in the LineItemPageView
|
|
13
|
+
export default class ToLineItemTransformation extends Transformation {
|
|
14
|
+
constructor(name: string) {
|
|
15
|
+
super(name, LineItem.name)
|
|
16
|
+
if (this.constructor === ToLineItemTransformation) {
|
|
17
|
+
throw new TypeError('Can not construct abstract class.')
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
completeTransform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
parseResult.messages = []
|
|
23
|
+
parseResult.pages.forEach(page => {
|
|
24
|
+
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
25
|
+
page.items.forEach(item => (item.annotation = null))
|
|
26
|
+
})
|
|
27
|
+
return parseResult
|
|
28
|
+
}
|
|
29
|
+
}
|
|
@@ -1,28 +1,28 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Abstract base for pipeline stages that operate on raw `TextItem`
|
|
3
|
-
* page items (the earliest stages, currently only `CalculateGlobalStats`). Provides
|
|
4
|
-
* the standard `completeTransform` cleanup that removes `REMOVED_ANNOTATION` items
|
|
5
|
-
* and clears annotations before passing control to the next transformation.
|
|
6
|
-
*/
|
|
7
|
-
import Transformation from './transformation'
|
|
8
|
-
import TextItem from '../../models/text-item'
|
|
9
|
-
import ParseResult from '../../models/parse-result'
|
|
10
|
-
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
-
|
|
12
|
-
export default class ToTextItemTransformation extends Transformation {
|
|
13
|
-
constructor(name: string) {
|
|
14
|
-
super(name, TextItem.name)
|
|
15
|
-
if (this.constructor === ToTextItemTransformation) {
|
|
16
|
-
throw new TypeError('Can not construct abstract class.')
|
|
17
|
-
}
|
|
18
|
-
}
|
|
19
|
-
|
|
20
|
-
completeTransform(parseResult: ParseResult): ParseResult {
|
|
21
|
-
parseResult.messages = []
|
|
22
|
-
parseResult.pages.forEach(page => {
|
|
23
|
-
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
24
|
-
page.items.forEach(item => (item.annotation = null))
|
|
25
|
-
})
|
|
26
|
-
return parseResult
|
|
27
|
-
}
|
|
28
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Abstract base for pipeline stages that operate on raw `TextItem`
|
|
3
|
+
* page items (the earliest stages, currently only `CalculateGlobalStats`). Provides
|
|
4
|
+
* the standard `completeTransform` cleanup that removes `REMOVED_ANNOTATION` items
|
|
5
|
+
* and clears annotations before passing control to the next transformation.
|
|
6
|
+
*/
|
|
7
|
+
import Transformation from './transformation'
|
|
8
|
+
import TextItem from '../../models/text-item'
|
|
9
|
+
import ParseResult from '../../models/parse-result'
|
|
10
|
+
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
11
|
+
|
|
12
|
+
export default class ToTextItemTransformation extends Transformation {
|
|
13
|
+
constructor(name: string) {
|
|
14
|
+
super(name, TextItem.name)
|
|
15
|
+
if (this.constructor === ToTextItemTransformation) {
|
|
16
|
+
throw new TypeError('Can not construct abstract class.')
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
completeTransform(parseResult: ParseResult): ParseResult {
|
|
21
|
+
parseResult.messages = []
|
|
22
|
+
parseResult.pages.forEach(page => {
|
|
23
|
+
page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
|
|
24
|
+
page.items.forEach(item => (item.annotation = null))
|
|
25
|
+
})
|
|
26
|
+
return parseResult
|
|
27
|
+
}
|
|
28
|
+
}
|