extract-pdf 0.1.21 → 0.1.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -74
- package/package.json +1 -1
- package/src/models/annotation.ts +41 -41
- package/src/models/block-type.ts +203 -203
- package/src/models/line-converter.ts +224 -224
- package/src/models/metadata.ts +29 -29
- package/src/models/page.ts +16 -16
- package/src/models/parse-result.ts +32 -32
- package/src/models/parsed-elements.ts +29 -29
- package/src/models/stashing-stream.ts +86 -86
- package/src/models/text-item-line-grouper.ts +41 -41
- package/src/models/word.ts +31 -31
- package/src/pdf-to-html.ts +225 -225
- package/src/transforms/base/to-line-item-block-transform.ts +29 -29
- package/src/transforms/base/to-line-item-transform.ts +29 -29
- package/src/transforms/base/to-text-item-transform.ts +28 -28
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
- package/src/transforms/block/detect-list-levels.ts +64 -64
- package/src/transforms/block/gather-blocks.ts +113 -113
- package/src/transforms/calculate-global-stats.ts +132 -132
- package/src/transforms/line-item/compact-lines.ts +92 -92
- package/src/transforms/line-item/detect-headers.ts +173 -173
- package/src/transforms/line-item/detect-list-items.ts +68 -68
- package/src/transforms/line-item/detect-toc.ts +459 -459
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
- package/src/transforms/to-html.ts +46 -46
- package/src/utils/is-url-pdf.ts +33 -33
- package/src/utils/page-item-functions.ts +35 -35
- package/src/utils/string-functions.ts +124 -124
|
@@ -1,101 +1,101 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Line-item transformation that removes running page headers and
|
|
3
|
-
* footers. Hashes the topmost and bottommost line of each page (ignoring digits
|
|
4
|
-
* and whitespace so page numbers don't break the match), then marks lines whose
|
|
5
|
-
* hash appears on more than two-thirds of all pages with `REMOVED_ANNOTATION`.
|
|
6
|
-
*/
|
|
7
|
-
import ToLineItemTransformation from '../base/to-line-item-transform'
|
|
8
|
-
import ParseResult from '../../models/parse-result'
|
|
9
|
-
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
10
|
-
|
|
11
|
-
import { isDigit } from '../../utils/string-functions'
|
|
12
|
-
|
|
13
|
-
function hashCodeIgnoringSpacesAndNumbers(string: string): number {
|
|
14
|
-
var hash = 0
|
|
15
|
-
if (string.trim().length === 0) return hash
|
|
16
|
-
for (var i = 0; i < string.length; i++) {
|
|
17
|
-
const charCode = string.charCodeAt(i)
|
|
18
|
-
if (!isDigit(charCode) && charCode !== 32 && charCode !== 160) {
|
|
19
|
-
hash = ((hash << 5) - hash) + charCode
|
|
20
|
-
hash |= 0 // Convert to 32bit integer
|
|
21
|
-
}
|
|
22
|
-
}
|
|
23
|
-
return hash
|
|
24
|
-
}
|
|
25
|
-
|
|
26
|
-
// Remove elements with similar content on same page positions, like page numbers, licenes information, etc...
|
|
27
|
-
export default class RemoveRepetitiveElements extends ToLineItemTransformation {
|
|
28
|
-
constructor () {
|
|
29
|
-
super('Remove Repetitive Elements')
|
|
30
|
-
}
|
|
31
|
-
|
|
32
|
-
// The idea is the following:
|
|
33
|
-
// - For each page, collect all items of the first, and all items of the last line
|
|
34
|
-
// - Calculate how often these items occur accros all pages (hash ignoring numbers, whitespace, upper/lowercase)
|
|
35
|
-
// - Delete items occuring on more then 2/3 of all pages
|
|
36
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
37
|
-
// find first and last lines per page
|
|
38
|
-
const pageStore = []
|
|
39
|
-
const minLineHashRepetitions = {}
|
|
40
|
-
const maxLineHashRepetitions = {}
|
|
41
|
-
parseResult.pages.forEach(page => {
|
|
42
|
-
const minMaxItems = page.items.reduce((itemStore, item) => {
|
|
43
|
-
if (item.y < itemStore.minY) {
|
|
44
|
-
itemStore.minElements = [item]
|
|
45
|
-
itemStore.minY = item.y
|
|
46
|
-
} else if (item.y === itemStore.minY) {
|
|
47
|
-
itemStore.minElements.push(item)
|
|
48
|
-
}
|
|
49
|
-
if (item.y > itemStore.maxY) {
|
|
50
|
-
itemStore.maxElements = [item]
|
|
51
|
-
itemStore.maxY = item.y
|
|
52
|
-
} else if (item.y === itemStore.maxY) {
|
|
53
|
-
itemStore.maxElements.push(item)
|
|
54
|
-
}
|
|
55
|
-
return itemStore
|
|
56
|
-
}, {
|
|
57
|
-
minY: 999,
|
|
58
|
-
maxY: 0,
|
|
59
|
-
minElements: [],
|
|
60
|
-
maxElements: [],
|
|
61
|
-
})
|
|
62
|
-
|
|
63
|
-
const minLineHash = hashCodeIgnoringSpacesAndNumbers(minMaxItems.minElements.reduce((combinedString, item) => combinedString + item.text().toUpperCase(), ''))
|
|
64
|
-
const maxLineHash = hashCodeIgnoringSpacesAndNumbers(minMaxItems.maxElements.reduce((combinedString, item) => combinedString + item.text().toUpperCase(), ''))
|
|
65
|
-
pageStore.push({
|
|
66
|
-
minElements: minMaxItems.minElements,
|
|
67
|
-
maxElements: minMaxItems.maxElements,
|
|
68
|
-
minLineHash: minLineHash,
|
|
69
|
-
maxLineHash: maxLineHash,
|
|
70
|
-
})
|
|
71
|
-
minLineHashRepetitions[minLineHash] = minLineHashRepetitions[minLineHash] ? minLineHashRepetitions[minLineHash] + 1 : 1
|
|
72
|
-
maxLineHashRepetitions[maxLineHash] = maxLineHashRepetitions[maxLineHash] ? maxLineHashRepetitions[maxLineHash] + 1 : 1
|
|
73
|
-
})
|
|
74
|
-
|
|
75
|
-
// now annoate all removed items
|
|
76
|
-
var removedHeader = 0
|
|
77
|
-
var removedFooter = 0
|
|
78
|
-
parseResult.pages.forEach((page, i) => {
|
|
79
|
-
if (minLineHashRepetitions[pageStore[i].minLineHash] >= Math.max(3, parseResult.pages.length * 2 / 3)) {
|
|
80
|
-
pageStore[i].minElements.forEach(item => {
|
|
81
|
-
item.annotation = REMOVED_ANNOTATION
|
|
82
|
-
})
|
|
83
|
-
removedFooter++
|
|
84
|
-
}
|
|
85
|
-
if (maxLineHashRepetitions[pageStore[i].maxLineHash] >= Math.max(3, parseResult.pages.length * 2 / 3)) {
|
|
86
|
-
pageStore[i].maxElements.forEach(item => {
|
|
87
|
-
item.annotation = REMOVED_ANNOTATION
|
|
88
|
-
})
|
|
89
|
-
removedHeader++
|
|
90
|
-
}
|
|
91
|
-
})
|
|
92
|
-
|
|
93
|
-
return new ParseResult({
|
|
94
|
-
...parseResult,
|
|
95
|
-
messages: [
|
|
96
|
-
'Removed Header: ' + removedHeader,
|
|
97
|
-
'Removed Footers: ' + removedFooter,
|
|
98
|
-
],
|
|
99
|
-
})
|
|
100
|
-
}
|
|
101
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Line-item transformation that removes running page headers and
|
|
3
|
+
* footers. Hashes the topmost and bottommost line of each page (ignoring digits
|
|
4
|
+
* and whitespace so page numbers don't break the match), then marks lines whose
|
|
5
|
+
* hash appears on more than two-thirds of all pages with `REMOVED_ANNOTATION`.
|
|
6
|
+
*/
|
|
7
|
+
import ToLineItemTransformation from '../base/to-line-item-transform'
|
|
8
|
+
import ParseResult from '../../models/parse-result'
|
|
9
|
+
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
10
|
+
|
|
11
|
+
import { isDigit } from '../../utils/string-functions'
|
|
12
|
+
|
|
13
|
+
function hashCodeIgnoringSpacesAndNumbers(string: string): number {
|
|
14
|
+
var hash = 0
|
|
15
|
+
if (string.trim().length === 0) return hash
|
|
16
|
+
for (var i = 0; i < string.length; i++) {
|
|
17
|
+
const charCode = string.charCodeAt(i)
|
|
18
|
+
if (!isDigit(charCode) && charCode !== 32 && charCode !== 160) {
|
|
19
|
+
hash = ((hash << 5) - hash) + charCode
|
|
20
|
+
hash |= 0 // Convert to 32bit integer
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
return hash
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// Remove elements with similar content on same page positions, like page numbers, licenes information, etc...
|
|
27
|
+
export default class RemoveRepetitiveElements extends ToLineItemTransformation {
|
|
28
|
+
constructor () {
|
|
29
|
+
super('Remove Repetitive Elements')
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// The idea is the following:
|
|
33
|
+
// - For each page, collect all items of the first, and all items of the last line
|
|
34
|
+
// - Calculate how often these items occur accros all pages (hash ignoring numbers, whitespace, upper/lowercase)
|
|
35
|
+
// - Delete items occuring on more then 2/3 of all pages
|
|
36
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
37
|
+
// find first and last lines per page
|
|
38
|
+
const pageStore = []
|
|
39
|
+
const minLineHashRepetitions = {}
|
|
40
|
+
const maxLineHashRepetitions = {}
|
|
41
|
+
parseResult.pages.forEach(page => {
|
|
42
|
+
const minMaxItems = page.items.reduce((itemStore, item) => {
|
|
43
|
+
if (item.y < itemStore.minY) {
|
|
44
|
+
itemStore.minElements = [item]
|
|
45
|
+
itemStore.minY = item.y
|
|
46
|
+
} else if (item.y === itemStore.minY) {
|
|
47
|
+
itemStore.minElements.push(item)
|
|
48
|
+
}
|
|
49
|
+
if (item.y > itemStore.maxY) {
|
|
50
|
+
itemStore.maxElements = [item]
|
|
51
|
+
itemStore.maxY = item.y
|
|
52
|
+
} else if (item.y === itemStore.maxY) {
|
|
53
|
+
itemStore.maxElements.push(item)
|
|
54
|
+
}
|
|
55
|
+
return itemStore
|
|
56
|
+
}, {
|
|
57
|
+
minY: 999,
|
|
58
|
+
maxY: 0,
|
|
59
|
+
minElements: [],
|
|
60
|
+
maxElements: [],
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
const minLineHash = hashCodeIgnoringSpacesAndNumbers(minMaxItems.minElements.reduce((combinedString, item) => combinedString + item.text().toUpperCase(), ''))
|
|
64
|
+
const maxLineHash = hashCodeIgnoringSpacesAndNumbers(minMaxItems.maxElements.reduce((combinedString, item) => combinedString + item.text().toUpperCase(), ''))
|
|
65
|
+
pageStore.push({
|
|
66
|
+
minElements: minMaxItems.minElements,
|
|
67
|
+
maxElements: minMaxItems.maxElements,
|
|
68
|
+
minLineHash: minLineHash,
|
|
69
|
+
maxLineHash: maxLineHash,
|
|
70
|
+
})
|
|
71
|
+
minLineHashRepetitions[minLineHash] = minLineHashRepetitions[minLineHash] ? minLineHashRepetitions[minLineHash] + 1 : 1
|
|
72
|
+
maxLineHashRepetitions[maxLineHash] = maxLineHashRepetitions[maxLineHash] ? maxLineHashRepetitions[maxLineHash] + 1 : 1
|
|
73
|
+
})
|
|
74
|
+
|
|
75
|
+
// now annoate all removed items
|
|
76
|
+
var removedHeader = 0
|
|
77
|
+
var removedFooter = 0
|
|
78
|
+
parseResult.pages.forEach((page, i) => {
|
|
79
|
+
if (minLineHashRepetitions[pageStore[i].minLineHash] >= Math.max(3, parseResult.pages.length * 2 / 3)) {
|
|
80
|
+
pageStore[i].minElements.forEach(item => {
|
|
81
|
+
item.annotation = REMOVED_ANNOTATION
|
|
82
|
+
})
|
|
83
|
+
removedFooter++
|
|
84
|
+
}
|
|
85
|
+
if (maxLineHashRepetitions[pageStore[i].maxLineHash] >= Math.max(3, parseResult.pages.length * 2 / 3)) {
|
|
86
|
+
pageStore[i].maxElements.forEach(item => {
|
|
87
|
+
item.annotation = REMOVED_ANNOTATION
|
|
88
|
+
})
|
|
89
|
+
removedHeader++
|
|
90
|
+
}
|
|
91
|
+
})
|
|
92
|
+
|
|
93
|
+
return new ParseResult({
|
|
94
|
+
...parseResult,
|
|
95
|
+
messages: [
|
|
96
|
+
'Removed Header: ' + removedHeader,
|
|
97
|
+
'Removed Footers: ' + removedFooter,
|
|
98
|
+
],
|
|
99
|
+
})
|
|
100
|
+
}
|
|
101
|
+
}
|
|
@@ -1,90 +1,90 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Line-item transformation that recovers rotated sidebar text.
|
|
3
|
-
* PDF extracts vertically-oriented labels as a sequence of single-character
|
|
4
|
-
* lines. `VerticalsStream` (a `StashingStream` subclass) detects runs of 6+
|
|
5
|
-
* such lines and merges them into one horizontal `LineItem`, combining their
|
|
6
|
-
* words and computing the correct bounding box.
|
|
7
|
-
*/
|
|
8
|
-
import ToLineItemTransformation from "../base/to-line-item-transform";
|
|
9
|
-
import ParseResult from "../../models/parse-result";
|
|
10
|
-
import LineItem from "../../models/line-item";
|
|
11
|
-
import StashingStream from "../../models/stashing-stream";
|
|
12
|
-
import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
|
|
13
|
-
|
|
14
|
-
// Converts vertical text to horizontal
|
|
15
|
-
export default class VerticalToHorizontal extends ToLineItemTransformation {
|
|
16
|
-
constructor() {
|
|
17
|
-
super("Vertical to Horizontal Text");
|
|
18
|
-
}
|
|
19
|
-
|
|
20
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
21
|
-
var foundVerticals = 0;
|
|
22
|
-
parseResult.pages.forEach((page) => {
|
|
23
|
-
const stream = new VerticalsStream();
|
|
24
|
-
stream.consumeAll(page.items);
|
|
25
|
-
page.items = stream.complete();
|
|
26
|
-
foundVerticals += stream.foundVerticals;
|
|
27
|
-
});
|
|
28
|
-
|
|
29
|
-
return new ParseResult({
|
|
30
|
-
...parseResult,
|
|
31
|
-
messages: ["Converted " + foundVerticals + " verticals"],
|
|
32
|
-
});
|
|
33
|
-
}
|
|
34
|
-
}
|
|
35
|
-
|
|
36
|
-
class VerticalsStream extends StashingStream {
|
|
37
|
-
foundVerticals: number;
|
|
38
|
-
|
|
39
|
-
constructor() {
|
|
40
|
-
super();
|
|
41
|
-
this.foundVerticals = 0;
|
|
42
|
-
}
|
|
43
|
-
|
|
44
|
-
shouldStash(item: any): boolean {
|
|
45
|
-
return item.words.length === 1 && item.words[0].string.length === 1;
|
|
46
|
-
}
|
|
47
|
-
|
|
48
|
-
doMatchesStash(lastItem: any, item: any): boolean {
|
|
49
|
-
return (
|
|
50
|
-
lastItem.y - item.y > 5 && lastItem.words[0].type === item.words[0].type
|
|
51
|
-
);
|
|
52
|
-
}
|
|
53
|
-
|
|
54
|
-
doFlushStash(stash: any[], results: any[]): void {
|
|
55
|
-
if (stash.length > 5) {
|
|
56
|
-
// unite
|
|
57
|
-
var combinedWords = [];
|
|
58
|
-
var minX = 999;
|
|
59
|
-
var maxY = 0;
|
|
60
|
-
var sumWidth = 0;
|
|
61
|
-
var maxHeight = 0;
|
|
62
|
-
stash.forEach((oneCharacterLine) => {
|
|
63
|
-
oneCharacterLine.annotation = REMOVED_ANNOTATION;
|
|
64
|
-
results.push(oneCharacterLine);
|
|
65
|
-
combinedWords.push(oneCharacterLine.words[0]);
|
|
66
|
-
minX = Math.min(minX, oneCharacterLine.x);
|
|
67
|
-
maxY = Math.max(maxY, oneCharacterLine.y);
|
|
68
|
-
sumWidth += oneCharacterLine.width;
|
|
69
|
-
maxHeight = Math.max(maxHeight, oneCharacterLine.height);
|
|
70
|
-
});
|
|
71
|
-
results.push(
|
|
72
|
-
new LineItem({
|
|
73
|
-
...stash[0],
|
|
74
|
-
x: minX,
|
|
75
|
-
y: maxY,
|
|
76
|
-
width: sumWidth,
|
|
77
|
-
height: maxHeight,
|
|
78
|
-
words: combinedWords,
|
|
79
|
-
annotation: ADDED_ANNOTATION,
|
|
80
|
-
}),
|
|
81
|
-
);
|
|
82
|
-
this.foundVerticals++;
|
|
83
|
-
} else {
|
|
84
|
-
// add as singles
|
|
85
|
-
results.push(...stash);
|
|
86
|
-
}
|
|
87
|
-
}
|
|
88
|
-
}
|
|
89
|
-
|
|
90
|
-
|
|
1
|
+
/**
|
|
2
|
+
* @description Line-item transformation that recovers rotated sidebar text.
|
|
3
|
+
* PDF extracts vertically-oriented labels as a sequence of single-character
|
|
4
|
+
* lines. `VerticalsStream` (a `StashingStream` subclass) detects runs of 6+
|
|
5
|
+
* such lines and merges them into one horizontal `LineItem`, combining their
|
|
6
|
+
* words and computing the correct bounding box.
|
|
7
|
+
*/
|
|
8
|
+
import ToLineItemTransformation from "../base/to-line-item-transform";
|
|
9
|
+
import ParseResult from "../../models/parse-result";
|
|
10
|
+
import LineItem from "../../models/line-item";
|
|
11
|
+
import StashingStream from "../../models/stashing-stream";
|
|
12
|
+
import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
|
|
13
|
+
|
|
14
|
+
// Converts vertical text to horizontal
|
|
15
|
+
export default class VerticalToHorizontal extends ToLineItemTransformation {
|
|
16
|
+
constructor() {
|
|
17
|
+
super("Vertical to Horizontal Text");
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
21
|
+
var foundVerticals = 0;
|
|
22
|
+
parseResult.pages.forEach((page) => {
|
|
23
|
+
const stream = new VerticalsStream();
|
|
24
|
+
stream.consumeAll(page.items);
|
|
25
|
+
page.items = stream.complete();
|
|
26
|
+
foundVerticals += stream.foundVerticals;
|
|
27
|
+
});
|
|
28
|
+
|
|
29
|
+
return new ParseResult({
|
|
30
|
+
...parseResult,
|
|
31
|
+
messages: ["Converted " + foundVerticals + " verticals"],
|
|
32
|
+
});
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
class VerticalsStream extends StashingStream {
|
|
37
|
+
foundVerticals: number;
|
|
38
|
+
|
|
39
|
+
constructor() {
|
|
40
|
+
super();
|
|
41
|
+
this.foundVerticals = 0;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
shouldStash(item: any): boolean {
|
|
45
|
+
return item.words.length === 1 && item.words[0].string.length === 1;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
doMatchesStash(lastItem: any, item: any): boolean {
|
|
49
|
+
return (
|
|
50
|
+
lastItem.y - item.y > 5 && lastItem.words[0].type === item.words[0].type
|
|
51
|
+
);
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
doFlushStash(stash: any[], results: any[]): void {
|
|
55
|
+
if (stash.length > 5) {
|
|
56
|
+
// unite
|
|
57
|
+
var combinedWords = [];
|
|
58
|
+
var minX = 999;
|
|
59
|
+
var maxY = 0;
|
|
60
|
+
var sumWidth = 0;
|
|
61
|
+
var maxHeight = 0;
|
|
62
|
+
stash.forEach((oneCharacterLine) => {
|
|
63
|
+
oneCharacterLine.annotation = REMOVED_ANNOTATION;
|
|
64
|
+
results.push(oneCharacterLine);
|
|
65
|
+
combinedWords.push(oneCharacterLine.words[0]);
|
|
66
|
+
minX = Math.min(minX, oneCharacterLine.x);
|
|
67
|
+
maxY = Math.max(maxY, oneCharacterLine.y);
|
|
68
|
+
sumWidth += oneCharacterLine.width;
|
|
69
|
+
maxHeight = Math.max(maxHeight, oneCharacterLine.height);
|
|
70
|
+
});
|
|
71
|
+
results.push(
|
|
72
|
+
new LineItem({
|
|
73
|
+
...stash[0],
|
|
74
|
+
x: minX,
|
|
75
|
+
y: maxY,
|
|
76
|
+
width: sumWidth,
|
|
77
|
+
height: maxHeight,
|
|
78
|
+
words: combinedWords,
|
|
79
|
+
annotation: ADDED_ANNOTATION,
|
|
80
|
+
}),
|
|
81
|
+
);
|
|
82
|
+
this.foundVerticals++;
|
|
83
|
+
} else {
|
|
84
|
+
// add as singles
|
|
85
|
+
results.push(...stash);
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
|
|
@@ -1,46 +1,46 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Final serialization stage that converts each page's text block
|
|
3
|
-
* items into `<p>`-wrapped HTML. TOC blocks are emitted verbatim; hyphenated
|
|
4
|
-
* line-break artifacts are stripped from non-list blocks; `<code>` fencing is
|
|
5
|
-
* removed from CODE-category blocks (treated as prose in this simplified output).
|
|
6
|
-
*/
|
|
7
|
-
|
|
8
|
-
import Transformation from './base/transformation'
|
|
9
|
-
import ParseResult from '../models/parse-result'
|
|
10
|
-
|
|
11
|
-
export default class ToHTML extends Transformation {
|
|
12
|
-
constructor () {
|
|
13
|
-
super('To HTML', 'String')
|
|
14
|
-
}
|
|
15
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
16
|
-
parseResult.pages.forEach(page => {
|
|
17
|
-
var text = ''
|
|
18
|
-
page.items.forEach(block => {
|
|
19
|
-
// Concatenate all words in the same block, unless it's a Table of Contents block
|
|
20
|
-
let concatText
|
|
21
|
-
if (block.category === 'TOC') {
|
|
22
|
-
concatText = block.text
|
|
23
|
-
} else {
|
|
24
|
-
concatText = block.text.replace(/(\r\n|\n|\r)/gm, '\n')
|
|
25
|
-
}
|
|
26
|
-
|
|
27
|
-
// Concatenate words that were previously broken up by newline
|
|
28
|
-
if (block.category !== 'LIST') {
|
|
29
|
-
concatText = concatText.split('- ').join('')
|
|
30
|
-
}
|
|
31
|
-
|
|
32
|
-
// Assume there are no code blocks in our documents
|
|
33
|
-
if (block.category === 'CODE') {
|
|
34
|
-
concatText = concatText.split('<code>').join('').split('</code>').join('')
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
text += `<p>${concatText}</p>\n\n`
|
|
38
|
-
})
|
|
39
|
-
|
|
40
|
-
page.items = [text]
|
|
41
|
-
})
|
|
42
|
-
return new ParseResult({
|
|
43
|
-
...parseResult,
|
|
44
|
-
})
|
|
45
|
-
}
|
|
46
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Final serialization stage that converts each page's text block
|
|
3
|
+
* items into `<p>`-wrapped HTML. TOC blocks are emitted verbatim; hyphenated
|
|
4
|
+
* line-break artifacts are stripped from non-list blocks; `<code>` fencing is
|
|
5
|
+
* removed from CODE-category blocks (treated as prose in this simplified output).
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import Transformation from './base/transformation'
|
|
9
|
+
import ParseResult from '../models/parse-result'
|
|
10
|
+
|
|
11
|
+
export default class ToHTML extends Transformation {
|
|
12
|
+
constructor () {
|
|
13
|
+
super('To HTML', 'String')
|
|
14
|
+
}
|
|
15
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
16
|
+
parseResult.pages.forEach(page => {
|
|
17
|
+
var text = ''
|
|
18
|
+
page.items.forEach(block => {
|
|
19
|
+
// Concatenate all words in the same block, unless it's a Table of Contents block
|
|
20
|
+
let concatText
|
|
21
|
+
if (block.category === 'TOC') {
|
|
22
|
+
concatText = block.text
|
|
23
|
+
} else {
|
|
24
|
+
concatText = block.text.replace(/(\r\n|\n|\r)/gm, '\n')
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
// Concatenate words that were previously broken up by newline
|
|
28
|
+
if (block.category !== 'LIST') {
|
|
29
|
+
concatText = concatText.split('- ').join('')
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// Assume there are no code blocks in our documents
|
|
33
|
+
if (block.category === 'CODE') {
|
|
34
|
+
concatText = concatText.split('<code>').join('').split('</code>').join('')
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
text += `<p>${concatText}</p>\n\n`
|
|
38
|
+
})
|
|
39
|
+
|
|
40
|
+
page.items = [text]
|
|
41
|
+
})
|
|
42
|
+
return new ParseResult({
|
|
43
|
+
...parseResult,
|
|
44
|
+
})
|
|
45
|
+
}
|
|
46
|
+
}
|
package/src/utils/is-url-pdf.ts
CHANGED
|
@@ -1,33 +1,33 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Detects if a given URL points to a PDF file by checking
|
|
3
|
-
* the stream's first bytes for %PDF- then ends the request.
|
|
4
|
-
* Useful for hidden pdf url that does not end with pdf
|
|
5
|
-
* @category Extract
|
|
6
|
-
*/
|
|
7
|
-
import grab from "grab-url";
|
|
8
|
-
|
|
9
|
-
export async function isUrlPDF(url: string) {
|
|
10
|
-
try {
|
|
11
|
-
// Fetch the URL as an arraybuffer
|
|
12
|
-
const buffer = await grab(url, {
|
|
13
|
-
responseType: "arraybuffer",
|
|
14
|
-
timeout: 10,
|
|
15
|
-
});
|
|
16
|
-
|
|
17
|
-
if (!buffer || buffer.byteLength < 5) return false;
|
|
18
|
-
|
|
19
|
-
const chunk = new Uint8Array(buffer);
|
|
20
|
-
|
|
21
|
-
// Check if the bytes match the PDF signature
|
|
22
|
-
return (
|
|
23
|
-
chunk[0] === 0x25 && // %
|
|
24
|
-
chunk[1] === 0x50 && // P
|
|
25
|
-
chunk[2] === 0x44 && // D
|
|
26
|
-
chunk[3] === 0x46 && // F
|
|
27
|
-
chunk[4] === 0x2d
|
|
28
|
-
); // -
|
|
29
|
-
} catch (error) {
|
|
30
|
-
console.error("Error checking URL:", error);
|
|
31
|
-
return false;
|
|
32
|
-
}
|
|
33
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* Detects if a given URL points to a PDF file by checking
|
|
3
|
+
* the stream's first bytes for %PDF- then ends the request.
|
|
4
|
+
* Useful for hidden pdf url that does not end with pdf
|
|
5
|
+
* @category Extract
|
|
6
|
+
*/
|
|
7
|
+
import grab from "grab-url";
|
|
8
|
+
|
|
9
|
+
export async function isUrlPDF(url: string) {
|
|
10
|
+
try {
|
|
11
|
+
// Fetch the URL as an arraybuffer
|
|
12
|
+
const buffer = await grab(url, {
|
|
13
|
+
responseType: "arraybuffer",
|
|
14
|
+
timeout: 10,
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
if (!buffer || buffer.byteLength < 5) return false;
|
|
18
|
+
|
|
19
|
+
const chunk = new Uint8Array(buffer);
|
|
20
|
+
|
|
21
|
+
// Check if the bytes match the PDF signature
|
|
22
|
+
return (
|
|
23
|
+
chunk[0] === 0x25 && // %
|
|
24
|
+
chunk[1] === 0x50 && // P
|
|
25
|
+
chunk[2] === 0x44 && // D
|
|
26
|
+
chunk[3] === 0x46 && // F
|
|
27
|
+
chunk[4] === 0x2d
|
|
28
|
+
); // -
|
|
29
|
+
} catch (error) {
|
|
30
|
+
console.error("Error checking URL:", error);
|
|
31
|
+
return false;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
@@ -1,35 +1,35 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Geometry utility functions shared across the transformation pipeline.
|
|
3
|
-
* `minXFromBlocks` finds the minimum x across all items inside an array of blocks;
|
|
4
|
-
* `minXFromPageItems` finds the minimum x across a flat page item array; `sortByX`
|
|
5
|
-
* sorts items in place by x coordinate for left-to-right reading order.
|
|
6
|
-
*/
|
|
7
|
-
import LineItemBlock from '../models/line-item-block'
|
|
8
|
-
|
|
9
|
-
export function minXFromBlocks(blocks: LineItemBlock[]): number | null {
|
|
10
|
-
var minX = 999
|
|
11
|
-
blocks.forEach(block => {
|
|
12
|
-
block.items.forEach(item => {
|
|
13
|
-
minX = Math.min(minX, item.x)
|
|
14
|
-
})
|
|
15
|
-
})
|
|
16
|
-
if (minX === 999) {
|
|
17
|
-
return null
|
|
18
|
-
}
|
|
19
|
-
return minX
|
|
20
|
-
}
|
|
21
|
-
|
|
22
|
-
export function minXFromPageItems(items: Array<{ x: number }>): number | null {
|
|
23
|
-
var minX = 999
|
|
24
|
-
items.forEach(item => {
|
|
25
|
-
minX = Math.min(minX, item.x)
|
|
26
|
-
})
|
|
27
|
-
if (minX === 999) {
|
|
28
|
-
return null
|
|
29
|
-
}
|
|
30
|
-
return minX
|
|
31
|
-
}
|
|
32
|
-
|
|
33
|
-
export function sortByX(items: Array<{ x: number }>): void {
|
|
34
|
-
items.sort((a, b) => a.x - b.x)
|
|
35
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Geometry utility functions shared across the transformation pipeline.
|
|
3
|
+
* `minXFromBlocks` finds the minimum x across all items inside an array of blocks;
|
|
4
|
+
* `minXFromPageItems` finds the minimum x across a flat page item array; `sortByX`
|
|
5
|
+
* sorts items in place by x coordinate for left-to-right reading order.
|
|
6
|
+
*/
|
|
7
|
+
import LineItemBlock from '../models/line-item-block'
|
|
8
|
+
|
|
9
|
+
export function minXFromBlocks(blocks: LineItemBlock[]): number | null {
|
|
10
|
+
var minX = 999
|
|
11
|
+
blocks.forEach(block => {
|
|
12
|
+
block.items.forEach(item => {
|
|
13
|
+
minX = Math.min(minX, item.x)
|
|
14
|
+
})
|
|
15
|
+
})
|
|
16
|
+
if (minX === 999) {
|
|
17
|
+
return null
|
|
18
|
+
}
|
|
19
|
+
return minX
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export function minXFromPageItems(items: Array<{ x: number }>): number | null {
|
|
23
|
+
var minX = 999
|
|
24
|
+
items.forEach(item => {
|
|
25
|
+
minX = Math.min(minX, item.x)
|
|
26
|
+
})
|
|
27
|
+
if (minX === 999) {
|
|
28
|
+
return null
|
|
29
|
+
}
|
|
30
|
+
return minX
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function sortByX(items: Array<{ x: number }>): void {
|
|
34
|
+
items.sort((a, b) => a.x - b.x)
|
|
35
|
+
}
|