extract-pdf 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -10
- package/dist/models/annotation.d.ts +20 -0
- package/dist/models/block-type.d.ts +10 -0
- package/dist/models/headline-finder.d.ts +11 -0
- package/dist/models/line-converter.d.ts +10 -0
- package/dist/models/line-item-block.d.ts +14 -0
- package/dist/models/line-item.d.ts +25 -0
- package/dist/models/metadata.d.ts +23 -0
- package/dist/models/page-item.d.ts +21 -0
- package/dist/models/page.d.ts +14 -0
- package/dist/models/parse-result.d.ts +24 -0
- package/dist/models/parsed-elements.d.ts +20 -0
- package/dist/models/stashing-stream.d.ts +23 -0
- package/dist/models/text-item-line-grouper.d.ts +8 -0
- package/dist/models/text-item.d.ts +29 -0
- package/dist/models/word.d.ts +27 -0
- package/dist/pdf-to-html.cjs.js +1 -1
- package/dist/pdf-to-html.d.ts +36 -41
- package/dist/pdf-to-html.es.js +1 -1
- package/dist/transforms/base/to-line-item-block-transform.d.ts +6 -0
- package/dist/transforms/base/to-line-item-transform.d.ts +6 -0
- package/dist/transforms/base/to-text-item-transform.d.ts +6 -0
- package/dist/transforms/base/transformation.d.ts +8 -0
- package/dist/transforms/block/detect-code-quote-blocks.d.ts +6 -0
- package/dist/transforms/block/detect-list-levels.d.ts +6 -0
- package/dist/transforms/block/gather-blocks.d.ts +6 -0
- package/dist/transforms/calculate-global-stats.d.ts +11 -0
- package/dist/transforms/line-item/compact-lines.d.ts +6 -0
- package/dist/transforms/line-item/detect-headers.d.ts +6 -0
- package/dist/transforms/line-item/detect-list-items.d.ts +6 -0
- package/dist/transforms/line-item/detect-toc.d.ts +6 -0
- package/dist/transforms/line-item/remove-repetitive-elements.d.ts +6 -0
- package/dist/transforms/line-item/vertical-to-horizontal.d.ts +6 -0
- package/dist/transforms/to-html.d.ts +6 -0
- package/dist/transforms/to-text-blocks.d.ts +6 -0
- package/dist/utils/is-url-pdf.d.ts +1 -0
- package/dist/utils/page-item-functions.d.ts +8 -0
- package/dist/utils/page-number-functions.d.ts +14 -0
- package/dist/utils/string-functions.d.ts +14 -0
- package/package.json +9 -12
- package/src/models/annotation.ts +41 -0
- package/src/models/block-type.ts +203 -0
- package/src/models/headline-finder.ts +53 -0
- package/src/models/line-converter.ts +224 -0
- package/src/models/line-item-block.ts +51 -0
- package/src/models/line-item.ts +59 -0
- package/src/models/metadata.ts +29 -0
- package/src/models/page-item.ts +36 -0
- package/src/models/page.ts +16 -0
- package/src/models/parse-result.ts +32 -0
- package/src/models/parsed-elements.ts +29 -0
- package/src/models/stashing-stream.ts +86 -0
- package/src/models/text-item-line-grouper.ts +41 -0
- package/src/models/text-item.ts +50 -0
- package/src/models/word.ts +31 -0
- package/src/pdf-to-html.ts +225 -0
- package/src/transforms/base/to-line-item-block-transform.ts +29 -0
- package/src/transforms/base/to-line-item-transform.ts +29 -0
- package/src/transforms/base/to-text-item-transform.ts +28 -0
- package/src/transforms/base/transformation.ts +35 -0
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -0
- package/src/transforms/block/detect-list-levels.ts +64 -0
- package/src/transforms/block/gather-blocks.ts +113 -0
- package/src/transforms/calculate-global-stats.ts +132 -0
- package/src/transforms/line-item/compact-lines.ts +92 -0
- package/src/transforms/line-item/detect-headers.ts +173 -0
- package/src/transforms/line-item/detect-list-items.ts +68 -0
- package/src/transforms/line-item/detect-toc.ts +459 -0
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -0
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -0
- package/src/transforms/to-html.ts +46 -0
- package/src/transforms/to-text-blocks.ts +38 -0
- package/src/utils/is-url-pdf.ts +33 -0
- package/src/utils/page-item-functions.ts +35 -0
- package/src/utils/page-number-functions.ts +109 -0
- package/src/utils/string-functions.ts +124 -0
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Flattens each page's parsed block items into plain text objects,
|
|
3
|
+
* replacing structured PDF block nodes with `{ category, text }` pairs. The
|
|
4
|
+
* `category` is derived from the block's `BlockType` name (e.g. "Paragraph",
|
|
5
|
+
* "Heading", "List") or falls back to "Unknown" when no type is assigned.
|
|
6
|
+
* `text` is extracted via `BlockType.blockToText`, which concatenates all
|
|
7
|
+
* inline text spans within the block. The result is a new `ParseResult` whose
|
|
8
|
+
* pages contain only these lightweight text items, discarding geometry and
|
|
9
|
+
* style metadata — making downstream NLP processing and serialization simpler.
|
|
10
|
+
*/
|
|
11
|
+
import Transformation from "./base/transformation";
|
|
12
|
+
import ParseResult from "../models/parse-result";
|
|
13
|
+
import BlockType from "../models/block-type";
|
|
14
|
+
|
|
15
|
+
export default class ToTextBlocks extends Transformation {
|
|
16
|
+
constructor() {
|
|
17
|
+
super("To Text Blocks", "TextBlock");
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
21
|
+
parseResult.pages.forEach((page) => {
|
|
22
|
+
const textItems: Array<{ category: string; text: string }> = [];
|
|
23
|
+
page.items.forEach((block) => {
|
|
24
|
+
// TODO category to type (before have no unknowns, have paragraph)
|
|
25
|
+
const category = block.type ? block.type.name : "Unknown";
|
|
26
|
+
textItems.push({
|
|
27
|
+
category: category,
|
|
28
|
+
text: BlockType.blockToText(block),
|
|
29
|
+
});
|
|
30
|
+
});
|
|
31
|
+
page.items = textItems;
|
|
32
|
+
});
|
|
33
|
+
return new ParseResult({
|
|
34
|
+
...parseResult,
|
|
35
|
+
});
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Detects if a given URL points to a PDF file by checking
|
|
3
|
+
* the stream's first bytes for %PDF- then ends the request.
|
|
4
|
+
* Useful for hidden pdf url that does not end with pdf
|
|
5
|
+
* @category Extract
|
|
6
|
+
*/
|
|
7
|
+
import grab from "grab-url";
|
|
8
|
+
|
|
9
|
+
export async function isUrlPDF(url: string) {
|
|
10
|
+
try {
|
|
11
|
+
// Fetch the URL as an arraybuffer
|
|
12
|
+
const buffer = await grab(url, {
|
|
13
|
+
responseType: "arraybuffer",
|
|
14
|
+
timeout: 10,
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
if (!buffer || buffer.byteLength < 5) return false;
|
|
18
|
+
|
|
19
|
+
const chunk = new Uint8Array(buffer);
|
|
20
|
+
|
|
21
|
+
// Check if the bytes match the PDF signature
|
|
22
|
+
return (
|
|
23
|
+
chunk[0] === 0x25 && // %
|
|
24
|
+
chunk[1] === 0x50 && // P
|
|
25
|
+
chunk[2] === 0x44 && // D
|
|
26
|
+
chunk[3] === 0x46 && // F
|
|
27
|
+
chunk[4] === 0x2d
|
|
28
|
+
); // -
|
|
29
|
+
} catch (error) {
|
|
30
|
+
console.error("Error checking URL:", error);
|
|
31
|
+
return false;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Geometry utility functions shared across the transformation pipeline.
|
|
3
|
+
* `minXFromBlocks` finds the minimum x across all items inside an array of blocks;
|
|
4
|
+
* `minXFromPageItems` finds the minimum x across a flat page item array; `sortByX`
|
|
5
|
+
* sorts items in place by x coordinate for left-to-right reading order.
|
|
6
|
+
*/
|
|
7
|
+
import LineItemBlock from '../models/line-item-block'
|
|
8
|
+
|
|
9
|
+
export function minXFromBlocks(blocks: LineItemBlock[]): number | null {
|
|
10
|
+
var minX = 999
|
|
11
|
+
blocks.forEach(block => {
|
|
12
|
+
block.items.forEach(item => {
|
|
13
|
+
minX = Math.min(minX, item.x)
|
|
14
|
+
})
|
|
15
|
+
})
|
|
16
|
+
if (minX === 999) {
|
|
17
|
+
return null
|
|
18
|
+
}
|
|
19
|
+
return minX
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export function minXFromPageItems(items: Array<{ x: number }>): number | null {
|
|
23
|
+
var minX = 999
|
|
24
|
+
items.forEach(item => {
|
|
25
|
+
minX = Math.min(minX, item.x)
|
|
26
|
+
})
|
|
27
|
+
if (minX === 999) {
|
|
28
|
+
return null
|
|
29
|
+
}
|
|
30
|
+
return minX
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function sortByX(items: Array<{ x: number }>): void {
|
|
34
|
+
items.sort((a, b) => a.x - b.x)
|
|
35
|
+
}
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Detects and strips printed page numbers from PDF text content.
|
|
3
|
+
* Scans the top and bottom sixths of each page's item list for standalone numeric
|
|
4
|
+
* strings, builds a `pageIndex → pageNum` map, finds the first page where numbers
|
|
5
|
+
* begin incrementing consecutively, and provides `removePageNumber` to filter
|
|
6
|
+
* those items out before further processing.
|
|
7
|
+
*/
|
|
8
|
+
import {
|
|
9
|
+
removeLeadingWhitespaces,
|
|
10
|
+
removeTrailingWhitespaces,
|
|
11
|
+
isNumber,
|
|
12
|
+
} from "./string-functions";
|
|
13
|
+
|
|
14
|
+
interface TextContentItem {
|
|
15
|
+
str: string;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
interface TextContent {
|
|
19
|
+
items: TextContentItem[];
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
type PageIndexNumMap = Record<string, number[]>;
|
|
23
|
+
|
|
24
|
+
const searchRange = (numerator: number, denominator: number, length: number): number => {
|
|
25
|
+
return Math.floor((numerator / denominator) * length);
|
|
26
|
+
};
|
|
27
|
+
|
|
28
|
+
const searchArea = (range: TextContentItem[], pageIndexNumMap: PageIndexNumMap, pageIndex: number): PageIndexNumMap => {
|
|
29
|
+
for (const { str } of range) {
|
|
30
|
+
const trimLeadingWhitespaces = removeLeadingWhitespaces(str);
|
|
31
|
+
const trimWhitespaces = removeTrailingWhitespaces(trimLeadingWhitespaces);
|
|
32
|
+
if (isNumber(trimWhitespaces)) {
|
|
33
|
+
if (!pageIndexNumMap[pageIndex]) {
|
|
34
|
+
pageIndexNumMap[pageIndex] = [];
|
|
35
|
+
}
|
|
36
|
+
pageIndexNumMap[pageIndex].push(Number(trimWhitespaces));
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
return pageIndexNumMap;
|
|
40
|
+
};
|
|
41
|
+
|
|
42
|
+
const findPageNumbers = (pageIndexNumMap: PageIndexNumMap, pageIndex: number, items: TextContentItem[]): PageIndexNumMap => {
|
|
43
|
+
const topArea = searchRange(1, 6, items.length);
|
|
44
|
+
const bottomArea = searchRange(5, 6, items.length);
|
|
45
|
+
|
|
46
|
+
const topAreaResult = searchArea(
|
|
47
|
+
items.slice(0, topArea),
|
|
48
|
+
pageIndexNumMap,
|
|
49
|
+
pageIndex,
|
|
50
|
+
);
|
|
51
|
+
return searchArea(items.slice(bottomArea), topAreaResult, pageIndex);
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
const findFirstPage = (pageIndexNumMap: PageIndexNumMap): { pageIndex: number; pageNum: number } | undefined => {
|
|
55
|
+
let counter = 0;
|
|
56
|
+
const keys = Object.keys(pageIndexNumMap);
|
|
57
|
+
if (keys.length === 0 || keys.length === 1) {
|
|
58
|
+
return;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
for (let x = 0; x < keys.length - 1; x++) {
|
|
62
|
+
const firstPage = pageIndexNumMap[keys[x]];
|
|
63
|
+
const secondPage = pageIndexNumMap[keys[x + 1]];
|
|
64
|
+
const prevCounter = counter;
|
|
65
|
+
|
|
66
|
+
for (let y = 0; y < firstPage.length && counter < 2; y++) {
|
|
67
|
+
for (let z = 0; z < secondPage.length && counter < 2; z++) {
|
|
68
|
+
const pageDifference = Number(keys[x + 1]) - Number(keys[x]);
|
|
69
|
+
if (firstPage[y] + 1 === secondPage[z]) {
|
|
70
|
+
counter++;
|
|
71
|
+
} else if (
|
|
72
|
+
pageDifference > 1 &&
|
|
73
|
+
firstPage[y] + pageDifference === secondPage[z]
|
|
74
|
+
) {
|
|
75
|
+
counter++;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
let pageDetails =
|
|
81
|
+
x > 0
|
|
82
|
+
? Object.entries(pageIndexNumMap)[x - 1]
|
|
83
|
+
: Object.entries(pageIndexNumMap)[x];
|
|
84
|
+
if (prevCounter === counter) {
|
|
85
|
+
counter = 0;
|
|
86
|
+
pageDetails = Object.entries(pageIndexNumMap)[x];
|
|
87
|
+
} else if (counter >= 2) {
|
|
88
|
+
return { pageIndex: Number(pageDetails[0]), pageNum: pageDetails[1][0] };
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
const removePageNumber = (textContent: TextContent, pageNum: number): TextContent => {
|
|
94
|
+
const filteredContent = { items: [...textContent.items] };
|
|
95
|
+
const topArea = searchRange(1, 6, filteredContent.items.length);
|
|
96
|
+
const bottomArea = searchRange(5, 6, filteredContent.items.length);
|
|
97
|
+
|
|
98
|
+
filteredContent.items = filteredContent.items.filter((item, index) => {
|
|
99
|
+
const isAtTop = index > 0 && index < topArea;
|
|
100
|
+
const isAtBottom =
|
|
101
|
+
index > bottomArea && index < filteredContent.items.length;
|
|
102
|
+
|
|
103
|
+
return isAtTop || isAtBottom ? Number(item.str) !== Number(pageNum) : item;
|
|
104
|
+
});
|
|
105
|
+
return filteredContent;
|
|
106
|
+
};
|
|
107
|
+
|
|
108
|
+
export { findPageNumbers, findFirstPage, removePageNumber };
|
|
109
|
+
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Character-level string utilities for the PDF text pipeline:
|
|
3
|
+
* digit and number detection, leading/trailing whitespace trimming, list-item
|
|
4
|
+
* character and pattern matching (bullet chars, numbered items), word-overlap
|
|
5
|
+
* scoring (`wordMatch`), char-code normalisation for headline matching
|
|
6
|
+
* (`normalizedCharCodeArray`), and camel-case detection
|
|
7
|
+
* (`hasUpperCaseCharacterInMiddleOfWord`).
|
|
8
|
+
*/
|
|
9
|
+
const MIN_DIGIT_CHAR_CODE = 48
|
|
10
|
+
const MAX_DIGIT_CHAR_CODE = 57
|
|
11
|
+
const WHITESPACE_CHAR_CODE = 32
|
|
12
|
+
const TAB_CHAR_CODE = 9
|
|
13
|
+
const DOT_CHAR_CODE = 46
|
|
14
|
+
|
|
15
|
+
export function removeLeadingWhitespaces(string: string): string {
|
|
16
|
+
while (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
|
|
17
|
+
string = string.substring(1, string.length)
|
|
18
|
+
}
|
|
19
|
+
return string
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export function removeTrailingWhitespaces(string: string): string {
|
|
23
|
+
while (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
|
|
24
|
+
string = string.substring(0, string.length - 1)
|
|
25
|
+
}
|
|
26
|
+
return string
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export function isDigit(charCode: number): boolean {
|
|
30
|
+
return charCode >= MIN_DIGIT_CHAR_CODE && charCode <= MAX_DIGIT_CHAR_CODE
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function isNumber(string: string): boolean {
|
|
34
|
+
for (var i = 0; i < string.length; i++) {
|
|
35
|
+
const charCode = string.charCodeAt(i)
|
|
36
|
+
if (!isDigit(charCode)) {
|
|
37
|
+
return false
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
return true
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export function hasOnly(string: string, char: string): boolean {
|
|
44
|
+
const charCode = char.charCodeAt(0)
|
|
45
|
+
for (var i = 0; i < string.length; i++) {
|
|
46
|
+
const aCharCode = string.charCodeAt(i)
|
|
47
|
+
if (aCharCode !== charCode) {
|
|
48
|
+
return false
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
return true
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function hasUpperCaseCharacterInMiddleOfWord(text: string): boolean {
|
|
55
|
+
var beginningOfWord = true
|
|
56
|
+
for (var i = 0; i < text.length; i++) {
|
|
57
|
+
const character = text.charAt(i)
|
|
58
|
+
if (character === ' ') {
|
|
59
|
+
beginningOfWord = true
|
|
60
|
+
} else {
|
|
61
|
+
if (!beginningOfWord && isNaN(character as any * 1) && character === character.toUpperCase() && character.toUpperCase() !== character.toLowerCase()) {
|
|
62
|
+
return true
|
|
63
|
+
}
|
|
64
|
+
beginningOfWord = false
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
return false
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// Remove whitespace/dots + to uppercase
|
|
71
|
+
export function normalizedCharCodeArray(string: string): number[] {
|
|
72
|
+
string = string.toUpperCase()
|
|
73
|
+
return charCodeArray(string).filter(charCode => charCode !== WHITESPACE_CHAR_CODE && charCode !== TAB_CHAR_CODE && charCode !== DOT_CHAR_CODE)
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export function charCodeArray(string: string): number[] {
|
|
77
|
+
const charCodes: number[] = []
|
|
78
|
+
for (var i = 0; i < string.length; i++) {
|
|
79
|
+
charCodes.push(string.charCodeAt(i))
|
|
80
|
+
}
|
|
81
|
+
return charCodes
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
export function prefixAfterWhitespace(prefix: string, string: string): string {
|
|
85
|
+
if (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
|
|
86
|
+
string = removeLeadingWhitespaces(string)
|
|
87
|
+
return ' ' + prefix + string
|
|
88
|
+
} else {
|
|
89
|
+
return prefix + string
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function suffixBeforeWhitespace(string: string, suffix: string): string {
|
|
94
|
+
if (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
|
|
95
|
+
string = removeTrailingWhitespaces(string)
|
|
96
|
+
return string + suffix + ' '
|
|
97
|
+
} else {
|
|
98
|
+
return string + suffix
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export function isListItemCharacter(string: string): boolean {
|
|
103
|
+
if (string.length > 1) {
|
|
104
|
+
return false
|
|
105
|
+
}
|
|
106
|
+
const char = string.charAt(0)
|
|
107
|
+
return char === '-' || char === '•' || char === '–'
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
export function isListItem(string: string): boolean {
|
|
111
|
+
return /^[\s]*[-•–][\s].*$/g.test(string)
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
export function isNumberedListItem(string: string): boolean {
|
|
115
|
+
return /^[\s]*[\d]*[.][\s].*$/g.test(string)
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export function wordMatch(string1: string, string2: string): number {
|
|
119
|
+
const words1 = new Set(string1.toUpperCase().split(' '))
|
|
120
|
+
const words2 = new Set(string2.toUpperCase().split(' '))
|
|
121
|
+
const intersection = new Set(
|
|
122
|
+
[...words1].filter(x => words2.has(x)))
|
|
123
|
+
return intersection.size / Math.max(words1.size, words2.size)
|
|
124
|
+
}
|