extract-pdf 0.1.20 → 0.1.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -74
- package/package.json +1 -1
- package/src/models/annotation.ts +41 -41
- package/src/models/block-type.ts +203 -203
- package/src/models/line-converter.ts +224 -224
- package/src/models/metadata.ts +29 -29
- package/src/models/page.ts +16 -16
- package/src/models/parse-result.ts +32 -32
- package/src/models/parsed-elements.ts +29 -29
- package/src/models/stashing-stream.ts +86 -86
- package/src/models/text-item-line-grouper.ts +41 -41
- package/src/models/word.ts +31 -31
- package/src/pdf-to-html.ts +225 -225
- package/src/transforms/base/to-line-item-block-transform.ts +29 -29
- package/src/transforms/base/to-line-item-transform.ts +29 -29
- package/src/transforms/base/to-text-item-transform.ts +28 -28
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
- package/src/transforms/block/detect-list-levels.ts +64 -64
- package/src/transforms/block/gather-blocks.ts +113 -113
- package/src/transforms/calculate-global-stats.ts +132 -132
- package/src/transforms/line-item/compact-lines.ts +92 -92
- package/src/transforms/line-item/detect-headers.ts +173 -173
- package/src/transforms/line-item/detect-list-items.ts +68 -68
- package/src/transforms/line-item/detect-toc.ts +459 -459
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
- package/src/transforms/to-html.ts +46 -46
- package/src/utils/is-url-pdf.ts +33 -33
- package/src/utils/page-item-functions.ts +35 -35
- package/src/utils/string-functions.ts +124 -124
|
@@ -1,124 +1,124 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Character-level string utilities for the PDF text pipeline:
|
|
3
|
-
* digit and number detection, leading/trailing whitespace trimming, list-item
|
|
4
|
-
* character and pattern matching (bullet chars, numbered items), word-overlap
|
|
5
|
-
* scoring (`wordMatch`), char-code normalisation for headline matching
|
|
6
|
-
* (`normalizedCharCodeArray`), and camel-case detection
|
|
7
|
-
* (`hasUpperCaseCharacterInMiddleOfWord`).
|
|
8
|
-
*/
|
|
9
|
-
const MIN_DIGIT_CHAR_CODE = 48
|
|
10
|
-
const MAX_DIGIT_CHAR_CODE = 57
|
|
11
|
-
const WHITESPACE_CHAR_CODE = 32
|
|
12
|
-
const TAB_CHAR_CODE = 9
|
|
13
|
-
const DOT_CHAR_CODE = 46
|
|
14
|
-
|
|
15
|
-
export function removeLeadingWhitespaces(string: string): string {
|
|
16
|
-
while (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
|
|
17
|
-
string = string.substring(1, string.length)
|
|
18
|
-
}
|
|
19
|
-
return string
|
|
20
|
-
}
|
|
21
|
-
|
|
22
|
-
export function removeTrailingWhitespaces(string: string): string {
|
|
23
|
-
while (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
|
|
24
|
-
string = string.substring(0, string.length - 1)
|
|
25
|
-
}
|
|
26
|
-
return string
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
export function isDigit(charCode: number): boolean {
|
|
30
|
-
return charCode >= MIN_DIGIT_CHAR_CODE && charCode <= MAX_DIGIT_CHAR_CODE
|
|
31
|
-
}
|
|
32
|
-
|
|
33
|
-
export function isNumber(string: string): boolean {
|
|
34
|
-
for (var i = 0; i < string.length; i++) {
|
|
35
|
-
const charCode = string.charCodeAt(i)
|
|
36
|
-
if (!isDigit(charCode)) {
|
|
37
|
-
return false
|
|
38
|
-
}
|
|
39
|
-
}
|
|
40
|
-
return true
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
export function hasOnly(string: string, char: string): boolean {
|
|
44
|
-
const charCode = char.charCodeAt(0)
|
|
45
|
-
for (var i = 0; i < string.length; i++) {
|
|
46
|
-
const aCharCode = string.charCodeAt(i)
|
|
47
|
-
if (aCharCode !== charCode) {
|
|
48
|
-
return false
|
|
49
|
-
}
|
|
50
|
-
}
|
|
51
|
-
return true
|
|
52
|
-
}
|
|
53
|
-
|
|
54
|
-
export function hasUpperCaseCharacterInMiddleOfWord(text: string): boolean {
|
|
55
|
-
var beginningOfWord = true
|
|
56
|
-
for (var i = 0; i < text.length; i++) {
|
|
57
|
-
const character = text.charAt(i)
|
|
58
|
-
if (character === ' ') {
|
|
59
|
-
beginningOfWord = true
|
|
60
|
-
} else {
|
|
61
|
-
if (!beginningOfWord && isNaN(character as any * 1) && character === character.toUpperCase() && character.toUpperCase() !== character.toLowerCase()) {
|
|
62
|
-
return true
|
|
63
|
-
}
|
|
64
|
-
beginningOfWord = false
|
|
65
|
-
}
|
|
66
|
-
}
|
|
67
|
-
return false
|
|
68
|
-
}
|
|
69
|
-
|
|
70
|
-
// Remove whitespace/dots + to uppercase
|
|
71
|
-
export function normalizedCharCodeArray(string: string): number[] {
|
|
72
|
-
string = string.toUpperCase()
|
|
73
|
-
return charCodeArray(string).filter(charCode => charCode !== WHITESPACE_CHAR_CODE && charCode !== TAB_CHAR_CODE && charCode !== DOT_CHAR_CODE)
|
|
74
|
-
}
|
|
75
|
-
|
|
76
|
-
export function charCodeArray(string: string): number[] {
|
|
77
|
-
const charCodes: number[] = []
|
|
78
|
-
for (var i = 0; i < string.length; i++) {
|
|
79
|
-
charCodes.push(string.charCodeAt(i))
|
|
80
|
-
}
|
|
81
|
-
return charCodes
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
export function prefixAfterWhitespace(prefix: string, string: string): string {
|
|
85
|
-
if (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
|
|
86
|
-
string = removeLeadingWhitespaces(string)
|
|
87
|
-
return ' ' + prefix + string
|
|
88
|
-
} else {
|
|
89
|
-
return prefix + string
|
|
90
|
-
}
|
|
91
|
-
}
|
|
92
|
-
|
|
93
|
-
export function suffixBeforeWhitespace(string: string, suffix: string): string {
|
|
94
|
-
if (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
|
|
95
|
-
string = removeTrailingWhitespaces(string)
|
|
96
|
-
return string + suffix + ' '
|
|
97
|
-
} else {
|
|
98
|
-
return string + suffix
|
|
99
|
-
}
|
|
100
|
-
}
|
|
101
|
-
|
|
102
|
-
export function isListItemCharacter(string: string): boolean {
|
|
103
|
-
if (string.length > 1) {
|
|
104
|
-
return false
|
|
105
|
-
}
|
|
106
|
-
const char = string.charAt(0)
|
|
107
|
-
return char === '-' || char === '•' || char === '–'
|
|
108
|
-
}
|
|
109
|
-
|
|
110
|
-
export function isListItem(string: string): boolean {
|
|
111
|
-
return /^[\s]*[-•–][\s].*$/g.test(string)
|
|
112
|
-
}
|
|
113
|
-
|
|
114
|
-
export function isNumberedListItem(string: string): boolean {
|
|
115
|
-
return /^[\s]*[\d]*[.][\s].*$/g.test(string)
|
|
116
|
-
}
|
|
117
|
-
|
|
118
|
-
export function wordMatch(string1: string, string2: string): number {
|
|
119
|
-
const words1 = new Set(string1.toUpperCase().split(' '))
|
|
120
|
-
const words2 = new Set(string2.toUpperCase().split(' '))
|
|
121
|
-
const intersection = new Set(
|
|
122
|
-
[...words1].filter(x => words2.has(x)))
|
|
123
|
-
return intersection.size / Math.max(words1.size, words2.size)
|
|
124
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Character-level string utilities for the PDF text pipeline:
|
|
3
|
+
* digit and number detection, leading/trailing whitespace trimming, list-item
|
|
4
|
+
* character and pattern matching (bullet chars, numbered items), word-overlap
|
|
5
|
+
* scoring (`wordMatch`), char-code normalisation for headline matching
|
|
6
|
+
* (`normalizedCharCodeArray`), and camel-case detection
|
|
7
|
+
* (`hasUpperCaseCharacterInMiddleOfWord`).
|
|
8
|
+
*/
|
|
9
|
+
const MIN_DIGIT_CHAR_CODE = 48
|
|
10
|
+
const MAX_DIGIT_CHAR_CODE = 57
|
|
11
|
+
const WHITESPACE_CHAR_CODE = 32
|
|
12
|
+
const TAB_CHAR_CODE = 9
|
|
13
|
+
const DOT_CHAR_CODE = 46
|
|
14
|
+
|
|
15
|
+
export function removeLeadingWhitespaces(string: string): string {
|
|
16
|
+
while (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
|
|
17
|
+
string = string.substring(1, string.length)
|
|
18
|
+
}
|
|
19
|
+
return string
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
export function removeTrailingWhitespaces(string: string): string {
|
|
23
|
+
while (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
|
|
24
|
+
string = string.substring(0, string.length - 1)
|
|
25
|
+
}
|
|
26
|
+
return string
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export function isDigit(charCode: number): boolean {
|
|
30
|
+
return charCode >= MIN_DIGIT_CHAR_CODE && charCode <= MAX_DIGIT_CHAR_CODE
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function isNumber(string: string): boolean {
|
|
34
|
+
for (var i = 0; i < string.length; i++) {
|
|
35
|
+
const charCode = string.charCodeAt(i)
|
|
36
|
+
if (!isDigit(charCode)) {
|
|
37
|
+
return false
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
return true
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export function hasOnly(string: string, char: string): boolean {
|
|
44
|
+
const charCode = char.charCodeAt(0)
|
|
45
|
+
for (var i = 0; i < string.length; i++) {
|
|
46
|
+
const aCharCode = string.charCodeAt(i)
|
|
47
|
+
if (aCharCode !== charCode) {
|
|
48
|
+
return false
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
return true
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export function hasUpperCaseCharacterInMiddleOfWord(text: string): boolean {
|
|
55
|
+
var beginningOfWord = true
|
|
56
|
+
for (var i = 0; i < text.length; i++) {
|
|
57
|
+
const character = text.charAt(i)
|
|
58
|
+
if (character === ' ') {
|
|
59
|
+
beginningOfWord = true
|
|
60
|
+
} else {
|
|
61
|
+
if (!beginningOfWord && isNaN(character as any * 1) && character === character.toUpperCase() && character.toUpperCase() !== character.toLowerCase()) {
|
|
62
|
+
return true
|
|
63
|
+
}
|
|
64
|
+
beginningOfWord = false
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
return false
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// Remove whitespace/dots + to uppercase
|
|
71
|
+
export function normalizedCharCodeArray(string: string): number[] {
|
|
72
|
+
string = string.toUpperCase()
|
|
73
|
+
return charCodeArray(string).filter(charCode => charCode !== WHITESPACE_CHAR_CODE && charCode !== TAB_CHAR_CODE && charCode !== DOT_CHAR_CODE)
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export function charCodeArray(string: string): number[] {
|
|
77
|
+
const charCodes: number[] = []
|
|
78
|
+
for (var i = 0; i < string.length; i++) {
|
|
79
|
+
charCodes.push(string.charCodeAt(i))
|
|
80
|
+
}
|
|
81
|
+
return charCodes
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
export function prefixAfterWhitespace(prefix: string, string: string): string {
|
|
85
|
+
if (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
|
|
86
|
+
string = removeLeadingWhitespaces(string)
|
|
87
|
+
return ' ' + prefix + string
|
|
88
|
+
} else {
|
|
89
|
+
return prefix + string
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function suffixBeforeWhitespace(string: string, suffix: string): string {
|
|
94
|
+
if (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
|
|
95
|
+
string = removeTrailingWhitespaces(string)
|
|
96
|
+
return string + suffix + ' '
|
|
97
|
+
} else {
|
|
98
|
+
return string + suffix
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export function isListItemCharacter(string: string): boolean {
|
|
103
|
+
if (string.length > 1) {
|
|
104
|
+
return false
|
|
105
|
+
}
|
|
106
|
+
const char = string.charAt(0)
|
|
107
|
+
return char === '-' || char === '•' || char === '–'
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
export function isListItem(string: string): boolean {
|
|
111
|
+
return /^[\s]*[-•–][\s].*$/g.test(string)
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
export function isNumberedListItem(string: string): boolean {
|
|
115
|
+
return /^[\s]*[\d]*[.][\s].*$/g.test(string)
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export function wordMatch(string1: string, string2: string): number {
|
|
119
|
+
const words1 = new Set(string1.toUpperCase().split(' '))
|
|
120
|
+
const words2 = new Set(string2.toUpperCase().split(' '))
|
|
121
|
+
const intersection = new Set(
|
|
122
|
+
[...words1].filter(x => words2.has(x)))
|
|
123
|
+
return intersection.size / Math.max(words1.size, words2.size)
|
|
124
|
+
}
|