extract-pdf 0.1.20 → 0.1.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/README.md +74 -74
  2. package/package.json +1 -1
  3. package/src/models/annotation.ts +41 -41
  4. package/src/models/block-type.ts +203 -203
  5. package/src/models/line-converter.ts +224 -224
  6. package/src/models/metadata.ts +29 -29
  7. package/src/models/page.ts +16 -16
  8. package/src/models/parse-result.ts +32 -32
  9. package/src/models/parsed-elements.ts +29 -29
  10. package/src/models/stashing-stream.ts +86 -86
  11. package/src/models/text-item-line-grouper.ts +41 -41
  12. package/src/models/word.ts +31 -31
  13. package/src/pdf-to-html.ts +225 -225
  14. package/src/transforms/base/to-line-item-block-transform.ts +29 -29
  15. package/src/transforms/base/to-line-item-transform.ts +29 -29
  16. package/src/transforms/base/to-text-item-transform.ts +28 -28
  17. package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
  18. package/src/transforms/block/detect-list-levels.ts +64 -64
  19. package/src/transforms/block/gather-blocks.ts +113 -113
  20. package/src/transforms/calculate-global-stats.ts +132 -132
  21. package/src/transforms/line-item/compact-lines.ts +92 -92
  22. package/src/transforms/line-item/detect-headers.ts +173 -173
  23. package/src/transforms/line-item/detect-list-items.ts +68 -68
  24. package/src/transforms/line-item/detect-toc.ts +459 -459
  25. package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
  26. package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
  27. package/src/transforms/to-html.ts +46 -46
  28. package/src/utils/is-url-pdf.ts +33 -33
  29. package/src/utils/page-item-functions.ts +35 -35
  30. package/src/utils/string-functions.ts +124 -124
@@ -1,124 +1,124 @@
1
- /**
2
- * @description Character-level string utilities for the PDF text pipeline:
3
- * digit and number detection, leading/trailing whitespace trimming, list-item
4
- * character and pattern matching (bullet chars, numbered items), word-overlap
5
- * scoring (`wordMatch`), char-code normalisation for headline matching
6
- * (`normalizedCharCodeArray`), and camel-case detection
7
- * (`hasUpperCaseCharacterInMiddleOfWord`).
8
- */
9
- const MIN_DIGIT_CHAR_CODE = 48
10
- const MAX_DIGIT_CHAR_CODE = 57
11
- const WHITESPACE_CHAR_CODE = 32
12
- const TAB_CHAR_CODE = 9
13
- const DOT_CHAR_CODE = 46
14
-
15
- export function removeLeadingWhitespaces(string: string): string {
16
- while (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
17
- string = string.substring(1, string.length)
18
- }
19
- return string
20
- }
21
-
22
- export function removeTrailingWhitespaces(string: string): string {
23
- while (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
24
- string = string.substring(0, string.length - 1)
25
- }
26
- return string
27
- }
28
-
29
- export function isDigit(charCode: number): boolean {
30
- return charCode >= MIN_DIGIT_CHAR_CODE && charCode <= MAX_DIGIT_CHAR_CODE
31
- }
32
-
33
- export function isNumber(string: string): boolean {
34
- for (var i = 0; i < string.length; i++) {
35
- const charCode = string.charCodeAt(i)
36
- if (!isDigit(charCode)) {
37
- return false
38
- }
39
- }
40
- return true
41
- }
42
-
43
- export function hasOnly(string: string, char: string): boolean {
44
- const charCode = char.charCodeAt(0)
45
- for (var i = 0; i < string.length; i++) {
46
- const aCharCode = string.charCodeAt(i)
47
- if (aCharCode !== charCode) {
48
- return false
49
- }
50
- }
51
- return true
52
- }
53
-
54
- export function hasUpperCaseCharacterInMiddleOfWord(text: string): boolean {
55
- var beginningOfWord = true
56
- for (var i = 0; i < text.length; i++) {
57
- const character = text.charAt(i)
58
- if (character === ' ') {
59
- beginningOfWord = true
60
- } else {
61
- if (!beginningOfWord && isNaN(character as any * 1) && character === character.toUpperCase() && character.toUpperCase() !== character.toLowerCase()) {
62
- return true
63
- }
64
- beginningOfWord = false
65
- }
66
- }
67
- return false
68
- }
69
-
70
- // Remove whitespace/dots + to uppercase
71
- export function normalizedCharCodeArray(string: string): number[] {
72
- string = string.toUpperCase()
73
- return charCodeArray(string).filter(charCode => charCode !== WHITESPACE_CHAR_CODE && charCode !== TAB_CHAR_CODE && charCode !== DOT_CHAR_CODE)
74
- }
75
-
76
- export function charCodeArray(string: string): number[] {
77
- const charCodes: number[] = []
78
- for (var i = 0; i < string.length; i++) {
79
- charCodes.push(string.charCodeAt(i))
80
- }
81
- return charCodes
82
- }
83
-
84
- export function prefixAfterWhitespace(prefix: string, string: string): string {
85
- if (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
86
- string = removeLeadingWhitespaces(string)
87
- return ' ' + prefix + string
88
- } else {
89
- return prefix + string
90
- }
91
- }
92
-
93
- export function suffixBeforeWhitespace(string: string, suffix: string): string {
94
- if (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
95
- string = removeTrailingWhitespaces(string)
96
- return string + suffix + ' '
97
- } else {
98
- return string + suffix
99
- }
100
- }
101
-
102
- export function isListItemCharacter(string: string): boolean {
103
- if (string.length > 1) {
104
- return false
105
- }
106
- const char = string.charAt(0)
107
- return char === '-' || char === '•' || char === '–'
108
- }
109
-
110
- export function isListItem(string: string): boolean {
111
- return /^[\s]*[-•–][\s].*$/g.test(string)
112
- }
113
-
114
- export function isNumberedListItem(string: string): boolean {
115
- return /^[\s]*[\d]*[.][\s].*$/g.test(string)
116
- }
117
-
118
- export function wordMatch(string1: string, string2: string): number {
119
- const words1 = new Set(string1.toUpperCase().split(' '))
120
- const words2 = new Set(string2.toUpperCase().split(' '))
121
- const intersection = new Set(
122
- [...words1].filter(x => words2.has(x)))
123
- return intersection.size / Math.max(words1.size, words2.size)
124
- }
1
+ /**
2
+ * @description Character-level string utilities for the PDF text pipeline:
3
+ * digit and number detection, leading/trailing whitespace trimming, list-item
4
+ * character and pattern matching (bullet chars, numbered items), word-overlap
5
+ * scoring (`wordMatch`), char-code normalisation for headline matching
6
+ * (`normalizedCharCodeArray`), and camel-case detection
7
+ * (`hasUpperCaseCharacterInMiddleOfWord`).
8
+ */
9
+ const MIN_DIGIT_CHAR_CODE = 48
10
+ const MAX_DIGIT_CHAR_CODE = 57
11
+ const WHITESPACE_CHAR_CODE = 32
12
+ const TAB_CHAR_CODE = 9
13
+ const DOT_CHAR_CODE = 46
14
+
15
+ export function removeLeadingWhitespaces(string: string): string {
16
+ while (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
17
+ string = string.substring(1, string.length)
18
+ }
19
+ return string
20
+ }
21
+
22
+ export function removeTrailingWhitespaces(string: string): string {
23
+ while (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
24
+ string = string.substring(0, string.length - 1)
25
+ }
26
+ return string
27
+ }
28
+
29
+ export function isDigit(charCode: number): boolean {
30
+ return charCode >= MIN_DIGIT_CHAR_CODE && charCode <= MAX_DIGIT_CHAR_CODE
31
+ }
32
+
33
+ export function isNumber(string: string): boolean {
34
+ for (var i = 0; i < string.length; i++) {
35
+ const charCode = string.charCodeAt(i)
36
+ if (!isDigit(charCode)) {
37
+ return false
38
+ }
39
+ }
40
+ return true
41
+ }
42
+
43
+ export function hasOnly(string: string, char: string): boolean {
44
+ const charCode = char.charCodeAt(0)
45
+ for (var i = 0; i < string.length; i++) {
46
+ const aCharCode = string.charCodeAt(i)
47
+ if (aCharCode !== charCode) {
48
+ return false
49
+ }
50
+ }
51
+ return true
52
+ }
53
+
54
+ export function hasUpperCaseCharacterInMiddleOfWord(text: string): boolean {
55
+ var beginningOfWord = true
56
+ for (var i = 0; i < text.length; i++) {
57
+ const character = text.charAt(i)
58
+ if (character === ' ') {
59
+ beginningOfWord = true
60
+ } else {
61
+ if (!beginningOfWord && isNaN(character as any * 1) && character === character.toUpperCase() && character.toUpperCase() !== character.toLowerCase()) {
62
+ return true
63
+ }
64
+ beginningOfWord = false
65
+ }
66
+ }
67
+ return false
68
+ }
69
+
70
+ // Remove whitespace/dots + to uppercase
71
+ export function normalizedCharCodeArray(string: string): number[] {
72
+ string = string.toUpperCase()
73
+ return charCodeArray(string).filter(charCode => charCode !== WHITESPACE_CHAR_CODE && charCode !== TAB_CHAR_CODE && charCode !== DOT_CHAR_CODE)
74
+ }
75
+
76
+ export function charCodeArray(string: string): number[] {
77
+ const charCodes: number[] = []
78
+ for (var i = 0; i < string.length; i++) {
79
+ charCodes.push(string.charCodeAt(i))
80
+ }
81
+ return charCodes
82
+ }
83
+
84
+ export function prefixAfterWhitespace(prefix: string, string: string): string {
85
+ if (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
86
+ string = removeLeadingWhitespaces(string)
87
+ return ' ' + prefix + string
88
+ } else {
89
+ return prefix + string
90
+ }
91
+ }
92
+
93
+ export function suffixBeforeWhitespace(string: string, suffix: string): string {
94
+ if (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
95
+ string = removeTrailingWhitespaces(string)
96
+ return string + suffix + ' '
97
+ } else {
98
+ return string + suffix
99
+ }
100
+ }
101
+
102
+ export function isListItemCharacter(string: string): boolean {
103
+ if (string.length > 1) {
104
+ return false
105
+ }
106
+ const char = string.charAt(0)
107
+ return char === '-' || char === '•' || char === '–'
108
+ }
109
+
110
+ export function isListItem(string: string): boolean {
111
+ return /^[\s]*[-•–][\s].*$/g.test(string)
112
+ }
113
+
114
+ export function isNumberedListItem(string: string): boolean {
115
+ return /^[\s]*[\d]*[.][\s].*$/g.test(string)
116
+ }
117
+
118
+ export function wordMatch(string1: string, string2: string): number {
119
+ const words1 = new Set(string1.toUpperCase().split(' '))
120
+ const words2 = new Set(string2.toUpperCase().split(' '))
121
+ const intersection = new Set(
122
+ [...words1].filter(x => words2.has(x)))
123
+ return intersection.size / Math.max(words1.size, words2.size)
124
+ }