extract-pdf 0.1.21 → 0.1.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/README.md +74 -74
  2. package/package.json +1 -1
  3. package/src/models/annotation.ts +41 -41
  4. package/src/models/block-type.ts +203 -203
  5. package/src/models/line-converter.ts +224 -224
  6. package/src/models/metadata.ts +29 -29
  7. package/src/models/page.ts +16 -16
  8. package/src/models/parse-result.ts +32 -32
  9. package/src/models/parsed-elements.ts +29 -29
  10. package/src/models/stashing-stream.ts +86 -86
  11. package/src/models/text-item-line-grouper.ts +41 -41
  12. package/src/models/word.ts +31 -31
  13. package/src/pdf-to-html.ts +225 -225
  14. package/src/transforms/base/to-line-item-block-transform.ts +29 -29
  15. package/src/transforms/base/to-line-item-transform.ts +29 -29
  16. package/src/transforms/base/to-text-item-transform.ts +28 -28
  17. package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
  18. package/src/transforms/block/detect-list-levels.ts +64 -64
  19. package/src/transforms/block/gather-blocks.ts +113 -113
  20. package/src/transforms/calculate-global-stats.ts +132 -132
  21. package/src/transforms/line-item/compact-lines.ts +92 -92
  22. package/src/transforms/line-item/detect-headers.ts +173 -173
  23. package/src/transforms/line-item/detect-list-items.ts +68 -68
  24. package/src/transforms/line-item/detect-toc.ts +459 -459
  25. package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
  26. package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
  27. package/src/transforms/to-html.ts +46 -46
  28. package/src/utils/is-url-pdf.ts +33 -33
  29. package/src/utils/page-item-functions.ts +35 -35
  30. package/src/utils/string-functions.ts +124 -124
@@ -1,173 +1,173 @@
1
- /**
2
- * @description Line-item transformation that assigns H1–H6 block types based on
3
- * font height relative to the document's body height. Handles three cases: title
4
- * pages (items at the document's maximum height), TOC-derived headline height
5
- * ranges when a table of contents was found, and a fallback that ranks all
6
- * above-average heights largest-first and maps them to heading levels. Also
7
- * detects all-uppercase body-font lines as the next lower heading level.
8
- */
9
- import ToLineItemTransformation from "../base/to-line-item-transform";
10
- import ParseResult from "../../models/parse-result";
11
- import { DETECTED_ANNOTATION } from "../../models/annotation";
12
- import BlockType from "../../models/block-type";
13
- import { isListItem } from "../../utils/string-functions";
14
-
15
- // Detect headlines based on heights
16
- export default class DetectHeaders extends ToLineItemTransformation {
17
- constructor() {
18
- super("Detect Headers");
19
- }
20
-
21
- transform(parseResult: ParseResult): ParseResult {
22
- const {
23
- tocPages,
24
- headlineTypeToHeightRange,
25
- mostUsedHeight,
26
- mostUsedDistance,
27
- mostUsedFont,
28
- maxHeight,
29
- } = parseResult.globals;
30
- const hasToc = tocPages.length > 0;
31
- var detectedHeaders = 0;
32
-
33
- // Handle title pages
34
- const pagesWithMaxHeight = findPagesWithMaxHeight(
35
- parseResult.pages,
36
- maxHeight,
37
- );
38
- const min2ndLevelHeaderHeigthOnMaxPage =
39
- mostUsedHeight + (maxHeight - mostUsedHeight) / 4;
40
- pagesWithMaxHeight.forEach((titlePage) => {
41
- titlePage.items.forEach((item) => {
42
- const height = item.height;
43
- if (!item.type && height > min2ndLevelHeaderHeigthOnMaxPage) {
44
- if (height === maxHeight) {
45
- item.type = BlockType.H1;
46
- } else {
47
- item.type = BlockType.H2;
48
- }
49
- item.annotation = DETECTED_ANNOTATION;
50
- detectedHeaders++;
51
- }
52
- });
53
- });
54
-
55
- if (hasToc) {
56
- // Use existing headline heights to find additional headlines
57
- const headlineTypes = Object.keys(headlineTypeToHeightRange);
58
- headlineTypes.forEach((headlineType) => {
59
- var range = headlineTypeToHeightRange[headlineType];
60
- if (range.max > mostUsedHeight) {
61
- // use only very clear headlines, only use max
62
- parseResult.pages.forEach((page) => {
63
- page.items.forEach((item) => {
64
- if (!item.type && item.height === range.max) {
65
- item.annotation = DETECTED_ANNOTATION;
66
- item.type = BlockType.enumValueOf(headlineType);
67
- detectedHeaders++;
68
- }
69
- });
70
- });
71
- }
72
- });
73
- } else {
74
- // Categorize headlines by the text heights
75
- const heights = [];
76
- var lastHeight;
77
- parseResult.pages.forEach((page) => {
78
- page.items.forEach((item) => {
79
- if (
80
- !item.type &&
81
- item.height > mostUsedHeight &&
82
- !isListItem(item.text())
83
- ) {
84
- if (
85
- !heights.includes(item.height) &&
86
- (!lastHeight || lastHeight > item.height)
87
- ) {
88
- heights.push(item.height);
89
- }
90
- }
91
- });
92
- });
93
- heights.sort((a, b) => b - a);
94
-
95
- heights.forEach((height, i) => {
96
- const headlineLevel = i + 2;
97
- if (headlineLevel <= 6) {
98
- const headlineType = BlockType.headlineByLevel(2 + i);
99
- parseResult.pages.forEach((page) => {
100
- page.items.forEach((item) => {
101
- if (
102
- !item.type &&
103
- item.height === height &&
104
- !isListItem(item.text())
105
- ) {
106
- detectedHeaders++;
107
- item.annotation = DETECTED_ANNOTATION;
108
- item.type = headlineType;
109
- }
110
- });
111
- });
112
- }
113
- });
114
- }
115
-
116
- // find headlines which have paragraph height
117
- var smallesHeadlineLevel = 1;
118
- parseResult.pages.forEach((page) => {
119
- page.items.forEach((item) => {
120
- if (item.type && item.type.headline) {
121
- smallesHeadlineLevel = Math.max(
122
- smallesHeadlineLevel,
123
- item.type.headlineLevel,
124
- );
125
- }
126
- });
127
- });
128
- if (smallesHeadlineLevel < 6) {
129
- const nextHeadlineType = BlockType.headlineByLevel(
130
- smallesHeadlineLevel + 1,
131
- );
132
- parseResult.pages.forEach((page) => {
133
- var lastItem;
134
- page.items.forEach((item) => {
135
- if (
136
- !item.type &&
137
- item.height === mostUsedHeight &&
138
- item.font !== mostUsedFont &&
139
- (!lastItem ||
140
- lastItem.y < item.y ||
141
- (lastItem.type && lastItem.type.headline) ||
142
- lastItem.y - item.y > mostUsedDistance * 2) &&
143
- item.text() === item.text().toUpperCase()
144
- ) {
145
- detectedHeaders++;
146
- item.annotation = DETECTED_ANNOTATION;
147
- item.type = nextHeadlineType;
148
- }
149
- lastItem = item;
150
- });
151
- });
152
- }
153
-
154
- return new ParseResult({
155
- ...parseResult,
156
- messages: ["Detected " + detectedHeaders + " headlines."],
157
- });
158
- }
159
- }
160
-
161
- function findPagesWithMaxHeight(pages: any[], maxHeight: number): Set<any> {
162
- const maxHeaderPagesSet = new Set<any>();
163
- pages.forEach((page) => {
164
- page.items.forEach((item) => {
165
- if (!item.type && item.height === maxHeight) {
166
- maxHeaderPagesSet.add(page);
167
- }
168
- });
169
- });
170
- return maxHeaderPagesSet;
171
- }
172
-
173
-
1
+ /**
2
+ * @description Line-item transformation that assigns H1–H6 block types based on
3
+ * font height relative to the document's body height. Handles three cases: title
4
+ * pages (items at the document's maximum height), TOC-derived headline height
5
+ * ranges when a table of contents was found, and a fallback that ranks all
6
+ * above-average heights largest-first and maps them to heading levels. Also
7
+ * detects all-uppercase body-font lines as the next lower heading level.
8
+ */
9
+ import ToLineItemTransformation from "../base/to-line-item-transform";
10
+ import ParseResult from "../../models/parse-result";
11
+ import { DETECTED_ANNOTATION } from "../../models/annotation";
12
+ import BlockType from "../../models/block-type";
13
+ import { isListItem } from "../../utils/string-functions";
14
+
15
+ // Detect headlines based on heights
16
+ export default class DetectHeaders extends ToLineItemTransformation {
17
+ constructor() {
18
+ super("Detect Headers");
19
+ }
20
+
21
+ transform(parseResult: ParseResult): ParseResult {
22
+ const {
23
+ tocPages,
24
+ headlineTypeToHeightRange,
25
+ mostUsedHeight,
26
+ mostUsedDistance,
27
+ mostUsedFont,
28
+ maxHeight,
29
+ } = parseResult.globals;
30
+ const hasToc = tocPages.length > 0;
31
+ var detectedHeaders = 0;
32
+
33
+ // Handle title pages
34
+ const pagesWithMaxHeight = findPagesWithMaxHeight(
35
+ parseResult.pages,
36
+ maxHeight,
37
+ );
38
+ const min2ndLevelHeaderHeigthOnMaxPage =
39
+ mostUsedHeight + (maxHeight - mostUsedHeight) / 4;
40
+ pagesWithMaxHeight.forEach((titlePage) => {
41
+ titlePage.items.forEach((item) => {
42
+ const height = item.height;
43
+ if (!item.type && height > min2ndLevelHeaderHeigthOnMaxPage) {
44
+ if (height === maxHeight) {
45
+ item.type = BlockType.H1;
46
+ } else {
47
+ item.type = BlockType.H2;
48
+ }
49
+ item.annotation = DETECTED_ANNOTATION;
50
+ detectedHeaders++;
51
+ }
52
+ });
53
+ });
54
+
55
+ if (hasToc) {
56
+ // Use existing headline heights to find additional headlines
57
+ const headlineTypes = Object.keys(headlineTypeToHeightRange);
58
+ headlineTypes.forEach((headlineType) => {
59
+ var range = headlineTypeToHeightRange[headlineType];
60
+ if (range.max > mostUsedHeight) {
61
+ // use only very clear headlines, only use max
62
+ parseResult.pages.forEach((page) => {
63
+ page.items.forEach((item) => {
64
+ if (!item.type && item.height === range.max) {
65
+ item.annotation = DETECTED_ANNOTATION;
66
+ item.type = BlockType.enumValueOf(headlineType);
67
+ detectedHeaders++;
68
+ }
69
+ });
70
+ });
71
+ }
72
+ });
73
+ } else {
74
+ // Categorize headlines by the text heights
75
+ const heights = [];
76
+ var lastHeight;
77
+ parseResult.pages.forEach((page) => {
78
+ page.items.forEach((item) => {
79
+ if (
80
+ !item.type &&
81
+ item.height > mostUsedHeight &&
82
+ !isListItem(item.text())
83
+ ) {
84
+ if (
85
+ !heights.includes(item.height) &&
86
+ (!lastHeight || lastHeight > item.height)
87
+ ) {
88
+ heights.push(item.height);
89
+ }
90
+ }
91
+ });
92
+ });
93
+ heights.sort((a, b) => b - a);
94
+
95
+ heights.forEach((height, i) => {
96
+ const headlineLevel = i + 2;
97
+ if (headlineLevel <= 6) {
98
+ const headlineType = BlockType.headlineByLevel(2 + i);
99
+ parseResult.pages.forEach((page) => {
100
+ page.items.forEach((item) => {
101
+ if (
102
+ !item.type &&
103
+ item.height === height &&
104
+ !isListItem(item.text())
105
+ ) {
106
+ detectedHeaders++;
107
+ item.annotation = DETECTED_ANNOTATION;
108
+ item.type = headlineType;
109
+ }
110
+ });
111
+ });
112
+ }
113
+ });
114
+ }
115
+
116
+ // find headlines which have paragraph height
117
+ var smallesHeadlineLevel = 1;
118
+ parseResult.pages.forEach((page) => {
119
+ page.items.forEach((item) => {
120
+ if (item.type && item.type.headline) {
121
+ smallesHeadlineLevel = Math.max(
122
+ smallesHeadlineLevel,
123
+ item.type.headlineLevel,
124
+ );
125
+ }
126
+ });
127
+ });
128
+ if (smallesHeadlineLevel < 6) {
129
+ const nextHeadlineType = BlockType.headlineByLevel(
130
+ smallesHeadlineLevel + 1,
131
+ );
132
+ parseResult.pages.forEach((page) => {
133
+ var lastItem;
134
+ page.items.forEach((item) => {
135
+ if (
136
+ !item.type &&
137
+ item.height === mostUsedHeight &&
138
+ item.font !== mostUsedFont &&
139
+ (!lastItem ||
140
+ lastItem.y < item.y ||
141
+ (lastItem.type && lastItem.type.headline) ||
142
+ lastItem.y - item.y > mostUsedDistance * 2) &&
143
+ item.text() === item.text().toUpperCase()
144
+ ) {
145
+ detectedHeaders++;
146
+ item.annotation = DETECTED_ANNOTATION;
147
+ item.type = nextHeadlineType;
148
+ }
149
+ lastItem = item;
150
+ });
151
+ });
152
+ }
153
+
154
+ return new ParseResult({
155
+ ...parseResult,
156
+ messages: ["Detected " + detectedHeaders + " headlines."],
157
+ });
158
+ }
159
+ }
160
+
161
+ function findPagesWithMaxHeight(pages: any[], maxHeight: number): Set<any> {
162
+ const maxHeaderPagesSet = new Set<any>();
163
+ pages.forEach((page) => {
164
+ page.items.forEach((item) => {
165
+ if (!item.type && item.height === maxHeight) {
166
+ maxHeaderPagesSet.add(page);
167
+ }
168
+ });
169
+ });
170
+ return maxHeaderPagesSet;
171
+ }
172
+
173
+
@@ -1,68 +1,68 @@
1
- /**
2
- * @description Line-item transformation that identifies list items starting with
3
- * bullet characters (-, •, –) or numbered patterns (`1. text`) and assigns
4
- * `BlockType.LIST`. Non-dash bullet characters are normalised to `-` by marking
5
- * the original line REMOVED and inserting a replacement `LineItem` (ADDED),
6
- * preserving the debug diff trail.
7
- */
8
-
9
- import ToLineItemTransformation from '../base/to-line-item-transform'
10
- import ParseResult from '../../models/parse-result'
11
- import LineItem from '../../models/line-item'
12
- import Word from '../../models/word'
13
- import { REMOVED_ANNOTATION, ADDED_ANNOTATION, DETECTED_ANNOTATION } from '../../models/annotation'
14
- import BlockType from '../../models/block-type'
15
- import { isListItemCharacter, isNumberedListItem } from '../../utils/string-functions'
16
-
17
- // Detect items starting with -, \u2022, etc...
18
- export default class DetectListItems extends ToLineItemTransformation {
19
- constructor () {
20
- super('Detect List Items')
21
- }
22
-
23
- transform(parseResult: ParseResult): ParseResult {
24
- var foundListItems = 0
25
- var foundNumberedItems = 0
26
- parseResult.pages.forEach(page => {
27
- const newItems = []
28
- page.items.forEach(item => {
29
- newItems.push(item)
30
- if (!item.type) {
31
- var text = item.text()
32
- if (isListItemCharacter(item.words[0].string)) {
33
- foundListItems++
34
- if (item.words[0].string === '-') {
35
- item.annotation = DETECTED_ANNOTATION
36
- item.type = BlockType.LIST
37
- } else {
38
- item.annotation = REMOVED_ANNOTATION
39
- const newWords = item.words.map(word => new Word({
40
- ...word,
41
- }))
42
- newWords[0].string = '-'
43
- newItems.push(new LineItem({
44
- ...item,
45
- words: newWords,
46
- annotation: ADDED_ANNOTATION,
47
- type: BlockType.LIST,
48
- }))
49
- }
50
- } else if (isNumberedListItem(text)) { // TODO check that starts with 1 (kala chakra)
51
- foundNumberedItems++
52
- item.annotation = DETECTED_ANNOTATION
53
- item.type = BlockType.LIST
54
- }
55
- }
56
- })
57
- page.items = newItems
58
- })
59
-
60
- return new ParseResult({
61
- ...parseResult,
62
- messages: [
63
- 'Detected ' + foundListItems + ' plain list items.',
64
- 'Detected ' + foundNumberedItems + ' numbered list items.',
65
- ],
66
- })
67
- }
68
- }
1
+ /**
2
+ * @description Line-item transformation that identifies list items starting with
3
+ * bullet characters (-, •, –) or numbered patterns (`1. text`) and assigns
4
+ * `BlockType.LIST`. Non-dash bullet characters are normalised to `-` by marking
5
+ * the original line REMOVED and inserting a replacement `LineItem` (ADDED),
6
+ * preserving the debug diff trail.
7
+ */
8
+
9
+ import ToLineItemTransformation from '../base/to-line-item-transform'
10
+ import ParseResult from '../../models/parse-result'
11
+ import LineItem from '../../models/line-item'
12
+ import Word from '../../models/word'
13
+ import { REMOVED_ANNOTATION, ADDED_ANNOTATION, DETECTED_ANNOTATION } from '../../models/annotation'
14
+ import BlockType from '../../models/block-type'
15
+ import { isListItemCharacter, isNumberedListItem } from '../../utils/string-functions'
16
+
17
+ // Detect items starting with -, \u2022, etc...
18
+ export default class DetectListItems extends ToLineItemTransformation {
19
+ constructor () {
20
+ super('Detect List Items')
21
+ }
22
+
23
+ transform(parseResult: ParseResult): ParseResult {
24
+ var foundListItems = 0
25
+ var foundNumberedItems = 0
26
+ parseResult.pages.forEach(page => {
27
+ const newItems = []
28
+ page.items.forEach(item => {
29
+ newItems.push(item)
30
+ if (!item.type) {
31
+ var text = item.text()
32
+ if (isListItemCharacter(item.words[0].string)) {
33
+ foundListItems++
34
+ if (item.words[0].string === '-') {
35
+ item.annotation = DETECTED_ANNOTATION
36
+ item.type = BlockType.LIST
37
+ } else {
38
+ item.annotation = REMOVED_ANNOTATION
39
+ const newWords = item.words.map(word => new Word({
40
+ ...word,
41
+ }))
42
+ newWords[0].string = '-'
43
+ newItems.push(new LineItem({
44
+ ...item,
45
+ words: newWords,
46
+ annotation: ADDED_ANNOTATION,
47
+ type: BlockType.LIST,
48
+ }))
49
+ }
50
+ } else if (isNumberedListItem(text)) { // TODO check that starts with 1 (kala chakra)
51
+ foundNumberedItems++
52
+ item.annotation = DETECTED_ANNOTATION
53
+ item.type = BlockType.LIST
54
+ }
55
+ }
56
+ })
57
+ page.items = newItems
58
+ })
59
+
60
+ return new ParseResult({
61
+ ...parseResult,
62
+ messages: [
63
+ 'Detected ' + foundListItems + ' plain list items.',
64
+ 'Detected ' + foundNumberedItems + ' numbered list items.',
65
+ ],
66
+ })
67
+ }
68
+ }