extract-pdf 0.1.21 → 0.1.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -74
- package/package.json +1 -1
- package/src/models/annotation.ts +41 -41
- package/src/models/block-type.ts +203 -203
- package/src/models/line-converter.ts +224 -224
- package/src/models/metadata.ts +29 -29
- package/src/models/page.ts +16 -16
- package/src/models/parse-result.ts +32 -32
- package/src/models/parsed-elements.ts +29 -29
- package/src/models/stashing-stream.ts +86 -86
- package/src/models/text-item-line-grouper.ts +41 -41
- package/src/models/word.ts +31 -31
- package/src/pdf-to-html.ts +225 -225
- package/src/transforms/base/to-line-item-block-transform.ts +29 -29
- package/src/transforms/base/to-line-item-transform.ts +29 -29
- package/src/transforms/base/to-text-item-transform.ts +28 -28
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
- package/src/transforms/block/detect-list-levels.ts +64 -64
- package/src/transforms/block/gather-blocks.ts +113 -113
- package/src/transforms/calculate-global-stats.ts +132 -132
- package/src/transforms/line-item/compact-lines.ts +92 -92
- package/src/transforms/line-item/detect-headers.ts +173 -173
- package/src/transforms/line-item/detect-list-items.ts +68 -68
- package/src/transforms/line-item/detect-toc.ts +459 -459
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
- package/src/transforms/to-html.ts +46 -46
- package/src/utils/is-url-pdf.ts +33 -33
- package/src/utils/page-item-functions.ts +35 -35
- package/src/utils/string-functions.ts +124 -124
|
@@ -1,132 +1,132 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description First-pass transformation that scans all `TextItem`s to compute
|
|
3
|
-
* document-wide statistics stored in `parseResult.globals`: most-frequent text
|
|
4
|
-
* height (`mostUsedHeight`), most-frequent font, most-frequent line distance,
|
|
5
|
-
* maximum height, and a font-to-format map (BOLD/OBLIQUE/BOLD_OBLIQUE) derived
|
|
6
|
-
* from font names. Also deep-copies every page item so subsequent transforms
|
|
7
|
-
* never mutate the original parsed data.
|
|
8
|
-
*/
|
|
9
|
-
import ToTextItemTransformation from "./base/to-text-item-transform";
|
|
10
|
-
import ParseResult from "../models/parse-result";
|
|
11
|
-
import { WordFormat } from "../models/line-converter";
|
|
12
|
-
|
|
13
|
-
export default class CalculateGlobalStats extends ToTextItemTransformation {
|
|
14
|
-
fontMap: Map<string, { name: string }> | undefined;
|
|
15
|
-
|
|
16
|
-
constructor(fontMap?: Map<string, { name: string }>) {
|
|
17
|
-
super("Calculate Global Stats");
|
|
18
|
-
this.fontMap = fontMap;
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
22
|
-
const heightToOccurrence: Record<string, number> = {};
|
|
23
|
-
const fontToOccurrence: Record<string, number> = {};
|
|
24
|
-
var maxHeight = 0;
|
|
25
|
-
var maxHeightFont: string | undefined;
|
|
26
|
-
parseResult.pages.forEach((page) => {
|
|
27
|
-
page.items.forEach((item) => {
|
|
28
|
-
if (!item.height) return;
|
|
29
|
-
heightToOccurrence[item.height] = heightToOccurrence[item.height]
|
|
30
|
-
? heightToOccurrence[item.height] + 1
|
|
31
|
-
: 1;
|
|
32
|
-
fontToOccurrence[item.font] = fontToOccurrence[item.font]
|
|
33
|
-
? fontToOccurrence[item.font] + 1
|
|
34
|
-
: 1;
|
|
35
|
-
if (item.height > maxHeight) {
|
|
36
|
-
maxHeight = item.height;
|
|
37
|
-
maxHeightFont = item.font;
|
|
38
|
-
}
|
|
39
|
-
});
|
|
40
|
-
});
|
|
41
|
-
const mostUsedHeight = parseInt(getMostUsedKey(heightToOccurrence));
|
|
42
|
-
const mostUsedFont = getMostUsedKey(fontToOccurrence);
|
|
43
|
-
|
|
44
|
-
const distanceToOccurrence: Record<string, number> = {};
|
|
45
|
-
parseResult.pages.forEach((page) => {
|
|
46
|
-
var lastItemOfMostUsedHeight: any;
|
|
47
|
-
page.items.forEach((item) => {
|
|
48
|
-
if (item.height === mostUsedHeight && item.text.trim().length > 0) {
|
|
49
|
-
if (
|
|
50
|
-
lastItemOfMostUsedHeight &&
|
|
51
|
-
item.y !== lastItemOfMostUsedHeight.y
|
|
52
|
-
) {
|
|
53
|
-
const distance = lastItemOfMostUsedHeight.y - item.y;
|
|
54
|
-
if (distance > 0) {
|
|
55
|
-
distanceToOccurrence[distance] = distanceToOccurrence[distance]
|
|
56
|
-
? distanceToOccurrence[distance] + 1
|
|
57
|
-
: 1;
|
|
58
|
-
}
|
|
59
|
-
}
|
|
60
|
-
lastItemOfMostUsedHeight = item;
|
|
61
|
-
} else {
|
|
62
|
-
lastItemOfMostUsedHeight = null;
|
|
63
|
-
}
|
|
64
|
-
});
|
|
65
|
-
});
|
|
66
|
-
const mostUsedDistance = parseInt(getMostUsedKey(distanceToOccurrence));
|
|
67
|
-
const fontIdToName: string[] = [];
|
|
68
|
-
const fontToFormats = new Map<string, string>();
|
|
69
|
-
this.fontMap?.forEach(function (value, key) {
|
|
70
|
-
fontIdToName.push(key + " = " + value.name);
|
|
71
|
-
const fontName = value.name.toLowerCase();
|
|
72
|
-
var format: { name: string } | null = null;
|
|
73
|
-
if (key === mostUsedFont) {
|
|
74
|
-
format = null;
|
|
75
|
-
} else if (
|
|
76
|
-
fontName.includes("bold") &&
|
|
77
|
-
(fontName.includes("oblique") || fontName.includes("italic"))
|
|
78
|
-
) {
|
|
79
|
-
format = WordFormat.BOLD_OBLIQUE;
|
|
80
|
-
} else if (fontName.includes("bold")) {
|
|
81
|
-
format = WordFormat.BOLD;
|
|
82
|
-
} else if (fontName.includes("oblique") || fontName.includes("italic")) {
|
|
83
|
-
format = WordFormat.OBLIQUE;
|
|
84
|
-
} else if (fontName === maxHeightFont) {
|
|
85
|
-
format = WordFormat.BOLD;
|
|
86
|
-
}
|
|
87
|
-
if (format) {
|
|
88
|
-
fontToFormats.set(key, format.name);
|
|
89
|
-
}
|
|
90
|
-
});
|
|
91
|
-
fontIdToName.sort();
|
|
92
|
-
|
|
93
|
-
const newPages = parseResult.pages.map((page) => {
|
|
94
|
-
return {
|
|
95
|
-
...page,
|
|
96
|
-
items: page.items.map((textItem) => ({ ...textItem })),
|
|
97
|
-
};
|
|
98
|
-
});
|
|
99
|
-
return new ParseResult({
|
|
100
|
-
...parseResult,
|
|
101
|
-
pages: newPages,
|
|
102
|
-
globals: {
|
|
103
|
-
mostUsedHeight,
|
|
104
|
-
mostUsedFont,
|
|
105
|
-
mostUsedDistance,
|
|
106
|
-
maxHeight,
|
|
107
|
-
maxHeightFont,
|
|
108
|
-
fontToFormats,
|
|
109
|
-
},
|
|
110
|
-
messages: [
|
|
111
|
-
"Items per height: " + JSON.stringify(heightToOccurrence),
|
|
112
|
-
"Items per font: " + JSON.stringify(fontToOccurrence),
|
|
113
|
-
"Items per distance: " + JSON.stringify(distanceToOccurrence),
|
|
114
|
-
"Fonts:" + JSON.stringify(fontIdToName),
|
|
115
|
-
],
|
|
116
|
-
});
|
|
117
|
-
}
|
|
118
|
-
}
|
|
119
|
-
|
|
120
|
-
function getMostUsedKey(keyToOccurrence: Record<string, number>): string {
|
|
121
|
-
var maxOccurence = 0;
|
|
122
|
-
var maxKey = "";
|
|
123
|
-
Object.keys(keyToOccurrence).map((element) => {
|
|
124
|
-
if (!maxKey || keyToOccurrence[element] > maxOccurence) {
|
|
125
|
-
maxOccurence = keyToOccurrence[element];
|
|
126
|
-
maxKey = element;
|
|
127
|
-
}
|
|
128
|
-
});
|
|
129
|
-
return maxKey;
|
|
130
|
-
}
|
|
131
|
-
|
|
132
|
-
|
|
1
|
+
/**
|
|
2
|
+
* @description First-pass transformation that scans all `TextItem`s to compute
|
|
3
|
+
* document-wide statistics stored in `parseResult.globals`: most-frequent text
|
|
4
|
+
* height (`mostUsedHeight`), most-frequent font, most-frequent line distance,
|
|
5
|
+
* maximum height, and a font-to-format map (BOLD/OBLIQUE/BOLD_OBLIQUE) derived
|
|
6
|
+
* from font names. Also deep-copies every page item so subsequent transforms
|
|
7
|
+
* never mutate the original parsed data.
|
|
8
|
+
*/
|
|
9
|
+
import ToTextItemTransformation from "./base/to-text-item-transform";
|
|
10
|
+
import ParseResult from "../models/parse-result";
|
|
11
|
+
import { WordFormat } from "../models/line-converter";
|
|
12
|
+
|
|
13
|
+
export default class CalculateGlobalStats extends ToTextItemTransformation {
|
|
14
|
+
fontMap: Map<string, { name: string }> | undefined;
|
|
15
|
+
|
|
16
|
+
constructor(fontMap?: Map<string, { name: string }>) {
|
|
17
|
+
super("Calculate Global Stats");
|
|
18
|
+
this.fontMap = fontMap;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
const heightToOccurrence: Record<string, number> = {};
|
|
23
|
+
const fontToOccurrence: Record<string, number> = {};
|
|
24
|
+
var maxHeight = 0;
|
|
25
|
+
var maxHeightFont: string | undefined;
|
|
26
|
+
parseResult.pages.forEach((page) => {
|
|
27
|
+
page.items.forEach((item) => {
|
|
28
|
+
if (!item.height) return;
|
|
29
|
+
heightToOccurrence[item.height] = heightToOccurrence[item.height]
|
|
30
|
+
? heightToOccurrence[item.height] + 1
|
|
31
|
+
: 1;
|
|
32
|
+
fontToOccurrence[item.font] = fontToOccurrence[item.font]
|
|
33
|
+
? fontToOccurrence[item.font] + 1
|
|
34
|
+
: 1;
|
|
35
|
+
if (item.height > maxHeight) {
|
|
36
|
+
maxHeight = item.height;
|
|
37
|
+
maxHeightFont = item.font;
|
|
38
|
+
}
|
|
39
|
+
});
|
|
40
|
+
});
|
|
41
|
+
const mostUsedHeight = parseInt(getMostUsedKey(heightToOccurrence));
|
|
42
|
+
const mostUsedFont = getMostUsedKey(fontToOccurrence);
|
|
43
|
+
|
|
44
|
+
const distanceToOccurrence: Record<string, number> = {};
|
|
45
|
+
parseResult.pages.forEach((page) => {
|
|
46
|
+
var lastItemOfMostUsedHeight: any;
|
|
47
|
+
page.items.forEach((item) => {
|
|
48
|
+
if (item.height === mostUsedHeight && item.text.trim().length > 0) {
|
|
49
|
+
if (
|
|
50
|
+
lastItemOfMostUsedHeight &&
|
|
51
|
+
item.y !== lastItemOfMostUsedHeight.y
|
|
52
|
+
) {
|
|
53
|
+
const distance = lastItemOfMostUsedHeight.y - item.y;
|
|
54
|
+
if (distance > 0) {
|
|
55
|
+
distanceToOccurrence[distance] = distanceToOccurrence[distance]
|
|
56
|
+
? distanceToOccurrence[distance] + 1
|
|
57
|
+
: 1;
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
lastItemOfMostUsedHeight = item;
|
|
61
|
+
} else {
|
|
62
|
+
lastItemOfMostUsedHeight = null;
|
|
63
|
+
}
|
|
64
|
+
});
|
|
65
|
+
});
|
|
66
|
+
const mostUsedDistance = parseInt(getMostUsedKey(distanceToOccurrence));
|
|
67
|
+
const fontIdToName: string[] = [];
|
|
68
|
+
const fontToFormats = new Map<string, string>();
|
|
69
|
+
this.fontMap?.forEach(function (value, key) {
|
|
70
|
+
fontIdToName.push(key + " = " + value.name);
|
|
71
|
+
const fontName = value.name.toLowerCase();
|
|
72
|
+
var format: { name: string } | null = null;
|
|
73
|
+
if (key === mostUsedFont) {
|
|
74
|
+
format = null;
|
|
75
|
+
} else if (
|
|
76
|
+
fontName.includes("bold") &&
|
|
77
|
+
(fontName.includes("oblique") || fontName.includes("italic"))
|
|
78
|
+
) {
|
|
79
|
+
format = WordFormat.BOLD_OBLIQUE;
|
|
80
|
+
} else if (fontName.includes("bold")) {
|
|
81
|
+
format = WordFormat.BOLD;
|
|
82
|
+
} else if (fontName.includes("oblique") || fontName.includes("italic")) {
|
|
83
|
+
format = WordFormat.OBLIQUE;
|
|
84
|
+
} else if (fontName === maxHeightFont) {
|
|
85
|
+
format = WordFormat.BOLD;
|
|
86
|
+
}
|
|
87
|
+
if (format) {
|
|
88
|
+
fontToFormats.set(key, format.name);
|
|
89
|
+
}
|
|
90
|
+
});
|
|
91
|
+
fontIdToName.sort();
|
|
92
|
+
|
|
93
|
+
const newPages = parseResult.pages.map((page) => {
|
|
94
|
+
return {
|
|
95
|
+
...page,
|
|
96
|
+
items: page.items.map((textItem) => ({ ...textItem })),
|
|
97
|
+
};
|
|
98
|
+
});
|
|
99
|
+
return new ParseResult({
|
|
100
|
+
...parseResult,
|
|
101
|
+
pages: newPages,
|
|
102
|
+
globals: {
|
|
103
|
+
mostUsedHeight,
|
|
104
|
+
mostUsedFont,
|
|
105
|
+
mostUsedDistance,
|
|
106
|
+
maxHeight,
|
|
107
|
+
maxHeightFont,
|
|
108
|
+
fontToFormats,
|
|
109
|
+
},
|
|
110
|
+
messages: [
|
|
111
|
+
"Items per height: " + JSON.stringify(heightToOccurrence),
|
|
112
|
+
"Items per font: " + JSON.stringify(fontToOccurrence),
|
|
113
|
+
"Items per distance: " + JSON.stringify(distanceToOccurrence),
|
|
114
|
+
"Fonts:" + JSON.stringify(fontIdToName),
|
|
115
|
+
],
|
|
116
|
+
});
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function getMostUsedKey(keyToOccurrence: Record<string, number>): string {
|
|
121
|
+
var maxOccurence = 0;
|
|
122
|
+
var maxKey = "";
|
|
123
|
+
Object.keys(keyToOccurrence).map((element) => {
|
|
124
|
+
if (!maxKey || keyToOccurrence[element] > maxOccurence) {
|
|
125
|
+
maxOccurence = keyToOccurrence[element];
|
|
126
|
+
maxKey = element;
|
|
127
|
+
}
|
|
128
|
+
});
|
|
129
|
+
return maxKey;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
|
|
@@ -1,92 +1,92 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Line-item transformation that collapses `TextItem`s sharing the
|
|
3
|
-
* same y-coordinate into single `LineItem`s via `TextItemLineGrouper` (line
|
|
4
|
-
* bucketing) and `LineConverter` (word/format detection). Simultaneously detects
|
|
5
|
-
* and records footnote links, footnote definitions, hyperlinks, and bold/italic
|
|
6
|
-
* word counts, tagging items as ADDED/REMOVED for debug rendering.
|
|
7
|
-
*/
|
|
8
|
-
import ToLineItemTransformation from "../base/to-line-item-transform";
|
|
9
|
-
import ParseResult from "../../models/parse-result";
|
|
10
|
-
import LineItem from "../../models/line-item";
|
|
11
|
-
import TextItemLineGrouper from "../../models/text-item-line-grouper";
|
|
12
|
-
import LineConverter from "../../models/line-converter";
|
|
13
|
-
import BlockType from "../../models/block-type";
|
|
14
|
-
import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
|
|
15
|
-
|
|
16
|
-
// gathers text items on the same y line to one line item
|
|
17
|
-
export default class CompactLines extends ToLineItemTransformation {
|
|
18
|
-
constructor() {
|
|
19
|
-
super("Compact To Lines");
|
|
20
|
-
}
|
|
21
|
-
|
|
22
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
23
|
-
const { mostUsedDistance, fontToFormats } = parseResult.globals;
|
|
24
|
-
const foundFootnotes = [];
|
|
25
|
-
const foundFootnoteLinks = [];
|
|
26
|
-
var linkCount = 0;
|
|
27
|
-
var formattedWords = 0;
|
|
28
|
-
|
|
29
|
-
const lineGrouper = new TextItemLineGrouper({
|
|
30
|
-
mostUsedDistance: mostUsedDistance,
|
|
31
|
-
});
|
|
32
|
-
const lineCompactor = new LineConverter(fontToFormats);
|
|
33
|
-
|
|
34
|
-
parseResult.pages.forEach((page) => {
|
|
35
|
-
if (page.items.length > 0) {
|
|
36
|
-
const lineItems = [];
|
|
37
|
-
const textItemsGroupedByLine = lineGrouper.group(page.items);
|
|
38
|
-
textItemsGroupedByLine.forEach((lineTextItems) => {
|
|
39
|
-
const lineItem = lineCompactor.compact(lineTextItems);
|
|
40
|
-
if (lineTextItems.length > 1) {
|
|
41
|
-
lineItem.annotation = ADDED_ANNOTATION;
|
|
42
|
-
lineTextItems.forEach((item) => {
|
|
43
|
-
item.annotation = REMOVED_ANNOTATION;
|
|
44
|
-
lineItems.push(
|
|
45
|
-
new LineItem({
|
|
46
|
-
...item,
|
|
47
|
-
}),
|
|
48
|
-
);
|
|
49
|
-
});
|
|
50
|
-
}
|
|
51
|
-
if (lineItem.words.length === 0) {
|
|
52
|
-
lineItem.annotation = REMOVED_ANNOTATION;
|
|
53
|
-
}
|
|
54
|
-
lineItems.push(lineItem);
|
|
55
|
-
|
|
56
|
-
if (lineItem.parsedElements.formattedWords) {
|
|
57
|
-
formattedWords += lineItem.parsedElements.formattedWords;
|
|
58
|
-
}
|
|
59
|
-
if (lineItem.parsedElements.containLinks) {
|
|
60
|
-
linkCount++;
|
|
61
|
-
}
|
|
62
|
-
if (lineItem.parsedElements.footnoteLinks.length > 0) {
|
|
63
|
-
const footnoteLinks = lineItem.parsedElements.footnoteLinks.map(
|
|
64
|
-
(footnoteLink) => ({ footnoteLink, page: page.index + 1 }),
|
|
65
|
-
);
|
|
66
|
-
foundFootnoteLinks.push.apply(foundFootnoteLinks, footnoteLinks);
|
|
67
|
-
}
|
|
68
|
-
if (lineItem.parsedElements.footnotes.length > 0) {
|
|
69
|
-
lineItem.type = BlockType.FOOTNOTES;
|
|
70
|
-
const footnotes = lineItem.parsedElements.footnotes.map(
|
|
71
|
-
(footnote) => ({ footnote, page: page.index + 1 }),
|
|
72
|
-
);
|
|
73
|
-
foundFootnotes.push.apply(foundFootnotes, footnotes);
|
|
74
|
-
}
|
|
75
|
-
});
|
|
76
|
-
page.items = lineItems;
|
|
77
|
-
}
|
|
78
|
-
});
|
|
79
|
-
|
|
80
|
-
return new ParseResult({
|
|
81
|
-
...parseResult,
|
|
82
|
-
messages: [
|
|
83
|
-
"Detected " + formattedWords + " formatted words",
|
|
84
|
-
"Found " + linkCount + " links",
|
|
85
|
-
"Detected " + foundFootnoteLinks.length + " footnotes links",
|
|
86
|
-
"Detected " + foundFootnotes.length + " footnotes",
|
|
87
|
-
],
|
|
88
|
-
});
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
|
|
92
|
-
|
|
1
|
+
/**
|
|
2
|
+
* @description Line-item transformation that collapses `TextItem`s sharing the
|
|
3
|
+
* same y-coordinate into single `LineItem`s via `TextItemLineGrouper` (line
|
|
4
|
+
* bucketing) and `LineConverter` (word/format detection). Simultaneously detects
|
|
5
|
+
* and records footnote links, footnote definitions, hyperlinks, and bold/italic
|
|
6
|
+
* word counts, tagging items as ADDED/REMOVED for debug rendering.
|
|
7
|
+
*/
|
|
8
|
+
import ToLineItemTransformation from "../base/to-line-item-transform";
|
|
9
|
+
import ParseResult from "../../models/parse-result";
|
|
10
|
+
import LineItem from "../../models/line-item";
|
|
11
|
+
import TextItemLineGrouper from "../../models/text-item-line-grouper";
|
|
12
|
+
import LineConverter from "../../models/line-converter";
|
|
13
|
+
import BlockType from "../../models/block-type";
|
|
14
|
+
import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
|
|
15
|
+
|
|
16
|
+
// gathers text items on the same y line to one line item
|
|
17
|
+
export default class CompactLines extends ToLineItemTransformation {
|
|
18
|
+
constructor() {
|
|
19
|
+
super("Compact To Lines");
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
23
|
+
const { mostUsedDistance, fontToFormats } = parseResult.globals;
|
|
24
|
+
const foundFootnotes = [];
|
|
25
|
+
const foundFootnoteLinks = [];
|
|
26
|
+
var linkCount = 0;
|
|
27
|
+
var formattedWords = 0;
|
|
28
|
+
|
|
29
|
+
const lineGrouper = new TextItemLineGrouper({
|
|
30
|
+
mostUsedDistance: mostUsedDistance,
|
|
31
|
+
});
|
|
32
|
+
const lineCompactor = new LineConverter(fontToFormats);
|
|
33
|
+
|
|
34
|
+
parseResult.pages.forEach((page) => {
|
|
35
|
+
if (page.items.length > 0) {
|
|
36
|
+
const lineItems = [];
|
|
37
|
+
const textItemsGroupedByLine = lineGrouper.group(page.items);
|
|
38
|
+
textItemsGroupedByLine.forEach((lineTextItems) => {
|
|
39
|
+
const lineItem = lineCompactor.compact(lineTextItems);
|
|
40
|
+
if (lineTextItems.length > 1) {
|
|
41
|
+
lineItem.annotation = ADDED_ANNOTATION;
|
|
42
|
+
lineTextItems.forEach((item) => {
|
|
43
|
+
item.annotation = REMOVED_ANNOTATION;
|
|
44
|
+
lineItems.push(
|
|
45
|
+
new LineItem({
|
|
46
|
+
...item,
|
|
47
|
+
}),
|
|
48
|
+
);
|
|
49
|
+
});
|
|
50
|
+
}
|
|
51
|
+
if (lineItem.words.length === 0) {
|
|
52
|
+
lineItem.annotation = REMOVED_ANNOTATION;
|
|
53
|
+
}
|
|
54
|
+
lineItems.push(lineItem);
|
|
55
|
+
|
|
56
|
+
if (lineItem.parsedElements.formattedWords) {
|
|
57
|
+
formattedWords += lineItem.parsedElements.formattedWords;
|
|
58
|
+
}
|
|
59
|
+
if (lineItem.parsedElements.containLinks) {
|
|
60
|
+
linkCount++;
|
|
61
|
+
}
|
|
62
|
+
if (lineItem.parsedElements.footnoteLinks.length > 0) {
|
|
63
|
+
const footnoteLinks = lineItem.parsedElements.footnoteLinks.map(
|
|
64
|
+
(footnoteLink) => ({ footnoteLink, page: page.index + 1 }),
|
|
65
|
+
);
|
|
66
|
+
foundFootnoteLinks.push.apply(foundFootnoteLinks, footnoteLinks);
|
|
67
|
+
}
|
|
68
|
+
if (lineItem.parsedElements.footnotes.length > 0) {
|
|
69
|
+
lineItem.type = BlockType.FOOTNOTES;
|
|
70
|
+
const footnotes = lineItem.parsedElements.footnotes.map(
|
|
71
|
+
(footnote) => ({ footnote, page: page.index + 1 }),
|
|
72
|
+
);
|
|
73
|
+
foundFootnotes.push.apply(foundFootnotes, footnotes);
|
|
74
|
+
}
|
|
75
|
+
});
|
|
76
|
+
page.items = lineItems;
|
|
77
|
+
}
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
return new ParseResult({
|
|
81
|
+
...parseResult,
|
|
82
|
+
messages: [
|
|
83
|
+
"Detected " + formattedWords + " formatted words",
|
|
84
|
+
"Found " + linkCount + " links",
|
|
85
|
+
"Detected " + foundFootnoteLinks.length + " footnotes links",
|
|
86
|
+
"Detected " + foundFootnotes.length + " footnotes",
|
|
87
|
+
],
|
|
88
|
+
});
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
|