extract-pdf 0.1.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/README.md +16 -10
  2. package/dist/models/annotation.d.ts +20 -0
  3. package/dist/models/block-type.d.ts +10 -0
  4. package/dist/models/headline-finder.d.ts +11 -0
  5. package/dist/models/line-converter.d.ts +10 -0
  6. package/dist/models/line-item-block.d.ts +14 -0
  7. package/dist/models/line-item.d.ts +25 -0
  8. package/dist/models/metadata.d.ts +23 -0
  9. package/dist/models/page-item.d.ts +21 -0
  10. package/dist/models/page.d.ts +14 -0
  11. package/dist/models/parse-result.d.ts +24 -0
  12. package/dist/models/parsed-elements.d.ts +20 -0
  13. package/dist/models/stashing-stream.d.ts +23 -0
  14. package/dist/models/text-item-line-grouper.d.ts +8 -0
  15. package/dist/models/text-item.d.ts +29 -0
  16. package/dist/models/word.d.ts +27 -0
  17. package/dist/pdf-to-html.cjs.js +1 -1
  18. package/dist/pdf-to-html.d.ts +36 -41
  19. package/dist/pdf-to-html.es.js +1 -1
  20. package/dist/transforms/base/to-line-item-block-transform.d.ts +6 -0
  21. package/dist/transforms/base/to-line-item-transform.d.ts +6 -0
  22. package/dist/transforms/base/to-text-item-transform.d.ts +6 -0
  23. package/dist/transforms/base/transformation.d.ts +8 -0
  24. package/dist/transforms/block/detect-code-quote-blocks.d.ts +6 -0
  25. package/dist/transforms/block/detect-list-levels.d.ts +6 -0
  26. package/dist/transforms/block/gather-blocks.d.ts +6 -0
  27. package/dist/transforms/calculate-global-stats.d.ts +11 -0
  28. package/dist/transforms/line-item/compact-lines.d.ts +6 -0
  29. package/dist/transforms/line-item/detect-headers.d.ts +6 -0
  30. package/dist/transforms/line-item/detect-list-items.d.ts +6 -0
  31. package/dist/transforms/line-item/detect-toc.d.ts +6 -0
  32. package/dist/transforms/line-item/remove-repetitive-elements.d.ts +6 -0
  33. package/dist/transforms/line-item/vertical-to-horizontal.d.ts +6 -0
  34. package/dist/transforms/to-html.d.ts +6 -0
  35. package/dist/transforms/to-text-blocks.d.ts +6 -0
  36. package/dist/utils/is-url-pdf.d.ts +1 -0
  37. package/dist/utils/page-item-functions.d.ts +8 -0
  38. package/dist/utils/page-number-functions.d.ts +14 -0
  39. package/dist/utils/string-functions.d.ts +14 -0
  40. package/package.json +9 -12
  41. package/src/models/annotation.ts +41 -0
  42. package/src/models/block-type.ts +203 -0
  43. package/src/models/headline-finder.ts +53 -0
  44. package/src/models/line-converter.ts +224 -0
  45. package/src/models/line-item-block.ts +51 -0
  46. package/src/models/line-item.ts +59 -0
  47. package/src/models/metadata.ts +29 -0
  48. package/src/models/page-item.ts +36 -0
  49. package/src/models/page.ts +16 -0
  50. package/src/models/parse-result.ts +32 -0
  51. package/src/models/parsed-elements.ts +29 -0
  52. package/src/models/stashing-stream.ts +86 -0
  53. package/src/models/text-item-line-grouper.ts +41 -0
  54. package/src/models/text-item.ts +50 -0
  55. package/src/models/word.ts +31 -0
  56. package/src/pdf-to-html.ts +225 -0
  57. package/src/transforms/base/to-line-item-block-transform.ts +29 -0
  58. package/src/transforms/base/to-line-item-transform.ts +29 -0
  59. package/src/transforms/base/to-text-item-transform.ts +28 -0
  60. package/src/transforms/base/transformation.ts +35 -0
  61. package/src/transforms/block/detect-code-quote-blocks.ts +57 -0
  62. package/src/transforms/block/detect-list-levels.ts +64 -0
  63. package/src/transforms/block/gather-blocks.ts +113 -0
  64. package/src/transforms/calculate-global-stats.ts +132 -0
  65. package/src/transforms/line-item/compact-lines.ts +92 -0
  66. package/src/transforms/line-item/detect-headers.ts +173 -0
  67. package/src/transforms/line-item/detect-list-items.ts +68 -0
  68. package/src/transforms/line-item/detect-toc.ts +459 -0
  69. package/src/transforms/line-item/remove-repetitive-elements.ts +101 -0
  70. package/src/transforms/line-item/vertical-to-horizontal.ts +90 -0
  71. package/src/transforms/to-html.ts +46 -0
  72. package/src/transforms/to-text-blocks.ts +38 -0
  73. package/src/utils/is-url-pdf.ts +33 -0
  74. package/src/utils/page-item-functions.ts +35 -0
  75. package/src/utils/page-number-functions.ts +109 -0
  76. package/src/utils/string-functions.ts +124 -0
@@ -0,0 +1,132 @@
1
+ /**
2
+ * @description First-pass transformation that scans all `TextItem`s to compute
3
+ * document-wide statistics stored in `parseResult.globals`: most-frequent text
4
+ * height (`mostUsedHeight`), most-frequent font, most-frequent line distance,
5
+ * maximum height, and a font-to-format map (BOLD/OBLIQUE/BOLD_OBLIQUE) derived
6
+ * from font names. Also deep-copies every page item so subsequent transforms
7
+ * never mutate the original parsed data.
8
+ */
9
+ import ToTextItemTransformation from "./base/to-text-item-transform";
10
+ import ParseResult from "../models/parse-result";
11
+ import { WordFormat } from "../models/line-converter";
12
+
13
+ export default class CalculateGlobalStats extends ToTextItemTransformation {
14
+ fontMap: Map<string, { name: string }> | undefined;
15
+
16
+ constructor(fontMap?: Map<string, { name: string }>) {
17
+ super("Calculate Global Stats");
18
+ this.fontMap = fontMap;
19
+ }
20
+
21
+ transform(parseResult: ParseResult): ParseResult {
22
+ const heightToOccurrence: Record<string, number> = {};
23
+ const fontToOccurrence: Record<string, number> = {};
24
+ var maxHeight = 0;
25
+ var maxHeightFont: string | undefined;
26
+ parseResult.pages.forEach((page) => {
27
+ page.items.forEach((item) => {
28
+ if (!item.height) return;
29
+ heightToOccurrence[item.height] = heightToOccurrence[item.height]
30
+ ? heightToOccurrence[item.height] + 1
31
+ : 1;
32
+ fontToOccurrence[item.font] = fontToOccurrence[item.font]
33
+ ? fontToOccurrence[item.font] + 1
34
+ : 1;
35
+ if (item.height > maxHeight) {
36
+ maxHeight = item.height;
37
+ maxHeightFont = item.font;
38
+ }
39
+ });
40
+ });
41
+ const mostUsedHeight = parseInt(getMostUsedKey(heightToOccurrence));
42
+ const mostUsedFont = getMostUsedKey(fontToOccurrence);
43
+
44
+ const distanceToOccurrence: Record<string, number> = {};
45
+ parseResult.pages.forEach((page) => {
46
+ var lastItemOfMostUsedHeight: any;
47
+ page.items.forEach((item) => {
48
+ if (item.height === mostUsedHeight && item.text.trim().length > 0) {
49
+ if (
50
+ lastItemOfMostUsedHeight &&
51
+ item.y !== lastItemOfMostUsedHeight.y
52
+ ) {
53
+ const distance = lastItemOfMostUsedHeight.y - item.y;
54
+ if (distance > 0) {
55
+ distanceToOccurrence[distance] = distanceToOccurrence[distance]
56
+ ? distanceToOccurrence[distance] + 1
57
+ : 1;
58
+ }
59
+ }
60
+ lastItemOfMostUsedHeight = item;
61
+ } else {
62
+ lastItemOfMostUsedHeight = null;
63
+ }
64
+ });
65
+ });
66
+ const mostUsedDistance = parseInt(getMostUsedKey(distanceToOccurrence));
67
+ const fontIdToName: string[] = [];
68
+ const fontToFormats = new Map<string, string>();
69
+ this.fontMap?.forEach(function (value, key) {
70
+ fontIdToName.push(key + " = " + value.name);
71
+ const fontName = value.name.toLowerCase();
72
+ var format: { name: string } | null = null;
73
+ if (key === mostUsedFont) {
74
+ format = null;
75
+ } else if (
76
+ fontName.includes("bold") &&
77
+ (fontName.includes("oblique") || fontName.includes("italic"))
78
+ ) {
79
+ format = WordFormat.BOLD_OBLIQUE;
80
+ } else if (fontName.includes("bold")) {
81
+ format = WordFormat.BOLD;
82
+ } else if (fontName.includes("oblique") || fontName.includes("italic")) {
83
+ format = WordFormat.OBLIQUE;
84
+ } else if (fontName === maxHeightFont) {
85
+ format = WordFormat.BOLD;
86
+ }
87
+ if (format) {
88
+ fontToFormats.set(key, format.name);
89
+ }
90
+ });
91
+ fontIdToName.sort();
92
+
93
+ const newPages = parseResult.pages.map((page) => {
94
+ return {
95
+ ...page,
96
+ items: page.items.map((textItem) => ({ ...textItem })),
97
+ };
98
+ });
99
+ return new ParseResult({
100
+ ...parseResult,
101
+ pages: newPages,
102
+ globals: {
103
+ mostUsedHeight,
104
+ mostUsedFont,
105
+ mostUsedDistance,
106
+ maxHeight,
107
+ maxHeightFont,
108
+ fontToFormats,
109
+ },
110
+ messages: [
111
+ "Items per height: " + JSON.stringify(heightToOccurrence),
112
+ "Items per font: " + JSON.stringify(fontToOccurrence),
113
+ "Items per distance: " + JSON.stringify(distanceToOccurrence),
114
+ "Fonts:" + JSON.stringify(fontIdToName),
115
+ ],
116
+ });
117
+ }
118
+ }
119
+
120
+ function getMostUsedKey(keyToOccurrence: Record<string, number>): string {
121
+ var maxOccurence = 0;
122
+ var maxKey = "";
123
+ Object.keys(keyToOccurrence).map((element) => {
124
+ if (!maxKey || keyToOccurrence[element] > maxOccurence) {
125
+ maxOccurence = keyToOccurrence[element];
126
+ maxKey = element;
127
+ }
128
+ });
129
+ return maxKey;
130
+ }
131
+
132
+
@@ -0,0 +1,92 @@
1
+ /**
2
+ * @description Line-item transformation that collapses `TextItem`s sharing the
3
+ * same y-coordinate into single `LineItem`s via `TextItemLineGrouper` (line
4
+ * bucketing) and `LineConverter` (word/format detection). Simultaneously detects
5
+ * and records footnote links, footnote definitions, hyperlinks, and bold/italic
6
+ * word counts, tagging items as ADDED/REMOVED for debug rendering.
7
+ */
8
+ import ToLineItemTransformation from "../base/to-line-item-transform";
9
+ import ParseResult from "../../models/parse-result";
10
+ import LineItem from "../../models/line-item";
11
+ import TextItemLineGrouper from "../../models/text-item-line-grouper";
12
+ import LineConverter from "../../models/line-converter";
13
+ import BlockType from "../../models/block-type";
14
+ import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
15
+
16
+ // gathers text items on the same y line to one line item
17
+ export default class CompactLines extends ToLineItemTransformation {
18
+ constructor() {
19
+ super("Compact To Lines");
20
+ }
21
+
22
+ transform(parseResult: ParseResult): ParseResult {
23
+ const { mostUsedDistance, fontToFormats } = parseResult.globals;
24
+ const foundFootnotes = [];
25
+ const foundFootnoteLinks = [];
26
+ var linkCount = 0;
27
+ var formattedWords = 0;
28
+
29
+ const lineGrouper = new TextItemLineGrouper({
30
+ mostUsedDistance: mostUsedDistance,
31
+ });
32
+ const lineCompactor = new LineConverter(fontToFormats);
33
+
34
+ parseResult.pages.forEach((page) => {
35
+ if (page.items.length > 0) {
36
+ const lineItems = [];
37
+ const textItemsGroupedByLine = lineGrouper.group(page.items);
38
+ textItemsGroupedByLine.forEach((lineTextItems) => {
39
+ const lineItem = lineCompactor.compact(lineTextItems);
40
+ if (lineTextItems.length > 1) {
41
+ lineItem.annotation = ADDED_ANNOTATION;
42
+ lineTextItems.forEach((item) => {
43
+ item.annotation = REMOVED_ANNOTATION;
44
+ lineItems.push(
45
+ new LineItem({
46
+ ...item,
47
+ }),
48
+ );
49
+ });
50
+ }
51
+ if (lineItem.words.length === 0) {
52
+ lineItem.annotation = REMOVED_ANNOTATION;
53
+ }
54
+ lineItems.push(lineItem);
55
+
56
+ if (lineItem.parsedElements.formattedWords) {
57
+ formattedWords += lineItem.parsedElements.formattedWords;
58
+ }
59
+ if (lineItem.parsedElements.containLinks) {
60
+ linkCount++;
61
+ }
62
+ if (lineItem.parsedElements.footnoteLinks.length > 0) {
63
+ const footnoteLinks = lineItem.parsedElements.footnoteLinks.map(
64
+ (footnoteLink) => ({ footnoteLink, page: page.index + 1 }),
65
+ );
66
+ foundFootnoteLinks.push.apply(foundFootnoteLinks, footnoteLinks);
67
+ }
68
+ if (lineItem.parsedElements.footnotes.length > 0) {
69
+ lineItem.type = BlockType.FOOTNOTES;
70
+ const footnotes = lineItem.parsedElements.footnotes.map(
71
+ (footnote) => ({ footnote, page: page.index + 1 }),
72
+ );
73
+ foundFootnotes.push.apply(foundFootnotes, footnotes);
74
+ }
75
+ });
76
+ page.items = lineItems;
77
+ }
78
+ });
79
+
80
+ return new ParseResult({
81
+ ...parseResult,
82
+ messages: [
83
+ "Detected " + formattedWords + " formatted words",
84
+ "Found " + linkCount + " links",
85
+ "Detected " + foundFootnoteLinks.length + " footnotes links",
86
+ "Detected " + foundFootnotes.length + " footnotes",
87
+ ],
88
+ });
89
+ }
90
+ }
91
+
92
+
@@ -0,0 +1,173 @@
1
+ /**
2
+ * @description Line-item transformation that assigns H1–H6 block types based on
3
+ * font height relative to the document's body height. Handles three cases: title
4
+ * pages (items at the document's maximum height), TOC-derived headline height
5
+ * ranges when a table of contents was found, and a fallback that ranks all
6
+ * above-average heights largest-first and maps them to heading levels. Also
7
+ * detects all-uppercase body-font lines as the next lower heading level.
8
+ */
9
+ import ToLineItemTransformation from "../base/to-line-item-transform";
10
+ import ParseResult from "../../models/parse-result";
11
+ import { DETECTED_ANNOTATION } from "../../models/annotation";
12
+ import BlockType from "../../models/block-type";
13
+ import { isListItem } from "../../utils/string-functions";
14
+
15
+ // Detect headlines based on heights
16
+ export default class DetectHeaders extends ToLineItemTransformation {
17
+ constructor() {
18
+ super("Detect Headers");
19
+ }
20
+
21
+ transform(parseResult: ParseResult): ParseResult {
22
+ const {
23
+ tocPages,
24
+ headlineTypeToHeightRange,
25
+ mostUsedHeight,
26
+ mostUsedDistance,
27
+ mostUsedFont,
28
+ maxHeight,
29
+ } = parseResult.globals;
30
+ const hasToc = tocPages.length > 0;
31
+ var detectedHeaders = 0;
32
+
33
+ // Handle title pages
34
+ const pagesWithMaxHeight = findPagesWithMaxHeight(
35
+ parseResult.pages,
36
+ maxHeight,
37
+ );
38
+ const min2ndLevelHeaderHeigthOnMaxPage =
39
+ mostUsedHeight + (maxHeight - mostUsedHeight) / 4;
40
+ pagesWithMaxHeight.forEach((titlePage) => {
41
+ titlePage.items.forEach((item) => {
42
+ const height = item.height;
43
+ if (!item.type && height > min2ndLevelHeaderHeigthOnMaxPage) {
44
+ if (height === maxHeight) {
45
+ item.type = BlockType.H1;
46
+ } else {
47
+ item.type = BlockType.H2;
48
+ }
49
+ item.annotation = DETECTED_ANNOTATION;
50
+ detectedHeaders++;
51
+ }
52
+ });
53
+ });
54
+
55
+ if (hasToc) {
56
+ // Use existing headline heights to find additional headlines
57
+ const headlineTypes = Object.keys(headlineTypeToHeightRange);
58
+ headlineTypes.forEach((headlineType) => {
59
+ var range = headlineTypeToHeightRange[headlineType];
60
+ if (range.max > mostUsedHeight) {
61
+ // use only very clear headlines, only use max
62
+ parseResult.pages.forEach((page) => {
63
+ page.items.forEach((item) => {
64
+ if (!item.type && item.height === range.max) {
65
+ item.annotation = DETECTED_ANNOTATION;
66
+ item.type = BlockType.enumValueOf(headlineType);
67
+ detectedHeaders++;
68
+ }
69
+ });
70
+ });
71
+ }
72
+ });
73
+ } else {
74
+ // Categorize headlines by the text heights
75
+ const heights = [];
76
+ var lastHeight;
77
+ parseResult.pages.forEach((page) => {
78
+ page.items.forEach((item) => {
79
+ if (
80
+ !item.type &&
81
+ item.height > mostUsedHeight &&
82
+ !isListItem(item.text())
83
+ ) {
84
+ if (
85
+ !heights.includes(item.height) &&
86
+ (!lastHeight || lastHeight > item.height)
87
+ ) {
88
+ heights.push(item.height);
89
+ }
90
+ }
91
+ });
92
+ });
93
+ heights.sort((a, b) => b - a);
94
+
95
+ heights.forEach((height, i) => {
96
+ const headlineLevel = i + 2;
97
+ if (headlineLevel <= 6) {
98
+ const headlineType = BlockType.headlineByLevel(2 + i);
99
+ parseResult.pages.forEach((page) => {
100
+ page.items.forEach((item) => {
101
+ if (
102
+ !item.type &&
103
+ item.height === height &&
104
+ !isListItem(item.text())
105
+ ) {
106
+ detectedHeaders++;
107
+ item.annotation = DETECTED_ANNOTATION;
108
+ item.type = headlineType;
109
+ }
110
+ });
111
+ });
112
+ }
113
+ });
114
+ }
115
+
116
+ // find headlines which have paragraph height
117
+ var smallesHeadlineLevel = 1;
118
+ parseResult.pages.forEach((page) => {
119
+ page.items.forEach((item) => {
120
+ if (item.type && item.type.headline) {
121
+ smallesHeadlineLevel = Math.max(
122
+ smallesHeadlineLevel,
123
+ item.type.headlineLevel,
124
+ );
125
+ }
126
+ });
127
+ });
128
+ if (smallesHeadlineLevel < 6) {
129
+ const nextHeadlineType = BlockType.headlineByLevel(
130
+ smallesHeadlineLevel + 1,
131
+ );
132
+ parseResult.pages.forEach((page) => {
133
+ var lastItem;
134
+ page.items.forEach((item) => {
135
+ if (
136
+ !item.type &&
137
+ item.height === mostUsedHeight &&
138
+ item.font !== mostUsedFont &&
139
+ (!lastItem ||
140
+ lastItem.y < item.y ||
141
+ (lastItem.type && lastItem.type.headline) ||
142
+ lastItem.y - item.y > mostUsedDistance * 2) &&
143
+ item.text() === item.text().toUpperCase()
144
+ ) {
145
+ detectedHeaders++;
146
+ item.annotation = DETECTED_ANNOTATION;
147
+ item.type = nextHeadlineType;
148
+ }
149
+ lastItem = item;
150
+ });
151
+ });
152
+ }
153
+
154
+ return new ParseResult({
155
+ ...parseResult,
156
+ messages: ["Detected " + detectedHeaders + " headlines."],
157
+ });
158
+ }
159
+ }
160
+
161
+ function findPagesWithMaxHeight(pages: any[], maxHeight: number): Set<any> {
162
+ const maxHeaderPagesSet = new Set<any>();
163
+ pages.forEach((page) => {
164
+ page.items.forEach((item) => {
165
+ if (!item.type && item.height === maxHeight) {
166
+ maxHeaderPagesSet.add(page);
167
+ }
168
+ });
169
+ });
170
+ return maxHeaderPagesSet;
171
+ }
172
+
173
+
@@ -0,0 +1,68 @@
1
+ /**
2
+ * @description Line-item transformation that identifies list items starting with
3
+ * bullet characters (-, •, –) or numbered patterns (`1. text`) and assigns
4
+ * `BlockType.LIST`. Non-dash bullet characters are normalised to `-` by marking
5
+ * the original line REMOVED and inserting a replacement `LineItem` (ADDED),
6
+ * preserving the debug diff trail.
7
+ */
8
+
9
+ import ToLineItemTransformation from '../base/to-line-item-transform'
10
+ import ParseResult from '../../models/parse-result'
11
+ import LineItem from '../../models/line-item'
12
+ import Word from '../../models/word'
13
+ import { REMOVED_ANNOTATION, ADDED_ANNOTATION, DETECTED_ANNOTATION } from '../../models/annotation'
14
+ import BlockType from '../../models/block-type'
15
+ import { isListItemCharacter, isNumberedListItem } from '../../utils/string-functions'
16
+
17
+ // Detect items starting with -, \u2022, etc...
18
+ export default class DetectListItems extends ToLineItemTransformation {
19
+ constructor () {
20
+ super('Detect List Items')
21
+ }
22
+
23
+ transform(parseResult: ParseResult): ParseResult {
24
+ var foundListItems = 0
25
+ var foundNumberedItems = 0
26
+ parseResult.pages.forEach(page => {
27
+ const newItems = []
28
+ page.items.forEach(item => {
29
+ newItems.push(item)
30
+ if (!item.type) {
31
+ var text = item.text()
32
+ if (isListItemCharacter(item.words[0].string)) {
33
+ foundListItems++
34
+ if (item.words[0].string === '-') {
35
+ item.annotation = DETECTED_ANNOTATION
36
+ item.type = BlockType.LIST
37
+ } else {
38
+ item.annotation = REMOVED_ANNOTATION
39
+ const newWords = item.words.map(word => new Word({
40
+ ...word,
41
+ }))
42
+ newWords[0].string = '-'
43
+ newItems.push(new LineItem({
44
+ ...item,
45
+ words: newWords,
46
+ annotation: ADDED_ANNOTATION,
47
+ type: BlockType.LIST,
48
+ }))
49
+ }
50
+ } else if (isNumberedListItem(text)) { // TODO check that starts with 1 (kala chakra)
51
+ foundNumberedItems++
52
+ item.annotation = DETECTED_ANNOTATION
53
+ item.type = BlockType.LIST
54
+ }
55
+ }
56
+ })
57
+ page.items = newItems
58
+ })
59
+
60
+ return new ParseResult({
61
+ ...parseResult,
62
+ messages: [
63
+ 'Detected ' + foundListItems + ' plain list items.',
64
+ 'Detected ' + foundNumberedItems + ' numbered list items.',
65
+ ],
66
+ })
67
+ }
68
+ }