extract-pdf 0.1.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/README.md +16 -10
  2. package/dist/models/annotation.d.ts +20 -0
  3. package/dist/models/block-type.d.ts +10 -0
  4. package/dist/models/headline-finder.d.ts +11 -0
  5. package/dist/models/line-converter.d.ts +10 -0
  6. package/dist/models/line-item-block.d.ts +14 -0
  7. package/dist/models/line-item.d.ts +25 -0
  8. package/dist/models/metadata.d.ts +23 -0
  9. package/dist/models/page-item.d.ts +21 -0
  10. package/dist/models/page.d.ts +14 -0
  11. package/dist/models/parse-result.d.ts +24 -0
  12. package/dist/models/parsed-elements.d.ts +20 -0
  13. package/dist/models/stashing-stream.d.ts +23 -0
  14. package/dist/models/text-item-line-grouper.d.ts +8 -0
  15. package/dist/models/text-item.d.ts +29 -0
  16. package/dist/models/word.d.ts +27 -0
  17. package/dist/pdf-to-html.cjs.js +1 -1
  18. package/dist/pdf-to-html.d.ts +36 -41
  19. package/dist/pdf-to-html.es.js +1 -1
  20. package/dist/transforms/base/to-line-item-block-transform.d.ts +6 -0
  21. package/dist/transforms/base/to-line-item-transform.d.ts +6 -0
  22. package/dist/transforms/base/to-text-item-transform.d.ts +6 -0
  23. package/dist/transforms/base/transformation.d.ts +8 -0
  24. package/dist/transforms/block/detect-code-quote-blocks.d.ts +6 -0
  25. package/dist/transforms/block/detect-list-levels.d.ts +6 -0
  26. package/dist/transforms/block/gather-blocks.d.ts +6 -0
  27. package/dist/transforms/calculate-global-stats.d.ts +11 -0
  28. package/dist/transforms/line-item/compact-lines.d.ts +6 -0
  29. package/dist/transforms/line-item/detect-headers.d.ts +6 -0
  30. package/dist/transforms/line-item/detect-list-items.d.ts +6 -0
  31. package/dist/transforms/line-item/detect-toc.d.ts +6 -0
  32. package/dist/transforms/line-item/remove-repetitive-elements.d.ts +6 -0
  33. package/dist/transforms/line-item/vertical-to-horizontal.d.ts +6 -0
  34. package/dist/transforms/to-html.d.ts +6 -0
  35. package/dist/transforms/to-text-blocks.d.ts +6 -0
  36. package/dist/utils/is-url-pdf.d.ts +1 -0
  37. package/dist/utils/page-item-functions.d.ts +8 -0
  38. package/dist/utils/page-number-functions.d.ts +14 -0
  39. package/dist/utils/string-functions.d.ts +14 -0
  40. package/package.json +9 -12
  41. package/src/models/annotation.ts +41 -0
  42. package/src/models/block-type.ts +203 -0
  43. package/src/models/headline-finder.ts +53 -0
  44. package/src/models/line-converter.ts +224 -0
  45. package/src/models/line-item-block.ts +51 -0
  46. package/src/models/line-item.ts +59 -0
  47. package/src/models/metadata.ts +29 -0
  48. package/src/models/page-item.ts +36 -0
  49. package/src/models/page.ts +16 -0
  50. package/src/models/parse-result.ts +32 -0
  51. package/src/models/parsed-elements.ts +29 -0
  52. package/src/models/stashing-stream.ts +86 -0
  53. package/src/models/text-item-line-grouper.ts +41 -0
  54. package/src/models/text-item.ts +50 -0
  55. package/src/models/word.ts +31 -0
  56. package/src/pdf-to-html.ts +225 -0
  57. package/src/transforms/base/to-line-item-block-transform.ts +29 -0
  58. package/src/transforms/base/to-line-item-transform.ts +29 -0
  59. package/src/transforms/base/to-text-item-transform.ts +28 -0
  60. package/src/transforms/base/transformation.ts +35 -0
  61. package/src/transforms/block/detect-code-quote-blocks.ts +57 -0
  62. package/src/transforms/block/detect-list-levels.ts +64 -0
  63. package/src/transforms/block/gather-blocks.ts +113 -0
  64. package/src/transforms/calculate-global-stats.ts +132 -0
  65. package/src/transforms/line-item/compact-lines.ts +92 -0
  66. package/src/transforms/line-item/detect-headers.ts +173 -0
  67. package/src/transforms/line-item/detect-list-items.ts +68 -0
  68. package/src/transforms/line-item/detect-toc.ts +459 -0
  69. package/src/transforms/line-item/remove-repetitive-elements.ts +101 -0
  70. package/src/transforms/line-item/vertical-to-horizontal.ts +90 -0
  71. package/src/transforms/to-html.ts +46 -0
  72. package/src/transforms/to-text-blocks.ts +38 -0
  73. package/src/utils/is-url-pdf.ts +33 -0
  74. package/src/utils/page-item-functions.ts +35 -0
  75. package/src/utils/page-number-functions.ts +109 -0
  76. package/src/utils/string-functions.ts +124 -0
@@ -0,0 +1,459 @@
1
+ /**
2
+ * @description Line-item transformation that identifies Table of Contents pages
3
+ * (≥75% of lines end with a page number) and extracts TOC links with hierarchy
4
+ * levels determined by x-position or font differences. Then cross-references
5
+ * each TOC link against document pages via `HeadlineFinder` to tag the actual
6
+ * heading text with H2–H6 types, falling back to font-height range matching for
7
+ * headings that could not be located by exact text.
8
+ */
9
+
10
+ import ToLineItemTransformation from "../base/to-line-item-transform";
11
+ import ParseResult from "../../models/parse-result";
12
+ import LineItem from "../../models/line-item";
13
+ import Word from "../../models/word";
14
+ import HeadlineFinder from "../../models/headline-finder";
15
+ import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
16
+ import BlockType from "../../models/block-type";
17
+ import {
18
+ isDigit,
19
+ isNumber,
20
+ wordMatch,
21
+ hasOnly,
22
+ } from "../../utils/string-functions";
23
+
24
+ // Detect table of contents pages plus linked headlines
25
+ export default class DetectTOC extends ToLineItemTransformation {
26
+ constructor() {
27
+ super("Detect TOC");
28
+ }
29
+
30
+ transform(parseResult: ParseResult): ParseResult {
31
+ const tocPages = [];
32
+ const maxPagesToEvaluate = Math.min(20, parseResult.pages.length);
33
+ const linkLeveler = new LinkLeveler();
34
+
35
+ var tocLinks = [];
36
+ var lastTocPage;
37
+ var headlineItem;
38
+ parseResult.pages.slice(0, maxPagesToEvaluate).forEach((page) => {
39
+ var lineItemsWithDigits = 0;
40
+ const unknownLines = new Set();
41
+ const pageTocLinks = [];
42
+ var lastWordsWithoutNumber;
43
+ var lastLine;
44
+ // find lines with words containing only "." ...
45
+ const tocLines = page.items.filter((line) =>
46
+ line.words.includes((word) => hasOnly(word.string, ".")),
47
+ );
48
+ // ... and ending with a number per page
49
+ tocLines.forEach((line) => {
50
+ var words = line.words.filter((word) => !hasOnly(word.string, "."));
51
+ const digits = [];
52
+ while (words.length > 0 && isNumber(words[words.length - 1].string)) {
53
+ const lastWord = words.pop();
54
+ digits.unshift(lastWord.string);
55
+ }
56
+
57
+ if (digits.length === 0 && words.length > 0) {
58
+ const lastWord = words[words.length - 1];
59
+ while (
60
+ isDigit(lastWord.string.charCodeAt(lastWord.string.length - 1))
61
+ ) {
62
+ digits.unshift(lastWord.string.charAt(lastWord.string.length - 1));
63
+ lastWord.string = lastWord.string.substring(
64
+ 0,
65
+ lastWord.string.length - 1,
66
+ );
67
+ }
68
+ }
69
+ var endsWithDigit = digits.length > 0;
70
+ if (endsWithDigit) {
71
+ endsWithDigit = true;
72
+ if (lastWordsWithoutNumber) {
73
+ // 2-line item ?
74
+ words.push(...lastWordsWithoutNumber);
75
+ lastWordsWithoutNumber = null;
76
+ }
77
+ pageTocLinks.push(
78
+ new TocLink({
79
+ pageNumber: parseInt(digits.join("")),
80
+ lineItem: new LineItem({ ...line, words }),
81
+ }),
82
+ );
83
+ lineItemsWithDigits++;
84
+ } else {
85
+ if (!headlineItem) {
86
+ headlineItem = line;
87
+ } else {
88
+ if (lastWordsWithoutNumber) {
89
+ unknownLines.add(lastLine);
90
+ }
91
+ lastWordsWithoutNumber = words;
92
+ lastLine = line;
93
+ }
94
+ }
95
+ });
96
+
97
+ // page has been processed
98
+ if ((lineItemsWithDigits * 100) / page.items.length > 75) {
99
+ tocPages.push(page.index + 1);
100
+ lastTocPage = page;
101
+ linkLeveler.levelPageItems(pageTocLinks);
102
+ tocLinks.push(...pageTocLinks);
103
+
104
+ const newBlocks = [];
105
+ page.items.forEach((line) => {
106
+ if (!unknownLines.has(line)) {
107
+ line.annotation = REMOVED_ANNOTATION;
108
+ }
109
+ newBlocks.push(line);
110
+ if (line === headlineItem) {
111
+ newBlocks.push(
112
+ new LineItem({
113
+ ...line,
114
+ type: BlockType.H2,
115
+ annotation: ADDED_ANNOTATION,
116
+ }),
117
+ );
118
+ }
119
+ });
120
+ page.items = newBlocks;
121
+ } else {
122
+ headlineItem = null;
123
+ }
124
+ });
125
+
126
+ // all pages have been processed
127
+ var foundHeadlines = tocLinks.length;
128
+ const notFoundHeadlines = [];
129
+ const foundBySize = [];
130
+ const headlineTypeToHeightRange = {}; // H1={min:23, max:25}
131
+
132
+ if (tocPages.length > 0) {
133
+ // Add TOC items
134
+ tocLinks.forEach((tocLink) => {
135
+ lastTocPage.items.push(
136
+ new LineItem({
137
+ words: [
138
+ new Word({
139
+ string: " ".repeat(tocLink.level * 3) + "-",
140
+ }),
141
+ ].concat(tocLink.lineItem.words),
142
+ type: BlockType.TOC,
143
+ annotation: ADDED_ANNOTATION,
144
+ }),
145
+ );
146
+ });
147
+
148
+ // Add linked headers
149
+ const pageMapping = detectPageMappingNumber(
150
+ parseResult.pages.filter((page) => page.index > lastTocPage.index),
151
+ tocLinks,
152
+ );
153
+ tocLinks.forEach((tocLink) => {
154
+ var linkedPage = parseResult.pages[tocLink.pageNumber + pageMapping];
155
+ var foundHealineItems;
156
+ if (linkedPage) {
157
+ foundHealineItems = findHeadlineItems(
158
+ linkedPage,
159
+ tocLink.lineItem.text(),
160
+ );
161
+ if (!foundHealineItems) {
162
+ // pages are off by 1 ?
163
+ linkedPage =
164
+ parseResult.pages[tocLink.pageNumber + pageMapping + 1];
165
+ if (linkedPage) {
166
+ foundHealineItems = findHeadlineItems(
167
+ linkedPage,
168
+ tocLink.lineItem.text(),
169
+ );
170
+ }
171
+ }
172
+ }
173
+ if (foundHealineItems) {
174
+ addHeadlineItems(
175
+ linkedPage,
176
+ tocLink,
177
+ foundHealineItems,
178
+ headlineTypeToHeightRange,
179
+ );
180
+ } else {
181
+ notFoundHeadlines.push(tocLink);
182
+ }
183
+ });
184
+
185
+ // Try to find linked headers by height
186
+ var fromPage = lastTocPage.index + 2;
187
+ var lastNotFound = [];
188
+ const rollupLastNotFound = (currentPageNumber) => {
189
+ if (lastNotFound.length > 0) {
190
+ lastNotFound.forEach((notFoundTocLink) => {
191
+ const headlineType = BlockType.headlineByLevel(
192
+ notFoundTocLink.level + 2,
193
+ );
194
+ const heightRange = headlineTypeToHeightRange[headlineType.name];
195
+ if (heightRange) {
196
+ const [pageIndex, lineIndex] = findPageAndLineFromHeadline(
197
+ parseResult.pages,
198
+ notFoundTocLink,
199
+ heightRange,
200
+ fromPage,
201
+ currentPageNumber,
202
+ );
203
+ if (lineIndex > -1) {
204
+ const page = parseResult.pages[pageIndex];
205
+ page.items[lineIndex].annotation = REMOVED_ANNOTATION;
206
+ page.items.splice(
207
+ lineIndex + 1,
208
+ 0,
209
+ new LineItem({
210
+ ...notFoundTocLink.lineItem,
211
+ type: headlineType,
212
+ annotation: ADDED_ANNOTATION,
213
+ }),
214
+ );
215
+ foundBySize.push(notFoundTocLink);
216
+ }
217
+ }
218
+ });
219
+ lastNotFound = [];
220
+ }
221
+ };
222
+ if (notFoundHeadlines.length > 0) {
223
+ tocLinks.forEach((tocLink) => {
224
+ if (notFoundHeadlines.includes(tocLink)) {
225
+ lastNotFound.push(tocLink);
226
+ } else {
227
+ rollupLastNotFound(tocLink.pageNumber);
228
+ fromPage = tocLink.pageNumber;
229
+ }
230
+ });
231
+ if (lastNotFound.length > 0) {
232
+ rollupLastNotFound(parseResult.pages.length);
233
+ }
234
+ }
235
+ }
236
+
237
+ const messages = [];
238
+ messages.push("Detected " + tocPages.length + " table of content pages");
239
+ if (tocPages.length > 0) {
240
+ messages.push(
241
+ "TOC headline heights: " + JSON.stringify(headlineTypeToHeightRange),
242
+ );
243
+ messages.push(
244
+ "Found TOC headlines: " +
245
+ (foundHeadlines - notFoundHeadlines.length + foundBySize.length) +
246
+ "/" +
247
+ foundHeadlines,
248
+ );
249
+ }
250
+ if (notFoundHeadlines.length > 0) {
251
+ messages.push(
252
+ "Found TOC headlines (by size): " +
253
+ foundBySize.map((tocLink) => tocLink.lineItem.text()),
254
+ );
255
+ messages.push(
256
+ "Missing TOC headlines: " +
257
+ notFoundHeadlines
258
+ .filter((fTocLink) => !foundBySize.includes(fTocLink))
259
+ .map(
260
+ (tocLink) => tocLink.lineItem.text() + "=>" + tocLink.pageNumber,
261
+ ),
262
+ );
263
+ }
264
+ return new ParseResult({
265
+ ...parseResult,
266
+ globals: {
267
+ ...parseResult.globals,
268
+ tocPages,
269
+ headlineTypeToHeightRange,
270
+ },
271
+ messages,
272
+ });
273
+ }
274
+ }
275
+
276
+ // Find out how the TOC page link actualy translates to the page.index
277
+ function detectPageMappingNumber(pages, tocLinks) {
278
+ for (var tocLink of tocLinks) {
279
+ const page = findPageWithHeadline(pages, tocLink.lineItem.text());
280
+ if (page) {
281
+ return page.index - tocLink.pageNumber;
282
+ }
283
+ }
284
+ return null;
285
+ }
286
+
287
+ function findPageWithHeadline(pages, headline) {
288
+ for (var page of pages) {
289
+ if (findHeadlineItems(page, headline)) {
290
+ return page;
291
+ }
292
+ }
293
+ return null;
294
+ }
295
+
296
+ function findHeadlineItems(page, headline) {
297
+ const headlineFinder = new HeadlineFinder({ headline });
298
+ var lineIndex = 0;
299
+ for (var line of page.items) {
300
+ const headlineItems = headlineFinder.consume(line);
301
+ if (headlineItems) {
302
+ return { lineIndex, headlineItems };
303
+ }
304
+ lineIndex++;
305
+ }
306
+ return null;
307
+ }
308
+
309
+ function addHeadlineItems(
310
+ page,
311
+ tocLink,
312
+ foundItems,
313
+ headlineTypeToHeightRange,
314
+ ) {
315
+ foundItems.headlineItems.forEach(
316
+ (item) => (item.annotation = REMOVED_ANNOTATION),
317
+ );
318
+ const headlineType = BlockType.headlineByLevel(tocLink.level + 2);
319
+ const headlineHeight = foundItems.headlineItems.reduce(
320
+ (max, item) => Math.max(max, item.height),
321
+ 0,
322
+ );
323
+ page.items.splice(
324
+ foundItems.lineIndex + 1,
325
+ 0,
326
+ new LineItem({
327
+ ...foundItems.headlineItems[0],
328
+ words: tocLink.lineItem.words,
329
+ height: headlineHeight,
330
+ type: headlineType,
331
+ annotation: ADDED_ANNOTATION,
332
+ }),
333
+ );
334
+ var range = headlineTypeToHeightRange[headlineType.name];
335
+ if (range) {
336
+ range.min = Math.min(range.min, headlineHeight);
337
+ range.max = Math.max(range.max, headlineHeight);
338
+ } else {
339
+ range = {
340
+ min: headlineHeight,
341
+ max: headlineHeight,
342
+ };
343
+ headlineTypeToHeightRange[headlineType.name] = range;
344
+ }
345
+ }
346
+
347
+ function findPageAndLineFromHeadline(
348
+ pages,
349
+ tocLink,
350
+ heightRange,
351
+ fromPage,
352
+ toPage,
353
+ ) {
354
+ const linkText = tocLink.lineItem.text().toUpperCase();
355
+ for (var i = fromPage; i <= toPage; i++) {
356
+ const page = pages[i - 1];
357
+ if (page) {
358
+ const lineIndex = page.items.findIndex((line) => {
359
+ if (
360
+ !line.type &&
361
+ !line.annotation &&
362
+ line.height >= heightRange.min &&
363
+ line.height <= heightRange.max
364
+ ) {
365
+ const match = wordMatch(linkText, line.text());
366
+ return match >= 0.5;
367
+ }
368
+ return false;
369
+ });
370
+ if (lineIndex > -1) return [i - 1, lineIndex];
371
+ }
372
+ }
373
+ return [-1, -1];
374
+ }
375
+
376
+ class LinkLeveler {
377
+ levelByMethod: any;
378
+ uniqueFonts: any[];
379
+
380
+ constructor() {
381
+ this.levelByMethod = null;
382
+ this.uniqueFonts = [];
383
+ }
384
+
385
+ levelPageItems(tocLinks /*: TocLink[] */) {
386
+ if (!this.levelByMethod) {
387
+ const uniqueX = this.calculateUniqueX(tocLinks);
388
+ if (uniqueX.length > 1) {
389
+ this.levelByMethod = this.levelByXDiff;
390
+ } else {
391
+ const uniqueFonts = this.calculateUniqueFonts(tocLinks);
392
+ if (uniqueFonts.length > 1) {
393
+ this.uniqueFonts = uniqueFonts;
394
+ this.levelByMethod = this.levelByFont;
395
+ } else {
396
+ this.levelByMethod = this.levelToZero;
397
+ }
398
+ }
399
+ }
400
+ this.levelByMethod(tocLinks);
401
+ }
402
+
403
+ levelByXDiff(tocLinks) {
404
+ const uniqueX = this.calculateUniqueX(tocLinks);
405
+ tocLinks.forEach((link) => {
406
+ link.level = uniqueX.indexOf(link.lineItem.x);
407
+ });
408
+ }
409
+
410
+ levelByFont(tocLinks) {
411
+ tocLinks.forEach((link) => {
412
+ link.level = this.uniqueFonts.indexOf(link.lineItem.font);
413
+ });
414
+ }
415
+
416
+ levelToZero(tocLinks) {
417
+ tocLinks.forEach((link) => {
418
+ link.level = 0;
419
+ });
420
+ }
421
+
422
+ calculateUniqueX(tocLinks) {
423
+ var uniqueX = tocLinks.reduce(function (uniquesArray, link) {
424
+ if (uniquesArray.indexOf(link.lineItem.x) < 0)
425
+ uniquesArray.push(link.lineItem.x);
426
+ return uniquesArray;
427
+ }, []);
428
+
429
+ uniqueX.sort((a, b) => {
430
+ return a - b;
431
+ });
432
+
433
+ return uniqueX;
434
+ }
435
+
436
+ calculateUniqueFonts(tocLinks) {
437
+ var uniqueFont = tocLinks.reduce(function (uniquesArray, link) {
438
+ if (uniquesArray.indexOf(link.lineItem.font) < 0)
439
+ uniquesArray.push(link.lineItem.font);
440
+ return uniquesArray;
441
+ }, []);
442
+
443
+ return uniqueFont;
444
+ }
445
+ }
446
+
447
+ class TocLink {
448
+ lineItem: any;
449
+ pageNumber: number;
450
+ level: number;
451
+
452
+ constructor(options: any) {
453
+ this.lineItem = options.lineItem;
454
+ this.pageNumber = options.pageNumber;
455
+ this.level = 0;
456
+ }
457
+ }
458
+
459
+
@@ -0,0 +1,101 @@
1
+ /**
2
+ * @description Line-item transformation that removes running page headers and
3
+ * footers. Hashes the topmost and bottommost line of each page (ignoring digits
4
+ * and whitespace so page numbers don't break the match), then marks lines whose
5
+ * hash appears on more than two-thirds of all pages with `REMOVED_ANNOTATION`.
6
+ */
7
+ import ToLineItemTransformation from '../base/to-line-item-transform'
8
+ import ParseResult from '../../models/parse-result'
9
+ import { REMOVED_ANNOTATION } from '../../models/annotation'
10
+
11
+ import { isDigit } from '../../utils/string-functions'
12
+
13
+ function hashCodeIgnoringSpacesAndNumbers(string: string): number {
14
+ var hash = 0
15
+ if (string.trim().length === 0) return hash
16
+ for (var i = 0; i < string.length; i++) {
17
+ const charCode = string.charCodeAt(i)
18
+ if (!isDigit(charCode) && charCode !== 32 && charCode !== 160) {
19
+ hash = ((hash << 5) - hash) + charCode
20
+ hash |= 0 // Convert to 32bit integer
21
+ }
22
+ }
23
+ return hash
24
+ }
25
+
26
+ // Remove elements with similar content on same page positions, like page numbers, licenes information, etc...
27
+ export default class RemoveRepetitiveElements extends ToLineItemTransformation {
28
+ constructor () {
29
+ super('Remove Repetitive Elements')
30
+ }
31
+
32
+ // The idea is the following:
33
+ // - For each page, collect all items of the first, and all items of the last line
34
+ // - Calculate how often these items occur accros all pages (hash ignoring numbers, whitespace, upper/lowercase)
35
+ // - Delete items occuring on more then 2/3 of all pages
36
+ transform(parseResult: ParseResult): ParseResult {
37
+ // find first and last lines per page
38
+ const pageStore = []
39
+ const minLineHashRepetitions = {}
40
+ const maxLineHashRepetitions = {}
41
+ parseResult.pages.forEach(page => {
42
+ const minMaxItems = page.items.reduce((itemStore, item) => {
43
+ if (item.y < itemStore.minY) {
44
+ itemStore.minElements = [item]
45
+ itemStore.minY = item.y
46
+ } else if (item.y === itemStore.minY) {
47
+ itemStore.minElements.push(item)
48
+ }
49
+ if (item.y > itemStore.maxY) {
50
+ itemStore.maxElements = [item]
51
+ itemStore.maxY = item.y
52
+ } else if (item.y === itemStore.maxY) {
53
+ itemStore.maxElements.push(item)
54
+ }
55
+ return itemStore
56
+ }, {
57
+ minY: 999,
58
+ maxY: 0,
59
+ minElements: [],
60
+ maxElements: [],
61
+ })
62
+
63
+ const minLineHash = hashCodeIgnoringSpacesAndNumbers(minMaxItems.minElements.reduce((combinedString, item) => combinedString + item.text().toUpperCase(), ''))
64
+ const maxLineHash = hashCodeIgnoringSpacesAndNumbers(minMaxItems.maxElements.reduce((combinedString, item) => combinedString + item.text().toUpperCase(), ''))
65
+ pageStore.push({
66
+ minElements: minMaxItems.minElements,
67
+ maxElements: minMaxItems.maxElements,
68
+ minLineHash: minLineHash,
69
+ maxLineHash: maxLineHash,
70
+ })
71
+ minLineHashRepetitions[minLineHash] = minLineHashRepetitions[minLineHash] ? minLineHashRepetitions[minLineHash] + 1 : 1
72
+ maxLineHashRepetitions[maxLineHash] = maxLineHashRepetitions[maxLineHash] ? maxLineHashRepetitions[maxLineHash] + 1 : 1
73
+ })
74
+
75
+ // now annoate all removed items
76
+ var removedHeader = 0
77
+ var removedFooter = 0
78
+ parseResult.pages.forEach((page, i) => {
79
+ if (minLineHashRepetitions[pageStore[i].minLineHash] >= Math.max(3, parseResult.pages.length * 2 / 3)) {
80
+ pageStore[i].minElements.forEach(item => {
81
+ item.annotation = REMOVED_ANNOTATION
82
+ })
83
+ removedFooter++
84
+ }
85
+ if (maxLineHashRepetitions[pageStore[i].maxLineHash] >= Math.max(3, parseResult.pages.length * 2 / 3)) {
86
+ pageStore[i].maxElements.forEach(item => {
87
+ item.annotation = REMOVED_ANNOTATION
88
+ })
89
+ removedHeader++
90
+ }
91
+ })
92
+
93
+ return new ParseResult({
94
+ ...parseResult,
95
+ messages: [
96
+ 'Removed Header: ' + removedHeader,
97
+ 'Removed Footers: ' + removedFooter,
98
+ ],
99
+ })
100
+ }
101
+ }
@@ -0,0 +1,90 @@
1
+ /**
2
+ * @description Line-item transformation that recovers rotated sidebar text.
3
+ * PDF extracts vertically-oriented labels as a sequence of single-character
4
+ * lines. `VerticalsStream` (a `StashingStream` subclass) detects runs of 6+
5
+ * such lines and merges them into one horizontal `LineItem`, combining their
6
+ * words and computing the correct bounding box.
7
+ */
8
+ import ToLineItemTransformation from "../base/to-line-item-transform";
9
+ import ParseResult from "../../models/parse-result";
10
+ import LineItem from "../../models/line-item";
11
+ import StashingStream from "../../models/stashing-stream";
12
+ import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
13
+
14
+ // Converts vertical text to horizontal
15
+ export default class VerticalToHorizontal extends ToLineItemTransformation {
16
+ constructor() {
17
+ super("Vertical to Horizontal Text");
18
+ }
19
+
20
+ transform(parseResult: ParseResult): ParseResult {
21
+ var foundVerticals = 0;
22
+ parseResult.pages.forEach((page) => {
23
+ const stream = new VerticalsStream();
24
+ stream.consumeAll(page.items);
25
+ page.items = stream.complete();
26
+ foundVerticals += stream.foundVerticals;
27
+ });
28
+
29
+ return new ParseResult({
30
+ ...parseResult,
31
+ messages: ["Converted " + foundVerticals + " verticals"],
32
+ });
33
+ }
34
+ }
35
+
36
+ class VerticalsStream extends StashingStream {
37
+ foundVerticals: number;
38
+
39
+ constructor() {
40
+ super();
41
+ this.foundVerticals = 0;
42
+ }
43
+
44
+ shouldStash(item: any): boolean {
45
+ return item.words.length === 1 && item.words[0].string.length === 1;
46
+ }
47
+
48
+ doMatchesStash(lastItem: any, item: any): boolean {
49
+ return (
50
+ lastItem.y - item.y > 5 && lastItem.words[0].type === item.words[0].type
51
+ );
52
+ }
53
+
54
+ doFlushStash(stash: any[], results: any[]): void {
55
+ if (stash.length > 5) {
56
+ // unite
57
+ var combinedWords = [];
58
+ var minX = 999;
59
+ var maxY = 0;
60
+ var sumWidth = 0;
61
+ var maxHeight = 0;
62
+ stash.forEach((oneCharacterLine) => {
63
+ oneCharacterLine.annotation = REMOVED_ANNOTATION;
64
+ results.push(oneCharacterLine);
65
+ combinedWords.push(oneCharacterLine.words[0]);
66
+ minX = Math.min(minX, oneCharacterLine.x);
67
+ maxY = Math.max(maxY, oneCharacterLine.y);
68
+ sumWidth += oneCharacterLine.width;
69
+ maxHeight = Math.max(maxHeight, oneCharacterLine.height);
70
+ });
71
+ results.push(
72
+ new LineItem({
73
+ ...stash[0],
74
+ x: minX,
75
+ y: maxY,
76
+ width: sumWidth,
77
+ height: maxHeight,
78
+ words: combinedWords,
79
+ annotation: ADDED_ANNOTATION,
80
+ }),
81
+ );
82
+ this.foundVerticals++;
83
+ } else {
84
+ // add as singles
85
+ results.push(...stash);
86
+ }
87
+ }
88
+ }
89
+
90
+
@@ -0,0 +1,46 @@
1
+ /**
2
+ * @description Final serialization stage that converts each page's text block
3
+ * items into `<p>`-wrapped HTML. TOC blocks are emitted verbatim; hyphenated
4
+ * line-break artifacts are stripped from non-list blocks; `<code>` fencing is
5
+ * removed from CODE-category blocks (treated as prose in this simplified output).
6
+ */
7
+
8
+ import Transformation from './base/transformation'
9
+ import ParseResult from '../models/parse-result'
10
+
11
+ export default class ToHTML extends Transformation {
12
+ constructor () {
13
+ super('To HTML', 'String')
14
+ }
15
+ transform(parseResult: ParseResult): ParseResult {
16
+ parseResult.pages.forEach(page => {
17
+ var text = ''
18
+ page.items.forEach(block => {
19
+ // Concatenate all words in the same block, unless it's a Table of Contents block
20
+ let concatText
21
+ if (block.category === 'TOC') {
22
+ concatText = block.text
23
+ } else {
24
+ concatText = block.text.replace(/(\r\n|\n|\r)/gm, '\n')
25
+ }
26
+
27
+ // Concatenate words that were previously broken up by newline
28
+ if (block.category !== 'LIST') {
29
+ concatText = concatText.split('- ').join('')
30
+ }
31
+
32
+ // Assume there are no code blocks in our documents
33
+ if (block.category === 'CODE') {
34
+ concatText = concatText.split('<code>').join('').split('</code>').join('')
35
+ }
36
+
37
+ text += `<p>${concatText}</p>\n\n`
38
+ })
39
+
40
+ page.items = [text]
41
+ })
42
+ return new ParseResult({
43
+ ...parseResult,
44
+ })
45
+ }
46
+ }