extract-pdf 0.1.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/README.md +16 -10
  2. package/dist/models/annotation.d.ts +20 -0
  3. package/dist/models/block-type.d.ts +10 -0
  4. package/dist/models/headline-finder.d.ts +11 -0
  5. package/dist/models/line-converter.d.ts +10 -0
  6. package/dist/models/line-item-block.d.ts +14 -0
  7. package/dist/models/line-item.d.ts +25 -0
  8. package/dist/models/metadata.d.ts +23 -0
  9. package/dist/models/page-item.d.ts +21 -0
  10. package/dist/models/page.d.ts +14 -0
  11. package/dist/models/parse-result.d.ts +24 -0
  12. package/dist/models/parsed-elements.d.ts +20 -0
  13. package/dist/models/stashing-stream.d.ts +23 -0
  14. package/dist/models/text-item-line-grouper.d.ts +8 -0
  15. package/dist/models/text-item.d.ts +29 -0
  16. package/dist/models/word.d.ts +27 -0
  17. package/dist/pdf-to-html.cjs.js +1 -1
  18. package/dist/pdf-to-html.d.ts +36 -41
  19. package/dist/pdf-to-html.es.js +1 -1
  20. package/dist/transforms/base/to-line-item-block-transform.d.ts +6 -0
  21. package/dist/transforms/base/to-line-item-transform.d.ts +6 -0
  22. package/dist/transforms/base/to-text-item-transform.d.ts +6 -0
  23. package/dist/transforms/base/transformation.d.ts +8 -0
  24. package/dist/transforms/block/detect-code-quote-blocks.d.ts +6 -0
  25. package/dist/transforms/block/detect-list-levels.d.ts +6 -0
  26. package/dist/transforms/block/gather-blocks.d.ts +6 -0
  27. package/dist/transforms/calculate-global-stats.d.ts +11 -0
  28. package/dist/transforms/line-item/compact-lines.d.ts +6 -0
  29. package/dist/transforms/line-item/detect-headers.d.ts +6 -0
  30. package/dist/transforms/line-item/detect-list-items.d.ts +6 -0
  31. package/dist/transforms/line-item/detect-toc.d.ts +6 -0
  32. package/dist/transforms/line-item/remove-repetitive-elements.d.ts +6 -0
  33. package/dist/transforms/line-item/vertical-to-horizontal.d.ts +6 -0
  34. package/dist/transforms/to-html.d.ts +6 -0
  35. package/dist/transforms/to-text-blocks.d.ts +6 -0
  36. package/dist/utils/is-url-pdf.d.ts +1 -0
  37. package/dist/utils/page-item-functions.d.ts +8 -0
  38. package/dist/utils/page-number-functions.d.ts +14 -0
  39. package/dist/utils/string-functions.d.ts +14 -0
  40. package/package.json +9 -12
  41. package/src/models/annotation.ts +41 -0
  42. package/src/models/block-type.ts +203 -0
  43. package/src/models/headline-finder.ts +53 -0
  44. package/src/models/line-converter.ts +224 -0
  45. package/src/models/line-item-block.ts +51 -0
  46. package/src/models/line-item.ts +59 -0
  47. package/src/models/metadata.ts +29 -0
  48. package/src/models/page-item.ts +36 -0
  49. package/src/models/page.ts +16 -0
  50. package/src/models/parse-result.ts +32 -0
  51. package/src/models/parsed-elements.ts +29 -0
  52. package/src/models/stashing-stream.ts +86 -0
  53. package/src/models/text-item-line-grouper.ts +41 -0
  54. package/src/models/text-item.ts +50 -0
  55. package/src/models/word.ts +31 -0
  56. package/src/pdf-to-html.ts +225 -0
  57. package/src/transforms/base/to-line-item-block-transform.ts +29 -0
  58. package/src/transforms/base/to-line-item-transform.ts +29 -0
  59. package/src/transforms/base/to-text-item-transform.ts +28 -0
  60. package/src/transforms/base/transformation.ts +35 -0
  61. package/src/transforms/block/detect-code-quote-blocks.ts +57 -0
  62. package/src/transforms/block/detect-list-levels.ts +64 -0
  63. package/src/transforms/block/gather-blocks.ts +113 -0
  64. package/src/transforms/calculate-global-stats.ts +132 -0
  65. package/src/transforms/line-item/compact-lines.ts +92 -0
  66. package/src/transforms/line-item/detect-headers.ts +173 -0
  67. package/src/transforms/line-item/detect-list-items.ts +68 -0
  68. package/src/transforms/line-item/detect-toc.ts +459 -0
  69. package/src/transforms/line-item/remove-repetitive-elements.ts +101 -0
  70. package/src/transforms/line-item/vertical-to-horizontal.ts +90 -0
  71. package/src/transforms/to-html.ts +46 -0
  72. package/src/transforms/to-text-blocks.ts +38 -0
  73. package/src/utils/is-url-pdf.ts +33 -0
  74. package/src/utils/page-item-functions.ts +35 -0
  75. package/src/utils/page-number-functions.ts +109 -0
  76. package/src/utils/string-functions.ts +124 -0
@@ -0,0 +1,38 @@
1
+ /**
2
+ * @description Flattens each page's parsed block items into plain text objects,
3
+ * replacing structured PDF block nodes with `{ category, text }` pairs. The
4
+ * `category` is derived from the block's `BlockType` name (e.g. "Paragraph",
5
+ * "Heading", "List") or falls back to "Unknown" when no type is assigned.
6
+ * `text` is extracted via `BlockType.blockToText`, which concatenates all
7
+ * inline text spans within the block. The result is a new `ParseResult` whose
8
+ * pages contain only these lightweight text items, discarding geometry and
9
+ * style metadata — making downstream NLP processing and serialization simpler.
10
+ */
11
+ import Transformation from "./base/transformation";
12
+ import ParseResult from "../models/parse-result";
13
+ import BlockType from "../models/block-type";
14
+
15
+ export default class ToTextBlocks extends Transformation {
16
+ constructor() {
17
+ super("To Text Blocks", "TextBlock");
18
+ }
19
+
20
+ transform(parseResult: ParseResult): ParseResult {
21
+ parseResult.pages.forEach((page) => {
22
+ const textItems: Array<{ category: string; text: string }> = [];
23
+ page.items.forEach((block) => {
24
+ // TODO category to type (before have no unknowns, have paragraph)
25
+ const category = block.type ? block.type.name : "Unknown";
26
+ textItems.push({
27
+ category: category,
28
+ text: BlockType.blockToText(block),
29
+ });
30
+ });
31
+ page.items = textItems;
32
+ });
33
+ return new ParseResult({
34
+ ...parseResult,
35
+ });
36
+ }
37
+ }
38
+
@@ -0,0 +1,33 @@
1
+ /**
2
+ * Detects if a given URL points to a PDF file by checking
3
+ * the stream's first bytes for %PDF- then ends the request.
4
+ * Useful for hidden pdf url that does not end with pdf
5
+ * @category Extract
6
+ */
7
+ import grab from "grab-url";
8
+
9
+ export async function isUrlPDF(url: string) {
10
+ try {
11
+ // Fetch the URL as an arraybuffer
12
+ const buffer = await grab(url, {
13
+ responseType: "arraybuffer",
14
+ timeout: 10,
15
+ });
16
+
17
+ if (!buffer || buffer.byteLength < 5) return false;
18
+
19
+ const chunk = new Uint8Array(buffer);
20
+
21
+ // Check if the bytes match the PDF signature
22
+ return (
23
+ chunk[0] === 0x25 && // %
24
+ chunk[1] === 0x50 && // P
25
+ chunk[2] === 0x44 && // D
26
+ chunk[3] === 0x46 && // F
27
+ chunk[4] === 0x2d
28
+ ); // -
29
+ } catch (error) {
30
+ console.error("Error checking URL:", error);
31
+ return false;
32
+ }
33
+ }
@@ -0,0 +1,35 @@
1
+ /**
2
+ * @description Geometry utility functions shared across the transformation pipeline.
3
+ * `minXFromBlocks` finds the minimum x across all items inside an array of blocks;
4
+ * `minXFromPageItems` finds the minimum x across a flat page item array; `sortByX`
5
+ * sorts items in place by x coordinate for left-to-right reading order.
6
+ */
7
+ import LineItemBlock from '../models/line-item-block'
8
+
9
+ export function minXFromBlocks(blocks: LineItemBlock[]): number | null {
10
+ var minX = 999
11
+ blocks.forEach(block => {
12
+ block.items.forEach(item => {
13
+ minX = Math.min(minX, item.x)
14
+ })
15
+ })
16
+ if (minX === 999) {
17
+ return null
18
+ }
19
+ return minX
20
+ }
21
+
22
+ export function minXFromPageItems(items: Array<{ x: number }>): number | null {
23
+ var minX = 999
24
+ items.forEach(item => {
25
+ minX = Math.min(minX, item.x)
26
+ })
27
+ if (minX === 999) {
28
+ return null
29
+ }
30
+ return minX
31
+ }
32
+
33
+ export function sortByX(items: Array<{ x: number }>): void {
34
+ items.sort((a, b) => a.x - b.x)
35
+ }
@@ -0,0 +1,109 @@
1
+ /**
2
+ * @description Detects and strips printed page numbers from PDF text content.
3
+ * Scans the top and bottom sixths of each page's item list for standalone numeric
4
+ * strings, builds a `pageIndex → pageNum` map, finds the first page where numbers
5
+ * begin incrementing consecutively, and provides `removePageNumber` to filter
6
+ * those items out before further processing.
7
+ */
8
+ import {
9
+ removeLeadingWhitespaces,
10
+ removeTrailingWhitespaces,
11
+ isNumber,
12
+ } from "./string-functions";
13
+
14
+ interface TextContentItem {
15
+ str: string;
16
+ }
17
+
18
+ interface TextContent {
19
+ items: TextContentItem[];
20
+ }
21
+
22
+ type PageIndexNumMap = Record<string, number[]>;
23
+
24
+ const searchRange = (numerator: number, denominator: number, length: number): number => {
25
+ return Math.floor((numerator / denominator) * length);
26
+ };
27
+
28
+ const searchArea = (range: TextContentItem[], pageIndexNumMap: PageIndexNumMap, pageIndex: number): PageIndexNumMap => {
29
+ for (const { str } of range) {
30
+ const trimLeadingWhitespaces = removeLeadingWhitespaces(str);
31
+ const trimWhitespaces = removeTrailingWhitespaces(trimLeadingWhitespaces);
32
+ if (isNumber(trimWhitespaces)) {
33
+ if (!pageIndexNumMap[pageIndex]) {
34
+ pageIndexNumMap[pageIndex] = [];
35
+ }
36
+ pageIndexNumMap[pageIndex].push(Number(trimWhitespaces));
37
+ }
38
+ }
39
+ return pageIndexNumMap;
40
+ };
41
+
42
+ const findPageNumbers = (pageIndexNumMap: PageIndexNumMap, pageIndex: number, items: TextContentItem[]): PageIndexNumMap => {
43
+ const topArea = searchRange(1, 6, items.length);
44
+ const bottomArea = searchRange(5, 6, items.length);
45
+
46
+ const topAreaResult = searchArea(
47
+ items.slice(0, topArea),
48
+ pageIndexNumMap,
49
+ pageIndex,
50
+ );
51
+ return searchArea(items.slice(bottomArea), topAreaResult, pageIndex);
52
+ };
53
+
54
+ const findFirstPage = (pageIndexNumMap: PageIndexNumMap): { pageIndex: number; pageNum: number } | undefined => {
55
+ let counter = 0;
56
+ const keys = Object.keys(pageIndexNumMap);
57
+ if (keys.length === 0 || keys.length === 1) {
58
+ return;
59
+ }
60
+
61
+ for (let x = 0; x < keys.length - 1; x++) {
62
+ const firstPage = pageIndexNumMap[keys[x]];
63
+ const secondPage = pageIndexNumMap[keys[x + 1]];
64
+ const prevCounter = counter;
65
+
66
+ for (let y = 0; y < firstPage.length && counter < 2; y++) {
67
+ for (let z = 0; z < secondPage.length && counter < 2; z++) {
68
+ const pageDifference = Number(keys[x + 1]) - Number(keys[x]);
69
+ if (firstPage[y] + 1 === secondPage[z]) {
70
+ counter++;
71
+ } else if (
72
+ pageDifference > 1 &&
73
+ firstPage[y] + pageDifference === secondPage[z]
74
+ ) {
75
+ counter++;
76
+ }
77
+ }
78
+ }
79
+
80
+ let pageDetails =
81
+ x > 0
82
+ ? Object.entries(pageIndexNumMap)[x - 1]
83
+ : Object.entries(pageIndexNumMap)[x];
84
+ if (prevCounter === counter) {
85
+ counter = 0;
86
+ pageDetails = Object.entries(pageIndexNumMap)[x];
87
+ } else if (counter >= 2) {
88
+ return { pageIndex: Number(pageDetails[0]), pageNum: pageDetails[1][0] };
89
+ }
90
+ }
91
+ };
92
+
93
+ const removePageNumber = (textContent: TextContent, pageNum: number): TextContent => {
94
+ const filteredContent = { items: [...textContent.items] };
95
+ const topArea = searchRange(1, 6, filteredContent.items.length);
96
+ const bottomArea = searchRange(5, 6, filteredContent.items.length);
97
+
98
+ filteredContent.items = filteredContent.items.filter((item, index) => {
99
+ const isAtTop = index > 0 && index < topArea;
100
+ const isAtBottom =
101
+ index > bottomArea && index < filteredContent.items.length;
102
+
103
+ return isAtTop || isAtBottom ? Number(item.str) !== Number(pageNum) : item;
104
+ });
105
+ return filteredContent;
106
+ };
107
+
108
+ export { findPageNumbers, findFirstPage, removePageNumber };
109
+
@@ -0,0 +1,124 @@
1
+ /**
2
+ * @description Character-level string utilities for the PDF text pipeline:
3
+ * digit and number detection, leading/trailing whitespace trimming, list-item
4
+ * character and pattern matching (bullet chars, numbered items), word-overlap
5
+ * scoring (`wordMatch`), char-code normalisation for headline matching
6
+ * (`normalizedCharCodeArray`), and camel-case detection
7
+ * (`hasUpperCaseCharacterInMiddleOfWord`).
8
+ */
9
+ const MIN_DIGIT_CHAR_CODE = 48
10
+ const MAX_DIGIT_CHAR_CODE = 57
11
+ const WHITESPACE_CHAR_CODE = 32
12
+ const TAB_CHAR_CODE = 9
13
+ const DOT_CHAR_CODE = 46
14
+
15
+ export function removeLeadingWhitespaces(string: string): string {
16
+ while (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
17
+ string = string.substring(1, string.length)
18
+ }
19
+ return string
20
+ }
21
+
22
+ export function removeTrailingWhitespaces(string: string): string {
23
+ while (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
24
+ string = string.substring(0, string.length - 1)
25
+ }
26
+ return string
27
+ }
28
+
29
+ export function isDigit(charCode: number): boolean {
30
+ return charCode >= MIN_DIGIT_CHAR_CODE && charCode <= MAX_DIGIT_CHAR_CODE
31
+ }
32
+
33
+ export function isNumber(string: string): boolean {
34
+ for (var i = 0; i < string.length; i++) {
35
+ const charCode = string.charCodeAt(i)
36
+ if (!isDigit(charCode)) {
37
+ return false
38
+ }
39
+ }
40
+ return true
41
+ }
42
+
43
+ export function hasOnly(string: string, char: string): boolean {
44
+ const charCode = char.charCodeAt(0)
45
+ for (var i = 0; i < string.length; i++) {
46
+ const aCharCode = string.charCodeAt(i)
47
+ if (aCharCode !== charCode) {
48
+ return false
49
+ }
50
+ }
51
+ return true
52
+ }
53
+
54
+ export function hasUpperCaseCharacterInMiddleOfWord(text: string): boolean {
55
+ var beginningOfWord = true
56
+ for (var i = 0; i < text.length; i++) {
57
+ const character = text.charAt(i)
58
+ if (character === ' ') {
59
+ beginningOfWord = true
60
+ } else {
61
+ if (!beginningOfWord && isNaN(character as any * 1) && character === character.toUpperCase() && character.toUpperCase() !== character.toLowerCase()) {
62
+ return true
63
+ }
64
+ beginningOfWord = false
65
+ }
66
+ }
67
+ return false
68
+ }
69
+
70
+ // Remove whitespace/dots + to uppercase
71
+ export function normalizedCharCodeArray(string: string): number[] {
72
+ string = string.toUpperCase()
73
+ return charCodeArray(string).filter(charCode => charCode !== WHITESPACE_CHAR_CODE && charCode !== TAB_CHAR_CODE && charCode !== DOT_CHAR_CODE)
74
+ }
75
+
76
+ export function charCodeArray(string: string): number[] {
77
+ const charCodes: number[] = []
78
+ for (var i = 0; i < string.length; i++) {
79
+ charCodes.push(string.charCodeAt(i))
80
+ }
81
+ return charCodes
82
+ }
83
+
84
+ export function prefixAfterWhitespace(prefix: string, string: string): string {
85
+ if (string.charCodeAt(0) === WHITESPACE_CHAR_CODE) {
86
+ string = removeLeadingWhitespaces(string)
87
+ return ' ' + prefix + string
88
+ } else {
89
+ return prefix + string
90
+ }
91
+ }
92
+
93
+ export function suffixBeforeWhitespace(string: string, suffix: string): string {
94
+ if (string.charCodeAt(string.length - 1) === WHITESPACE_CHAR_CODE) {
95
+ string = removeTrailingWhitespaces(string)
96
+ return string + suffix + ' '
97
+ } else {
98
+ return string + suffix
99
+ }
100
+ }
101
+
102
+ export function isListItemCharacter(string: string): boolean {
103
+ if (string.length > 1) {
104
+ return false
105
+ }
106
+ const char = string.charAt(0)
107
+ return char === '-' || char === '•' || char === '–'
108
+ }
109
+
110
+ export function isListItem(string: string): boolean {
111
+ return /^[\s]*[-•–][\s].*$/g.test(string)
112
+ }
113
+
114
+ export function isNumberedListItem(string: string): boolean {
115
+ return /^[\s]*[\d]*[.][\s].*$/g.test(string)
116
+ }
117
+
118
+ export function wordMatch(string1: string, string2: string): number {
119
+ const words1 = new Set(string1.toUpperCase().split(' '))
120
+ const words2 = new Set(string2.toUpperCase().split(' '))
121
+ const intersection = new Set(
122
+ [...words1].filter(x => words2.has(x)))
123
+ return intersection.size / Math.max(words1.size, words2.size)
124
+ }