extract-pdf 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/README.md +16 -10
  2. package/dist/models/annotation.d.ts +20 -0
  3. package/dist/models/block-type.d.ts +10 -0
  4. package/dist/models/headline-finder.d.ts +11 -0
  5. package/dist/models/line-converter.d.ts +10 -0
  6. package/dist/models/line-item-block.d.ts +14 -0
  7. package/dist/models/line-item.d.ts +25 -0
  8. package/dist/models/metadata.d.ts +23 -0
  9. package/dist/models/page-item.d.ts +21 -0
  10. package/dist/models/page.d.ts +14 -0
  11. package/dist/models/parse-result.d.ts +24 -0
  12. package/dist/models/parsed-elements.d.ts +20 -0
  13. package/dist/models/stashing-stream.d.ts +23 -0
  14. package/dist/models/text-item-line-grouper.d.ts +8 -0
  15. package/dist/models/text-item.d.ts +29 -0
  16. package/dist/models/word.d.ts +27 -0
  17. package/dist/pdf-to-html.cjs.js +1 -1
  18. package/dist/pdf-to-html.d.ts +36 -41
  19. package/dist/pdf-to-html.es.js +1 -1
  20. package/dist/transforms/base/to-line-item-block-transform.d.ts +6 -0
  21. package/dist/transforms/base/to-line-item-transform.d.ts +6 -0
  22. package/dist/transforms/base/to-text-item-transform.d.ts +6 -0
  23. package/dist/transforms/base/transformation.d.ts +8 -0
  24. package/dist/transforms/block/detect-code-quote-blocks.d.ts +6 -0
  25. package/dist/transforms/block/detect-list-levels.d.ts +6 -0
  26. package/dist/transforms/block/gather-blocks.d.ts +6 -0
  27. package/dist/transforms/calculate-global-stats.d.ts +11 -0
  28. package/dist/transforms/line-item/compact-lines.d.ts +6 -0
  29. package/dist/transforms/line-item/detect-headers.d.ts +6 -0
  30. package/dist/transforms/line-item/detect-list-items.d.ts +6 -0
  31. package/dist/transforms/line-item/detect-toc.d.ts +6 -0
  32. package/dist/transforms/line-item/remove-repetitive-elements.d.ts +6 -0
  33. package/dist/transforms/line-item/vertical-to-horizontal.d.ts +6 -0
  34. package/dist/transforms/to-html.d.ts +6 -0
  35. package/dist/transforms/to-text-blocks.d.ts +6 -0
  36. package/dist/utils/is-url-pdf.d.ts +1 -0
  37. package/dist/utils/page-item-functions.d.ts +8 -0
  38. package/dist/utils/page-number-functions.d.ts +14 -0
  39. package/dist/utils/string-functions.d.ts +14 -0
  40. package/package.json +8 -11
  41. package/src/models/annotation.ts +41 -0
  42. package/src/models/block-type.ts +203 -0
  43. package/src/models/headline-finder.ts +53 -0
  44. package/src/models/line-converter.ts +224 -0
  45. package/src/models/line-item-block.ts +51 -0
  46. package/src/models/line-item.ts +59 -0
  47. package/src/models/metadata.ts +29 -0
  48. package/src/models/page-item.ts +36 -0
  49. package/src/models/page.ts +16 -0
  50. package/src/models/parse-result.ts +32 -0
  51. package/src/models/parsed-elements.ts +29 -0
  52. package/src/models/stashing-stream.ts +86 -0
  53. package/src/models/text-item-line-grouper.ts +41 -0
  54. package/src/models/text-item.ts +50 -0
  55. package/src/models/word.ts +31 -0
  56. package/src/pdf-to-html.ts +225 -0
  57. package/src/transforms/base/to-line-item-block-transform.ts +29 -0
  58. package/src/transforms/base/to-line-item-transform.ts +29 -0
  59. package/src/transforms/base/to-text-item-transform.ts +28 -0
  60. package/src/transforms/base/transformation.ts +35 -0
  61. package/src/transforms/block/detect-code-quote-blocks.ts +57 -0
  62. package/src/transforms/block/detect-list-levels.ts +64 -0
  63. package/src/transforms/block/gather-blocks.ts +113 -0
  64. package/src/transforms/calculate-global-stats.ts +132 -0
  65. package/src/transforms/line-item/compact-lines.ts +92 -0
  66. package/src/transforms/line-item/detect-headers.ts +173 -0
  67. package/src/transforms/line-item/detect-list-items.ts +68 -0
  68. package/src/transforms/line-item/detect-toc.ts +459 -0
  69. package/src/transforms/line-item/remove-repetitive-elements.ts +101 -0
  70. package/src/transforms/line-item/vertical-to-horizontal.ts +90 -0
  71. package/src/transforms/to-html.ts +46 -0
  72. package/src/transforms/to-text-blocks.ts +38 -0
  73. package/src/utils/is-url-pdf.ts +33 -0
  74. package/src/utils/page-item-functions.ts +35 -0
  75. package/src/utils/page-number-functions.ts +109 -0
  76. package/src/utils/string-functions.ts +124 -0
@@ -0,0 +1,53 @@
1
+ /**
2
+ * @description Incrementally matches a target headline string across consecutive
3
+ * `LineItem`s. Characters are normalised (uppercase, whitespace/dot stripped)
4
+ * before comparison so multi-line or split titles are found correctly. Returns
5
+ * the array of matching items once the full headline has been consumed, or `null`
6
+ * if the sequence breaks. Used by `DetectTOC` to locate heading text on content pages.
7
+ */
8
+ import LineItem from "./line-item";
9
+ import { normalizedCharCodeArray } from "../utils/string-functions";
10
+
11
+ export default class HeadlineFinder {
12
+ headlineCharCodes: number[];
13
+ stackedLineItems: LineItem[];
14
+ stackedChars: number;
15
+
16
+ constructor(options: { headline: string }) {
17
+ this.headlineCharCodes = normalizedCharCodeArray(options.headline);
18
+ this.stackedLineItems = [];
19
+ this.stackedChars = 0;
20
+ }
21
+
22
+ consume(lineItem: LineItem): LineItem[] | null {
23
+ const normalizedCharCodes = normalizedCharCodeArray(lineItem.text());
24
+ const matchAll = this.matchAll(normalizedCharCodes);
25
+ if (matchAll) {
26
+ this.stackedLineItems.push(lineItem);
27
+ this.stackedChars += normalizedCharCodes.length;
28
+ if (this.stackedChars === this.headlineCharCodes.length) {
29
+ return this.stackedLineItems;
30
+ }
31
+ } else {
32
+ if (this.stackedChars > 0) {
33
+ this.stackedChars = 0;
34
+ this.stackedLineItems = [];
35
+ this.consume(lineItem);
36
+ }
37
+ }
38
+ return null;
39
+ }
40
+
41
+ matchAll(normalizedCharCodes: number[]): boolean {
42
+ for (var i = 0; i < normalizedCharCodes.length; i++) {
43
+ const headlineChar = this.headlineCharCodes[this.stackedChars + i];
44
+ const textItemChar = normalizedCharCodes[i];
45
+ if (textItemChar !== headlineChar) {
46
+ return false;
47
+ }
48
+ }
49
+ return true;
50
+ }
51
+ }
52
+
53
+
@@ -0,0 +1,224 @@
1
+ /**
2
+ * @description Converts an array of spatially-grouped TextItems (one line) into a single
3
+ * LineItem by detecting inline formatting (bold, italic), footnote superscripts, footnote
4
+ * anchors, and hyperlinks. WordFormat carries HTML open/close symbols; WordType carries
5
+ * rendering helpers for links and footnotes. The inner WordDetectionStream extends
6
+ * StashingStream to buffer consecutive same-format items before flushing them as Word nodes.
7
+ */
8
+ import TextItem from "./text-item";
9
+ import Word, { WordFormatEntry, WordTypeEntry } from "./word";
10
+ import LineItem from "./line-item";
11
+ import StashingStream from "./stashing-stream";
12
+ import ParsedElements from "./parsed-elements";
13
+ import { isNumber, isListItemCharacter } from "../utils/string-functions";
14
+ import { sortByX } from "../utils/page-item-functions";
15
+
16
+ export const WordFormat: Record<string, WordFormatEntry> = {
17
+ BOLD: {
18
+ name: "BOLD",
19
+ startSymbol: "<strong>",
20
+ endSymbol: "</strong>",
21
+ },
22
+
23
+ OBLIQUE: {
24
+ name: "OBLIQUE",
25
+ startSymbol: "<em>",
26
+ endSymbol: "</em>",
27
+ },
28
+
29
+ BOLD_OBLIQUE: {
30
+ name: "BOLD_OBLIQUE",
31
+ startSymbol: "<strong><em>",
32
+ endSymbol: "</em></strong>",
33
+ },
34
+ };
35
+
36
+ export const WordType: Record<string, WordTypeEntry> = {
37
+ LINK: {
38
+ name: "LINK",
39
+ toText(string: string) {
40
+ return `<a href="${string}">${string}</a>`;
41
+ },
42
+ },
43
+
44
+ FOOTNOTE_LINK: {
45
+ name: "FOOTNOTE_LINK",
46
+ attachWithoutWhitespace: true,
47
+ plainTextFormat: true,
48
+ toText(string: string) {
49
+ return `<sup><a href="#${string}">${string}</a></sup>`;
50
+ },
51
+ },
52
+
53
+ FOOTNOTE: {
54
+ name: "FOOTNOTE",
55
+ toText(string: string) {
56
+ return `<p id="${string}">^${string}</p>`;
57
+ },
58
+ },
59
+ };
60
+
61
+ export default class LineConverter {
62
+ fontToFormats: Map<string, string>;
63
+
64
+ constructor(fontToFormats: Map<string, string>) {
65
+ this.fontToFormats = fontToFormats;
66
+ }
67
+
68
+ compact(textItems: TextItem[]): LineItem {
69
+ sortByX(textItems);
70
+
71
+ const wordStream = new WordDetectionStream(this.fontToFormats);
72
+ wordStream.consumeAll(textItems.map((item) => new TextItem({ ...item })));
73
+ const words = wordStream.complete();
74
+
75
+ var maxHeight = 0;
76
+ var widthSum = 0;
77
+ textItems.forEach((item) => {
78
+ maxHeight = Math.max(maxHeight, item.height);
79
+ widthSum += item.width;
80
+ });
81
+ return new LineItem({
82
+ x: textItems[0].x,
83
+ y: textItems[0].y,
84
+ height: maxHeight,
85
+ width: widthSum,
86
+ words: words,
87
+ parsedElements: new ParsedElements({
88
+ footnoteLinks: wordStream.footnoteLinks,
89
+ footnotes: wordStream.footnotes,
90
+ containLinks: wordStream.containLinks,
91
+ formattedWords: wordStream.formattedWords,
92
+ }),
93
+ });
94
+ }
95
+ }
96
+
97
+ class WordDetectionStream extends StashingStream {
98
+ fontToFormats: Map<string, string>;
99
+ footnoteLinks: number[];
100
+ footnotes: string[];
101
+ formattedWords: number;
102
+ containLinks: boolean;
103
+ stashedNumber: boolean;
104
+ firstY?: number;
105
+ currentItem: TextItem | null;
106
+
107
+ constructor(fontToFormats: Map<string, string>) {
108
+ super();
109
+ this.fontToFormats = fontToFormats;
110
+ this.footnoteLinks = [];
111
+ this.footnotes = [];
112
+ this.formattedWords = 0;
113
+ this.containLinks = false;
114
+ this.stashedNumber = false;
115
+ this.currentItem = null;
116
+ }
117
+
118
+ shouldStash(item: any): boolean {
119
+ if (!this.firstY) {
120
+ this.firstY = item.y;
121
+ }
122
+ this.currentItem = item;
123
+ return true;
124
+ }
125
+
126
+ onPushOnStash(item: any): void {
127
+ this.stashedNumber = isNumber(item.text.trim());
128
+ }
129
+
130
+ doMatchesStash(lastItem: any, item: any): boolean {
131
+ const lastItemFormat = this.fontToFormats.get(lastItem.font);
132
+ const itemFormat = this.fontToFormats.get(item.font);
133
+ if (lastItemFormat !== itemFormat) {
134
+ return false;
135
+ }
136
+ const itemIsANumber = isNumber(item.text.trim());
137
+ return this.stashedNumber === itemIsANumber;
138
+ }
139
+
140
+ doFlushStash(stash: any[], results: any[]): void {
141
+ if (this.stashedNumber) {
142
+ const joinedNumber = stash
143
+ .map((item: any) => item.text)
144
+ .join("")
145
+ .trim();
146
+ if (stash[0].y > this.firstY) {
147
+ results.push(
148
+ new Word({
149
+ string: `${joinedNumber}`,
150
+ type: WordType.FOOTNOTE_LINK,
151
+ }),
152
+ );
153
+ this.footnoteLinks.push(parseInt(joinedNumber));
154
+ } else if (this.currentItem && this.currentItem.y < stash[0].y) {
155
+ results.push(
156
+ new Word({
157
+ string: `${joinedNumber}`,
158
+ type: WordType.FOOTNOTE,
159
+ }),
160
+ );
161
+ this.footnotes.push(joinedNumber);
162
+ } else {
163
+ this.copyStashItemsAsText(stash, results);
164
+ }
165
+ } else {
166
+ this.copyStashItemsAsText(stash, results);
167
+ }
168
+ }
169
+
170
+ copyStashItemsAsText(stash: any[], results: any[]): void {
171
+ const format = this.fontToFormats.get(stash[0].font);
172
+ results.push(...this.itemsToWords(stash, format));
173
+ }
174
+
175
+ itemsToWords(items: TextItem[], formatName: string | null): Word[] {
176
+ const combinedText = combineText(items);
177
+ const words = combinedText.split(" ");
178
+ const format = formatName ? WordFormat[formatName] : null;
179
+ return words
180
+ .filter((w) => w.trim().length > 0)
181
+ .map((word) => {
182
+ var type: WordTypeEntry | null = null;
183
+ if (word.startsWith("http:")) {
184
+ this.containLinks = true;
185
+ type = WordType.LINK;
186
+ } else if (word.startsWith("www.")) {
187
+ this.containLinks = true;
188
+ word = `http://${word}`;
189
+ type = WordType.LINK;
190
+ }
191
+
192
+ if (format) {
193
+ this.formattedWords++;
194
+ }
195
+ return new Word({ string: word, type, format });
196
+ });
197
+ }
198
+ }
199
+
200
+ function combineText(textItems: TextItem[]): string {
201
+ var text = "";
202
+ var lastItem: TextItem | null = null;
203
+ textItems.forEach((textItem) => {
204
+ var textToAdd = textItem.text;
205
+ if (!text.endsWith(" ") && !textToAdd.startsWith(" ")) {
206
+ if (lastItem) {
207
+ const xDistance = textItem.x - lastItem.x - lastItem.width;
208
+ if (xDistance > 5) {
209
+ text += " ";
210
+ }
211
+ } else {
212
+ if (isListItemCharacter(textItem.text)) {
213
+ textToAdd += " ";
214
+ }
215
+ }
216
+ }
217
+ text += textToAdd;
218
+ lastItem = textItem;
219
+ });
220
+ return text;
221
+ }
222
+
223
+
224
+
@@ -0,0 +1,51 @@
1
+ /**
2
+ * @description A semantically coherent group of `LineItem`s that share the same
3
+ * block type (paragraph, list, heading, etc.). Extends `PageItem`. `addItem`
4
+ * enforces type consistency across all lines, merges `ParsedElements` metadata
5
+ * from each line into the block, and strips the type tag from individual lines
6
+ * so only the block-level type is authoritative.
7
+ */
8
+ import PageItem, { BlockTypeEntry } from "./page-item";
9
+ import Annotation from "./annotation";
10
+ import ParsedElements from "./parsed-elements";
11
+ import LineItem from "./line-item";
12
+
13
+ // A block of LineItem[] within a Page
14
+ export default class LineItemBlock extends PageItem {
15
+ items: LineItem[];
16
+
17
+ constructor(options: {
18
+ items?: LineItem[];
19
+ type?: BlockTypeEntry | null;
20
+ annotation?: Annotation | null;
21
+ parsedElements?: ParsedElements | null;
22
+ }) {
23
+ super(options);
24
+ this.items = [];
25
+ if (options.items) {
26
+ options.items.forEach((item) => this.addItem(item));
27
+ }
28
+ }
29
+
30
+ addItem(item: LineItem): void {
31
+ if (this.type && item.type && this.type !== item.type) {
32
+ throw new Error(
33
+ `Adding item of type ${item.type.name} to block of type ${this.type.name}`,
34
+ );
35
+ }
36
+ if (!this.type) {
37
+ this.type = item.type;
38
+ }
39
+ if (item.parsedElements) {
40
+ if (this.parsedElements) {
41
+ this.parsedElements.add(item.parsedElements);
42
+ } else {
43
+ this.parsedElements = item.parsedElements;
44
+ }
45
+ }
46
+ const copiedItem = new LineItem({ ...item });
47
+ copiedItem.type = null;
48
+ this.items.push(copiedItem);
49
+ }
50
+ }
51
+
@@ -0,0 +1,59 @@
1
+ /**
2
+ * @description A single typographic line on a PDF page. Extends `PageItem` with
3
+ * spatial coordinates (x, y, width, height) and a `words` array of `Word` tokens.
4
+ * Can be constructed from a pre-built `words` array or from a raw `text` string,
5
+ * which is split on spaces and wrapped into `Word` instances automatically.
6
+ */
7
+ import PageItem, { BlockTypeEntry } from "./page-item";
8
+ import Annotation from "./annotation";
9
+ import ParsedElements from "./parsed-elements";
10
+ import Word from "./word";
11
+
12
+ // A line within a page
13
+ export default class LineItem extends PageItem {
14
+ x: number;
15
+ y: number;
16
+ width: number;
17
+ height: number;
18
+ words: Word[];
19
+
20
+ constructor(options: {
21
+ x?: number;
22
+ y?: number;
23
+ width?: number;
24
+ height?: number;
25
+ words?: Word[];
26
+ text?: string;
27
+ type?: BlockTypeEntry | null;
28
+ annotation?: Annotation | null;
29
+ parsedElements?: ParsedElements | null;
30
+ font?: string;
31
+ }) {
32
+ super(options);
33
+ this.x = options.x ?? 0;
34
+ this.y = options.y ?? 0;
35
+ this.width = options.width ?? 0;
36
+ this.height = options.height ?? 0;
37
+ this.words = options.words || [];
38
+ if (options.text && !options.words) {
39
+ this.words = options.text
40
+ .split(" ")
41
+ .filter((string) => string.trim().length > 0)
42
+ .map(
43
+ (wordAsString) =>
44
+ new Word({
45
+ string: wordAsString,
46
+ }),
47
+ );
48
+ }
49
+ }
50
+
51
+ text(): string {
52
+ return this.wordStrings().join(" ");
53
+ }
54
+
55
+ wordStrings(): string[] {
56
+ return this.words.map((word) => word.string);
57
+ }
58
+ }
59
+
@@ -0,0 +1,29 @@
1
+ /**
2
+ * @description Normalised PDF document metadata (title, author, creator, producer).
3
+ * Handles both the modern XMP `metadata` API and the legacy `info` dictionary
4
+ * returned by pdfjs, so callers always receive a consistent shape regardless of
5
+ * which metadata format the PDF uses.
6
+ */
7
+ // Metadata of the PDF document
8
+ export default class Metadata {
9
+ title?: string;
10
+ creator?: string;
11
+ producer?: string;
12
+ author?: string;
13
+
14
+ constructor(originalMetadata: {
15
+ metadata?: { get(key: string): string | undefined };
16
+ info?: { Title?: string; Author?: string; Creator?: string; Producer?: string };
17
+ }) {
18
+ if (originalMetadata.metadata) {
19
+ this.title = originalMetadata.metadata.get("dc:title");
20
+ this.creator = originalMetadata.metadata.get("xap:creatortool");
21
+ this.producer = originalMetadata.metadata.get("pdf:producer");
22
+ } else {
23
+ this.title = originalMetadata.info?.Title;
24
+ this.author = originalMetadata.info?.Author;
25
+ this.creator = originalMetadata.info?.Creator;
26
+ this.producer = originalMetadata.info?.Producer;
27
+ }
28
+ }
29
+ }
@@ -0,0 +1,36 @@
1
+ /**
2
+ * @description Abstract base class for all page-level document nodes (`TextItem`,
3
+ * `LineItem`, `LineItemBlock`). Carries the shared fields `type` (semantic block
4
+ * category from `BlockType`), `annotation` (diff marker for debug rendering), and
5
+ * `parsedElements` (inline footnote/link metadata). Throws `TypeError` if
6
+ * instantiated directly.
7
+ */
8
+ import Annotation from "./annotation";
9
+ import ParsedElements from "./parsed-elements";
10
+
11
+ export interface BlockTypeEntry {
12
+ name: string;
13
+ headline?: boolean;
14
+ headlineLevel?: number;
15
+ mergeToBlock?: boolean;
16
+ mergeFollowingNonTypedItems?: boolean;
17
+ mergeFollowingNonTypedItemsWithSmallDistance?: boolean;
18
+ toText?(block: any): string;
19
+ }
20
+
21
+ // A abstract PageItem class, can be TextItem, LineItem or LineItemBlock
22
+ export default class PageItem {
23
+ type: BlockTypeEntry | null;
24
+ annotation: Annotation | null;
25
+ parsedElements: ParsedElements | null;
26
+
27
+ constructor(options: { type?: BlockTypeEntry | null; annotation?: Annotation | null; parsedElements?: ParsedElements | null }) {
28
+ if (this.constructor === PageItem) {
29
+ throw new TypeError("Can not construct abstract class.");
30
+ }
31
+ this.type = options.type ?? null;
32
+ this.annotation = options.annotation ?? null;
33
+ this.parsedElements = options.parsedElements ?? null;
34
+ }
35
+ }
36
+
@@ -0,0 +1,16 @@
1
+ /**
2
+ * @description Simple container for a single PDF page. Holds a 0-based `index`
3
+ * and an `items` array whose element type evolves through the transformation
4
+ * pipeline: raw `TextItem[]` → compacted `LineItem[]` → grouped `LineItemBlock[]`
5
+ * → final serialized text objects.
6
+ */
7
+ // A page which holds PageItems displayable via PdfPageView
8
+ export default class Page {
9
+ index: number;
10
+ items: any[];
11
+
12
+ constructor(options: { index: number; items?: any[] }) {
13
+ this.index = options.index;
14
+ this.items = options.items || [];
15
+ }
16
+ }
@@ -0,0 +1,32 @@
1
+ /**
2
+ * @description Immutable value object passed between every pipeline stage.
3
+ * `pages` holds the document's page array (items mutate in type each stage),
4
+ * `globals` holds cross-page statistics computed by `CalculateGlobalStats`
5
+ * (most-used height, font, line distance, font-format map), and `messages`
6
+ * carries diagnostic strings emitted by each transformation for debugging.
7
+ */
8
+ import Page from "./page";
9
+
10
+ export interface ParseGlobals {
11
+ mostUsedHeight?: number;
12
+ mostUsedFont?: string;
13
+ mostUsedDistance?: number;
14
+ maxHeight?: number;
15
+ maxHeightFont?: string;
16
+ fontToFormats?: Map<string, string>;
17
+ tocPages?: number[];
18
+ headlineTypeToHeightRange?: Record<string, { min: number; max: number }>;
19
+ }
20
+
21
+ // The result of a PDF parse respectively a Transformation
22
+ export default class ParseResult {
23
+ pages: Page[];
24
+ globals: ParseGlobals;
25
+ messages: string[];
26
+
27
+ constructor(options: { pages?: Page[]; globals?: ParseGlobals; messages?: string[] }) {
28
+ this.pages = options.pages || [];
29
+ this.globals = options.globals || {};
30
+ this.messages = options.messages || [];
31
+ }
32
+ }
@@ -0,0 +1,29 @@
1
+ /**
2
+ * @description Aggregated inline-element metadata for a `LineItem` or
3
+ * `LineItemBlock`. Tracks detected footnote link numbers, footnote definitions,
4
+ * whether any hyperlinks are present, and a count of bold/italic formatted words.
5
+ * The `add()` method merges a child item's `ParsedElements` into the parent
6
+ * block's running totals as lines are gathered into blocks.
7
+ */
8
+ export default class ParsedElements {
9
+ footnoteLinks: number[];
10
+ footnotes: string[];
11
+ containLinks: boolean;
12
+ formattedWords: number;
13
+
14
+ constructor(options: { footnoteLinks?: number[]; footnotes?: string[]; containLinks?: boolean; formattedWords?: number }) {
15
+ this.footnoteLinks = options.footnoteLinks || [];
16
+ this.footnotes = options.footnotes || [];
17
+ this.containLinks = options.containLinks ?? false;
18
+ this.formattedWords = options.formattedWords ?? 0;
19
+ }
20
+
21
+ add(parsedElements: ParsedElements): void {
22
+ this.footnoteLinks = this.footnoteLinks.concat(
23
+ parsedElements.footnoteLinks,
24
+ );
25
+ this.footnotes = this.footnotes.concat(parsedElements.footnotes);
26
+ this.containLinks = this.containLinks || parsedElements.containLinks;
27
+ this.formattedWords += parsedElements.formattedWords;
28
+ }
29
+ }
@@ -0,0 +1,86 @@
1
+ /**
2
+ * @description Abstract stream processor that buffers ("stashes") consecutive
3
+ * items matching a predicate before flushing them as a group. Subclasses
4
+ * implement `shouldStash`, `doMatchesStash`, and `doFlushStash` to define what
5
+ * constitutes a run. Used by `WordDetectionStream` in `LineConverter` (grouping
6
+ * same-format text spans) and `VerticalsStream` in `VerticalToHorizontal`
7
+ * (merging single-character vertical lines into one horizontal line).
8
+ */
9
+ // Abstract stream which allows stash items temporarily
10
+ export default class StashingStream {
11
+ results: any[];
12
+ stash: any[];
13
+
14
+ constructor() {
15
+ if (this.constructor === StashingStream) {
16
+ throw new TypeError("Can not construct abstract class.");
17
+ }
18
+ this.results = [];
19
+ this.stash = [];
20
+ }
21
+
22
+ consumeAll(items: any[]): void {
23
+ items.forEach((item) => this.consume(item));
24
+ }
25
+
26
+ consume(item: any): void {
27
+ if (this.shouldStash(item)) {
28
+ if (!this.matchesStash(item)) {
29
+ this.flushStash();
30
+ }
31
+ this.pushOnStash(item);
32
+ } else {
33
+ if (this.stash.length > 0) {
34
+ this.flushStash();
35
+ }
36
+ this.results.push(item);
37
+ }
38
+ }
39
+
40
+ pushOnStash(item: any): void {
41
+ this.onPushOnStash(item);
42
+ this.stash.push(item);
43
+ }
44
+
45
+ complete(): any[] {
46
+ if (this.stash.length > 0) {
47
+ this.flushStash();
48
+ }
49
+ return this.results;
50
+ }
51
+
52
+ matchesStash(item: any): boolean {
53
+ if (this.stash.length === 0) {
54
+ return true;
55
+ }
56
+ const lastItem = this.stash[this.stash.length - 1];
57
+ return this.doMatchesStash(lastItem, item);
58
+ }
59
+
60
+ flushStash(): void {
61
+ if (this.stash.length > 0) {
62
+ this.doFlushStash(this.stash, this.results);
63
+ this.stash = [];
64
+ }
65
+ }
66
+
67
+ onPushOnStash(item: any): void {
68
+ // sub-classes may override
69
+ }
70
+
71
+ shouldStash(item: any): boolean {
72
+ throw new TypeError(" Do not call abstract method foo from child." + item);
73
+ }
74
+
75
+ doMatchesStash(lastItem: any, item: any): boolean {
76
+ throw new TypeError(
77
+ " Do not call abstract method foo from child." + lastItem + item,
78
+ );
79
+ }
80
+
81
+ doFlushStash(stash: any[], results: any[]): void {
82
+ throw new TypeError(
83
+ " Do not call abstract method foo from child." + stash + results,
84
+ );
85
+ }
86
+ }
@@ -0,0 +1,41 @@
1
+ /**
2
+ * @description Groups a flat array of `TextItem`s into per-line buckets by
3
+ * clustering items whose y-coordinates differ by less than half of
4
+ * `mostUsedDistance`. Each bucket is then sorted by x so items read left-to-right,
5
+ * which is critical for footnote detection and word ordering in `CompactLines`.
6
+ */
7
+
8
+ import TextItem from "./text-item";
9
+ import { sortByX } from "../utils/page-item-functions";
10
+
11
+ // Groups all text items which are on the same y line
12
+ export default class TextItemLineGrouper {
13
+ mostUsedDistance: number;
14
+
15
+ constructor(options: { mostUsedDistance?: number }) {
16
+ this.mostUsedDistance = options.mostUsedDistance || 12;
17
+ }
18
+
19
+ group(textItems: TextItem[]): TextItem[][] {
20
+ const lines: TextItem[][] = [];
21
+ var currentLine: TextItem[] = [];
22
+ textItems.forEach((item) => {
23
+ if (
24
+ currentLine.length > 0 &&
25
+ Math.abs(currentLine[0].y - item.y) >= this.mostUsedDistance / 2
26
+ ) {
27
+ lines.push(currentLine);
28
+ currentLine = [];
29
+ }
30
+ currentLine.push(item);
31
+ });
32
+ lines.push(currentLine);
33
+
34
+ lines.forEach((lineItems) => {
35
+ sortByX(lineItems);
36
+ });
37
+ return lines;
38
+ }
39
+ }
40
+
41
+