extract-pdf 0.1.20 → 0.1.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/README.md +74 -74
  2. package/package.json +1 -1
  3. package/src/models/annotation.ts +41 -41
  4. package/src/models/block-type.ts +203 -203
  5. package/src/models/line-converter.ts +224 -224
  6. package/src/models/metadata.ts +29 -29
  7. package/src/models/page.ts +16 -16
  8. package/src/models/parse-result.ts +32 -32
  9. package/src/models/parsed-elements.ts +29 -29
  10. package/src/models/stashing-stream.ts +86 -86
  11. package/src/models/text-item-line-grouper.ts +41 -41
  12. package/src/models/word.ts +31 -31
  13. package/src/pdf-to-html.ts +225 -225
  14. package/src/transforms/base/to-line-item-block-transform.ts +29 -29
  15. package/src/transforms/base/to-line-item-transform.ts +29 -29
  16. package/src/transforms/base/to-text-item-transform.ts +28 -28
  17. package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
  18. package/src/transforms/block/detect-list-levels.ts +64 -64
  19. package/src/transforms/block/gather-blocks.ts +113 -113
  20. package/src/transforms/calculate-global-stats.ts +132 -132
  21. package/src/transforms/line-item/compact-lines.ts +92 -92
  22. package/src/transforms/line-item/detect-headers.ts +173 -173
  23. package/src/transforms/line-item/detect-list-items.ts +68 -68
  24. package/src/transforms/line-item/detect-toc.ts +459 -459
  25. package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
  26. package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
  27. package/src/transforms/to-html.ts +46 -46
  28. package/src/utils/is-url-pdf.ts +33 -33
  29. package/src/utils/page-item-functions.ts +35 -35
  30. package/src/utils/string-functions.ts +124 -124
@@ -1,224 +1,224 @@
1
- /**
2
- * @description Converts an array of spatially-grouped TextItems (one line) into a single
3
- * LineItem by detecting inline formatting (bold, italic), footnote superscripts, footnote
4
- * anchors, and hyperlinks. WordFormat carries HTML open/close symbols; WordType carries
5
- * rendering helpers for links and footnotes. The inner WordDetectionStream extends
6
- * StashingStream to buffer consecutive same-format items before flushing them as Word nodes.
7
- */
8
- import TextItem from "./text-item";
9
- import Word, { WordFormatEntry, WordTypeEntry } from "./word";
10
- import LineItem from "./line-item";
11
- import StashingStream from "./stashing-stream";
12
- import ParsedElements from "./parsed-elements";
13
- import { isNumber, isListItemCharacter } from "../utils/string-functions";
14
- import { sortByX } from "../utils/page-item-functions";
15
-
16
- export const WordFormat: Record<string, WordFormatEntry> = {
17
- BOLD: {
18
- name: "BOLD",
19
- startSymbol: "<strong>",
20
- endSymbol: "</strong>",
21
- },
22
-
23
- OBLIQUE: {
24
- name: "OBLIQUE",
25
- startSymbol: "<em>",
26
- endSymbol: "</em>",
27
- },
28
-
29
- BOLD_OBLIQUE: {
30
- name: "BOLD_OBLIQUE",
31
- startSymbol: "<strong><em>",
32
- endSymbol: "</em></strong>",
33
- },
34
- };
35
-
36
- export const WordType: Record<string, WordTypeEntry> = {
37
- LINK: {
38
- name: "LINK",
39
- toText(string: string) {
40
- return `<a href="${string}">${string}</a>`;
41
- },
42
- },
43
-
44
- FOOTNOTE_LINK: {
45
- name: "FOOTNOTE_LINK",
46
- attachWithoutWhitespace: true,
47
- plainTextFormat: true,
48
- toText(string: string) {
49
- return `<sup><a href="#${string}">${string}</a></sup>`;
50
- },
51
- },
52
-
53
- FOOTNOTE: {
54
- name: "FOOTNOTE",
55
- toText(string: string) {
56
- return `<p id="${string}">^${string}</p>`;
57
- },
58
- },
59
- };
60
-
61
- export default class LineConverter {
62
- fontToFormats: Map<string, string>;
63
-
64
- constructor(fontToFormats: Map<string, string>) {
65
- this.fontToFormats = fontToFormats;
66
- }
67
-
68
- compact(textItems: TextItem[]): LineItem {
69
- sortByX(textItems);
70
-
71
- const wordStream = new WordDetectionStream(this.fontToFormats);
72
- wordStream.consumeAll(textItems.map((item) => new TextItem({ ...item })));
73
- const words = wordStream.complete();
74
-
75
- var maxHeight = 0;
76
- var widthSum = 0;
77
- textItems.forEach((item) => {
78
- maxHeight = Math.max(maxHeight, item.height);
79
- widthSum += item.width;
80
- });
81
- return new LineItem({
82
- x: textItems[0].x,
83
- y: textItems[0].y,
84
- height: maxHeight,
85
- width: widthSum,
86
- words: words,
87
- parsedElements: new ParsedElements({
88
- footnoteLinks: wordStream.footnoteLinks,
89
- footnotes: wordStream.footnotes,
90
- containLinks: wordStream.containLinks,
91
- formattedWords: wordStream.formattedWords,
92
- }),
93
- });
94
- }
95
- }
96
-
97
- class WordDetectionStream extends StashingStream {
98
- fontToFormats: Map<string, string>;
99
- footnoteLinks: number[];
100
- footnotes: string[];
101
- formattedWords: number;
102
- containLinks: boolean;
103
- stashedNumber: boolean;
104
- firstY?: number;
105
- currentItem: TextItem | null;
106
-
107
- constructor(fontToFormats: Map<string, string>) {
108
- super();
109
- this.fontToFormats = fontToFormats;
110
- this.footnoteLinks = [];
111
- this.footnotes = [];
112
- this.formattedWords = 0;
113
- this.containLinks = false;
114
- this.stashedNumber = false;
115
- this.currentItem = null;
116
- }
117
-
118
- shouldStash(item: any): boolean {
119
- if (!this.firstY) {
120
- this.firstY = item.y;
121
- }
122
- this.currentItem = item;
123
- return true;
124
- }
125
-
126
- onPushOnStash(item: any): void {
127
- this.stashedNumber = isNumber(item.text.trim());
128
- }
129
-
130
- doMatchesStash(lastItem: any, item: any): boolean {
131
- const lastItemFormat = this.fontToFormats.get(lastItem.font);
132
- const itemFormat = this.fontToFormats.get(item.font);
133
- if (lastItemFormat !== itemFormat) {
134
- return false;
135
- }
136
- const itemIsANumber = isNumber(item.text.trim());
137
- return this.stashedNumber === itemIsANumber;
138
- }
139
-
140
- doFlushStash(stash: any[], results: any[]): void {
141
- if (this.stashedNumber) {
142
- const joinedNumber = stash
143
- .map((item: any) => item.text)
144
- .join("")
145
- .trim();
146
- if (stash[0].y > this.firstY) {
147
- results.push(
148
- new Word({
149
- string: `${joinedNumber}`,
150
- type: WordType.FOOTNOTE_LINK,
151
- }),
152
- );
153
- this.footnoteLinks.push(parseInt(joinedNumber));
154
- } else if (this.currentItem && this.currentItem.y < stash[0].y) {
155
- results.push(
156
- new Word({
157
- string: `${joinedNumber}`,
158
- type: WordType.FOOTNOTE,
159
- }),
160
- );
161
- this.footnotes.push(joinedNumber);
162
- } else {
163
- this.copyStashItemsAsText(stash, results);
164
- }
165
- } else {
166
- this.copyStashItemsAsText(stash, results);
167
- }
168
- }
169
-
170
- copyStashItemsAsText(stash: any[], results: any[]): void {
171
- const format = this.fontToFormats.get(stash[0].font);
172
- results.push(...this.itemsToWords(stash, format));
173
- }
174
-
175
- itemsToWords(items: TextItem[], formatName: string | null): Word[] {
176
- const combinedText = combineText(items);
177
- const words = combinedText.split(" ");
178
- const format = formatName ? WordFormat[formatName] : null;
179
- return words
180
- .filter((w) => w.trim().length > 0)
181
- .map((word) => {
182
- var type: WordTypeEntry | null = null;
183
- if (word.startsWith("http:")) {
184
- this.containLinks = true;
185
- type = WordType.LINK;
186
- } else if (word.startsWith("www.")) {
187
- this.containLinks = true;
188
- word = `http://${word}`;
189
- type = WordType.LINK;
190
- }
191
-
192
- if (format) {
193
- this.formattedWords++;
194
- }
195
- return new Word({ string: word, type, format });
196
- });
197
- }
198
- }
199
-
200
- function combineText(textItems: TextItem[]): string {
201
- var text = "";
202
- var lastItem: TextItem | null = null;
203
- textItems.forEach((textItem) => {
204
- var textToAdd = textItem.text;
205
- if (!text.endsWith(" ") && !textToAdd.startsWith(" ")) {
206
- if (lastItem) {
207
- const xDistance = textItem.x - lastItem.x - lastItem.width;
208
- if (xDistance > 5) {
209
- text += " ";
210
- }
211
- } else {
212
- if (isListItemCharacter(textItem.text)) {
213
- textToAdd += " ";
214
- }
215
- }
216
- }
217
- text += textToAdd;
218
- lastItem = textItem;
219
- });
220
- return text;
221
- }
222
-
223
-
224
-
1
+ /**
2
+ * @description Converts an array of spatially-grouped TextItems (one line) into a single
3
+ * LineItem by detecting inline formatting (bold, italic), footnote superscripts, footnote
4
+ * anchors, and hyperlinks. WordFormat carries HTML open/close symbols; WordType carries
5
+ * rendering helpers for links and footnotes. The inner WordDetectionStream extends
6
+ * StashingStream to buffer consecutive same-format items before flushing them as Word nodes.
7
+ */
8
+ import TextItem from "./text-item";
9
+ import Word, { WordFormatEntry, WordTypeEntry } from "./word";
10
+ import LineItem from "./line-item";
11
+ import StashingStream from "./stashing-stream";
12
+ import ParsedElements from "./parsed-elements";
13
+ import { isNumber, isListItemCharacter } from "../utils/string-functions";
14
+ import { sortByX } from "../utils/page-item-functions";
15
+
16
+ export const WordFormat: Record<string, WordFormatEntry> = {
17
+ BOLD: {
18
+ name: "BOLD",
19
+ startSymbol: "<strong>",
20
+ endSymbol: "</strong>",
21
+ },
22
+
23
+ OBLIQUE: {
24
+ name: "OBLIQUE",
25
+ startSymbol: "<em>",
26
+ endSymbol: "</em>",
27
+ },
28
+
29
+ BOLD_OBLIQUE: {
30
+ name: "BOLD_OBLIQUE",
31
+ startSymbol: "<strong><em>",
32
+ endSymbol: "</em></strong>",
33
+ },
34
+ };
35
+
36
+ export const WordType: Record<string, WordTypeEntry> = {
37
+ LINK: {
38
+ name: "LINK",
39
+ toText(string: string) {
40
+ return `<a href="${string}">${string}</a>`;
41
+ },
42
+ },
43
+
44
+ FOOTNOTE_LINK: {
45
+ name: "FOOTNOTE_LINK",
46
+ attachWithoutWhitespace: true,
47
+ plainTextFormat: true,
48
+ toText(string: string) {
49
+ return `<sup><a href="#${string}">${string}</a></sup>`;
50
+ },
51
+ },
52
+
53
+ FOOTNOTE: {
54
+ name: "FOOTNOTE",
55
+ toText(string: string) {
56
+ return `<p id="${string}">^${string}</p>`;
57
+ },
58
+ },
59
+ };
60
+
61
+ export default class LineConverter {
62
+ fontToFormats: Map<string, string>;
63
+
64
+ constructor(fontToFormats: Map<string, string>) {
65
+ this.fontToFormats = fontToFormats;
66
+ }
67
+
68
+ compact(textItems: TextItem[]): LineItem {
69
+ sortByX(textItems);
70
+
71
+ const wordStream = new WordDetectionStream(this.fontToFormats);
72
+ wordStream.consumeAll(textItems.map((item) => new TextItem({ ...item })));
73
+ const words = wordStream.complete();
74
+
75
+ var maxHeight = 0;
76
+ var widthSum = 0;
77
+ textItems.forEach((item) => {
78
+ maxHeight = Math.max(maxHeight, item.height);
79
+ widthSum += item.width;
80
+ });
81
+ return new LineItem({
82
+ x: textItems[0].x,
83
+ y: textItems[0].y,
84
+ height: maxHeight,
85
+ width: widthSum,
86
+ words: words,
87
+ parsedElements: new ParsedElements({
88
+ footnoteLinks: wordStream.footnoteLinks,
89
+ footnotes: wordStream.footnotes,
90
+ containLinks: wordStream.containLinks,
91
+ formattedWords: wordStream.formattedWords,
92
+ }),
93
+ });
94
+ }
95
+ }
96
+
97
+ class WordDetectionStream extends StashingStream {
98
+ fontToFormats: Map<string, string>;
99
+ footnoteLinks: number[];
100
+ footnotes: string[];
101
+ formattedWords: number;
102
+ containLinks: boolean;
103
+ stashedNumber: boolean;
104
+ firstY?: number;
105
+ currentItem: TextItem | null;
106
+
107
+ constructor(fontToFormats: Map<string, string>) {
108
+ super();
109
+ this.fontToFormats = fontToFormats;
110
+ this.footnoteLinks = [];
111
+ this.footnotes = [];
112
+ this.formattedWords = 0;
113
+ this.containLinks = false;
114
+ this.stashedNumber = false;
115
+ this.currentItem = null;
116
+ }
117
+
118
+ shouldStash(item: any): boolean {
119
+ if (!this.firstY) {
120
+ this.firstY = item.y;
121
+ }
122
+ this.currentItem = item;
123
+ return true;
124
+ }
125
+
126
+ onPushOnStash(item: any): void {
127
+ this.stashedNumber = isNumber(item.text.trim());
128
+ }
129
+
130
+ doMatchesStash(lastItem: any, item: any): boolean {
131
+ const lastItemFormat = this.fontToFormats.get(lastItem.font);
132
+ const itemFormat = this.fontToFormats.get(item.font);
133
+ if (lastItemFormat !== itemFormat) {
134
+ return false;
135
+ }
136
+ const itemIsANumber = isNumber(item.text.trim());
137
+ return this.stashedNumber === itemIsANumber;
138
+ }
139
+
140
+ doFlushStash(stash: any[], results: any[]): void {
141
+ if (this.stashedNumber) {
142
+ const joinedNumber = stash
143
+ .map((item: any) => item.text)
144
+ .join("")
145
+ .trim();
146
+ if (stash[0].y > this.firstY) {
147
+ results.push(
148
+ new Word({
149
+ string: `${joinedNumber}`,
150
+ type: WordType.FOOTNOTE_LINK,
151
+ }),
152
+ );
153
+ this.footnoteLinks.push(parseInt(joinedNumber));
154
+ } else if (this.currentItem && this.currentItem.y < stash[0].y) {
155
+ results.push(
156
+ new Word({
157
+ string: `${joinedNumber}`,
158
+ type: WordType.FOOTNOTE,
159
+ }),
160
+ );
161
+ this.footnotes.push(joinedNumber);
162
+ } else {
163
+ this.copyStashItemsAsText(stash, results);
164
+ }
165
+ } else {
166
+ this.copyStashItemsAsText(stash, results);
167
+ }
168
+ }
169
+
170
+ copyStashItemsAsText(stash: any[], results: any[]): void {
171
+ const format = this.fontToFormats.get(stash[0].font);
172
+ results.push(...this.itemsToWords(stash, format));
173
+ }
174
+
175
+ itemsToWords(items: TextItem[], formatName: string | null): Word[] {
176
+ const combinedText = combineText(items);
177
+ const words = combinedText.split(" ");
178
+ const format = formatName ? WordFormat[formatName] : null;
179
+ return words
180
+ .filter((w) => w.trim().length > 0)
181
+ .map((word) => {
182
+ var type: WordTypeEntry | null = null;
183
+ if (word.startsWith("http:")) {
184
+ this.containLinks = true;
185
+ type = WordType.LINK;
186
+ } else if (word.startsWith("www.")) {
187
+ this.containLinks = true;
188
+ word = `http://${word}`;
189
+ type = WordType.LINK;
190
+ }
191
+
192
+ if (format) {
193
+ this.formattedWords++;
194
+ }
195
+ return new Word({ string: word, type, format });
196
+ });
197
+ }
198
+ }
199
+
200
+ function combineText(textItems: TextItem[]): string {
201
+ var text = "";
202
+ var lastItem: TextItem | null = null;
203
+ textItems.forEach((textItem) => {
204
+ var textToAdd = textItem.text;
205
+ if (!text.endsWith(" ") && !textToAdd.startsWith(" ")) {
206
+ if (lastItem) {
207
+ const xDistance = textItem.x - lastItem.x - lastItem.width;
208
+ if (xDistance > 5) {
209
+ text += " ";
210
+ }
211
+ } else {
212
+ if (isListItemCharacter(textItem.text)) {
213
+ textToAdd += " ";
214
+ }
215
+ }
216
+ }
217
+ text += textToAdd;
218
+ lastItem = textItem;
219
+ });
220
+ return text;
221
+ }
222
+
223
+
224
+
@@ -1,29 +1,29 @@
1
- /**
2
- * @description Normalised PDF document metadata (title, author, creator, producer).
3
- * Handles both the modern XMP `metadata` API and the legacy `info` dictionary
4
- * returned by pdfjs, so callers always receive a consistent shape regardless of
5
- * which metadata format the PDF uses.
6
- */
7
- // Metadata of the PDF document
8
- export default class Metadata {
9
- title?: string;
10
- creator?: string;
11
- producer?: string;
12
- author?: string;
13
-
14
- constructor(originalMetadata: {
15
- metadata?: { get(key: string): string | undefined };
16
- info?: { Title?: string; Author?: string; Creator?: string; Producer?: string };
17
- }) {
18
- if (originalMetadata.metadata) {
19
- this.title = originalMetadata.metadata.get("dc:title");
20
- this.creator = originalMetadata.metadata.get("xap:creatortool");
21
- this.producer = originalMetadata.metadata.get("pdf:producer");
22
- } else {
23
- this.title = originalMetadata.info?.Title;
24
- this.author = originalMetadata.info?.Author;
25
- this.creator = originalMetadata.info?.Creator;
26
- this.producer = originalMetadata.info?.Producer;
27
- }
28
- }
29
- }
1
+ /**
2
+ * @description Normalised PDF document metadata (title, author, creator, producer).
3
+ * Handles both the modern XMP `metadata` API and the legacy `info` dictionary
4
+ * returned by pdfjs, so callers always receive a consistent shape regardless of
5
+ * which metadata format the PDF uses.
6
+ */
7
+ // Metadata of the PDF document
8
+ export default class Metadata {
9
+ title?: string;
10
+ creator?: string;
11
+ producer?: string;
12
+ author?: string;
13
+
14
+ constructor(originalMetadata: {
15
+ metadata?: { get(key: string): string | undefined };
16
+ info?: { Title?: string; Author?: string; Creator?: string; Producer?: string };
17
+ }) {
18
+ if (originalMetadata.metadata) {
19
+ this.title = originalMetadata.metadata.get("dc:title");
20
+ this.creator = originalMetadata.metadata.get("xap:creatortool");
21
+ this.producer = originalMetadata.metadata.get("pdf:producer");
22
+ } else {
23
+ this.title = originalMetadata.info?.Title;
24
+ this.author = originalMetadata.info?.Author;
25
+ this.creator = originalMetadata.info?.Creator;
26
+ this.producer = originalMetadata.info?.Producer;
27
+ }
28
+ }
29
+ }
@@ -1,16 +1,16 @@
1
- /**
2
- * @description Simple container for a single PDF page. Holds a 0-based `index`
3
- * and an `items` array whose element type evolves through the transformation
4
- * pipeline: raw `TextItem[]` → compacted `LineItem[]` → grouped `LineItemBlock[]`
5
- * → final serialized text objects.
6
- */
7
- // A page which holds PageItems displayable via PdfPageView
8
- export default class Page {
9
- index: number;
10
- items: any[];
11
-
12
- constructor(options: { index: number; items?: any[] }) {
13
- this.index = options.index;
14
- this.items = options.items || [];
15
- }
16
- }
1
+ /**
2
+ * @description Simple container for a single PDF page. Holds a 0-based `index`
3
+ * and an `items` array whose element type evolves through the transformation
4
+ * pipeline: raw `TextItem[]` → compacted `LineItem[]` → grouped `LineItemBlock[]`
5
+ * → final serialized text objects.
6
+ */
7
+ // A page which holds PageItems displayable via PdfPageView
8
+ export default class Page {
9
+ index: number;
10
+ items: any[];
11
+
12
+ constructor(options: { index: number; items?: any[] }) {
13
+ this.index = options.index;
14
+ this.items = options.items || [];
15
+ }
16
+ }
@@ -1,32 +1,32 @@
1
- /**
2
- * @description Immutable value object passed between every pipeline stage.
3
- * `pages` holds the document's page array (items mutate in type each stage),
4
- * `globals` holds cross-page statistics computed by `CalculateGlobalStats`
5
- * (most-used height, font, line distance, font-format map), and `messages`
6
- * carries diagnostic strings emitted by each transformation for debugging.
7
- */
8
- import Page from "./page";
9
-
10
- export interface ParseGlobals {
11
- mostUsedHeight?: number;
12
- mostUsedFont?: string;
13
- mostUsedDistance?: number;
14
- maxHeight?: number;
15
- maxHeightFont?: string;
16
- fontToFormats?: Map<string, string>;
17
- tocPages?: number[];
18
- headlineTypeToHeightRange?: Record<string, { min: number; max: number }>;
19
- }
20
-
21
- // The result of a PDF parse respectively a Transformation
22
- export default class ParseResult {
23
- pages: Page[];
24
- globals: ParseGlobals;
25
- messages: string[];
26
-
27
- constructor(options: { pages?: Page[]; globals?: ParseGlobals; messages?: string[] }) {
28
- this.pages = options.pages || [];
29
- this.globals = options.globals || {};
30
- this.messages = options.messages || [];
31
- }
32
- }
1
+ /**
2
+ * @description Immutable value object passed between every pipeline stage.
3
+ * `pages` holds the document's page array (items mutate in type each stage),
4
+ * `globals` holds cross-page statistics computed by `CalculateGlobalStats`
5
+ * (most-used height, font, line distance, font-format map), and `messages`
6
+ * carries diagnostic strings emitted by each transformation for debugging.
7
+ */
8
+ import Page from "./page";
9
+
10
+ export interface ParseGlobals {
11
+ mostUsedHeight?: number;
12
+ mostUsedFont?: string;
13
+ mostUsedDistance?: number;
14
+ maxHeight?: number;
15
+ maxHeightFont?: string;
16
+ fontToFormats?: Map<string, string>;
17
+ tocPages?: number[];
18
+ headlineTypeToHeightRange?: Record<string, { min: number; max: number }>;
19
+ }
20
+
21
+ // The result of a PDF parse respectively a Transformation
22
+ export default class ParseResult {
23
+ pages: Page[];
24
+ globals: ParseGlobals;
25
+ messages: string[];
26
+
27
+ constructor(options: { pages?: Page[]; globals?: ParseGlobals; messages?: string[] }) {
28
+ this.pages = options.pages || [];
29
+ this.globals = options.globals || {};
30
+ this.messages = options.messages || [];
31
+ }
32
+ }
@@ -1,29 +1,29 @@
1
- /**
2
- * @description Aggregated inline-element metadata for a `LineItem` or
3
- * `LineItemBlock`. Tracks detected footnote link numbers, footnote definitions,
4
- * whether any hyperlinks are present, and a count of bold/italic formatted words.
5
- * The `add()` method merges a child item's `ParsedElements` into the parent
6
- * block's running totals as lines are gathered into blocks.
7
- */
8
- export default class ParsedElements {
9
- footnoteLinks: number[];
10
- footnotes: string[];
11
- containLinks: boolean;
12
- formattedWords: number;
13
-
14
- constructor(options: { footnoteLinks?: number[]; footnotes?: string[]; containLinks?: boolean; formattedWords?: number }) {
15
- this.footnoteLinks = options.footnoteLinks || [];
16
- this.footnotes = options.footnotes || [];
17
- this.containLinks = options.containLinks ?? false;
18
- this.formattedWords = options.formattedWords ?? 0;
19
- }
20
-
21
- add(parsedElements: ParsedElements): void {
22
- this.footnoteLinks = this.footnoteLinks.concat(
23
- parsedElements.footnoteLinks,
24
- );
25
- this.footnotes = this.footnotes.concat(parsedElements.footnotes);
26
- this.containLinks = this.containLinks || parsedElements.containLinks;
27
- this.formattedWords += parsedElements.formattedWords;
28
- }
29
- }
1
+ /**
2
+ * @description Aggregated inline-element metadata for a `LineItem` or
3
+ * `LineItemBlock`. Tracks detected footnote link numbers, footnote definitions,
4
+ * whether any hyperlinks are present, and a count of bold/italic formatted words.
5
+ * The `add()` method merges a child item's `ParsedElements` into the parent
6
+ * block's running totals as lines are gathered into blocks.
7
+ */
8
+ export default class ParsedElements {
9
+ footnoteLinks: number[];
10
+ footnotes: string[];
11
+ containLinks: boolean;
12
+ formattedWords: number;
13
+
14
+ constructor(options: { footnoteLinks?: number[]; footnotes?: string[]; containLinks?: boolean; formattedWords?: number }) {
15
+ this.footnoteLinks = options.footnoteLinks || [];
16
+ this.footnotes = options.footnotes || [];
17
+ this.containLinks = options.containLinks ?? false;
18
+ this.formattedWords = options.formattedWords ?? 0;
19
+ }
20
+
21
+ add(parsedElements: ParsedElements): void {
22
+ this.footnoteLinks = this.footnoteLinks.concat(
23
+ parsedElements.footnoteLinks,
24
+ );
25
+ this.footnotes = this.footnotes.concat(parsedElements.footnotes);
26
+ this.containLinks = this.containLinks || parsedElements.containLinks;
27
+ this.formattedWords += parsedElements.formattedWords;
28
+ }
29
+ }