extract-pdf 0.1.21 → 0.1.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/README.md +74 -74
  2. package/package.json +1 -1
  3. package/src/models/annotation.ts +41 -41
  4. package/src/models/block-type.ts +203 -203
  5. package/src/models/line-converter.ts +224 -224
  6. package/src/models/metadata.ts +29 -29
  7. package/src/models/page.ts +16 -16
  8. package/src/models/parse-result.ts +32 -32
  9. package/src/models/parsed-elements.ts +29 -29
  10. package/src/models/stashing-stream.ts +86 -86
  11. package/src/models/text-item-line-grouper.ts +41 -41
  12. package/src/models/word.ts +31 -31
  13. package/src/pdf-to-html.ts +225 -225
  14. package/src/transforms/base/to-line-item-block-transform.ts +29 -29
  15. package/src/transforms/base/to-line-item-transform.ts +29 -29
  16. package/src/transforms/base/to-text-item-transform.ts +28 -28
  17. package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
  18. package/src/transforms/block/detect-list-levels.ts +64 -64
  19. package/src/transforms/block/gather-blocks.ts +113 -113
  20. package/src/transforms/calculate-global-stats.ts +132 -132
  21. package/src/transforms/line-item/compact-lines.ts +92 -92
  22. package/src/transforms/line-item/detect-headers.ts +173 -173
  23. package/src/transforms/line-item/detect-list-items.ts +68 -68
  24. package/src/transforms/line-item/detect-toc.ts +459 -459
  25. package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
  26. package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
  27. package/src/transforms/to-html.ts +46 -46
  28. package/src/utils/is-url-pdf.ts +33 -33
  29. package/src/utils/page-item-functions.ts +35 -35
  30. package/src/utils/string-functions.ts +124 -124
@@ -1,225 +1,225 @@
1
- /**
2
- * @fileoverview High-fidelity PDF-to-HTML conversion pipeline.
3
- * Extracts structural elements (headers, lists, code blocks) and handles page-level metadata.
4
- */
5
- import {
6
- findPageNumbers,
7
- findFirstPage,
8
- removePageNumber,
9
- } from "./utils/page-number-functions";
10
- import TextItem from "./models/text-item";
11
- import Page from "./models/page";
12
-
13
- /**
14
- * Fetch wrapper for grabbing binary content
15
- */
16
- async function grab(url: string, options: { responseType?: string; timeout?: number } = {}) {
17
- const timeout = options.timeout ? options.timeout * 1000 : 10000;
18
- const controller = new AbortController();
19
- const timeoutId = setTimeout(() => controller.abort(), timeout);
20
-
21
- try {
22
- const response = await fetch(url, {
23
- signal: controller.signal,
24
- });
25
- clearTimeout(timeoutId);
26
-
27
- if (!response.ok) {
28
- throw new Error(`HTTP ${response.status}`);
29
- }
30
-
31
- if (options.responseType === "arraybuffer") {
32
- return await response.arrayBuffer();
33
- }
34
- return await response.text();
35
- } catch (error) {
36
- clearTimeout(timeoutId);
37
- throw error;
38
- }
39
- }
40
-
41
- import CalculateGlobalStats from "./transforms/calculate-global-stats";
42
- import CompactLines from "./transforms/line-item/compact-lines";
43
- import RemoveRepetitiveElements from "./transforms/line-item/remove-repetitive-elements";
44
- import VerticalToHorizontal from "./transforms/line-item/vertical-to-horizontal";
45
- import DetectTOC from "./transforms/line-item/detect-toc";
46
- import DetectListItems from "./transforms/line-item/detect-list-items";
47
- import DetectHeaders from "./transforms/line-item/detect-headers";
48
-
49
- import GatherBlocks from "./transforms/block/gather-blocks";
50
- import DetectCodeQuoteBlocks from "./transforms/block/detect-code-quote-blocks";
51
- import DetectListLevels from "./transforms/block/detect-list-levels";
52
- import ToTextBlocks from "./transforms/to-text-blocks";
53
- import ToHTML from "./transforms/to-html";
54
- import ParseResult from "./models/parse-result";
55
-
56
- /**
57
- * Extracts formatted text from PDF with parsing of linebreaks ,
58
- * page headers, footnotes, and section headings. Supports fonts, links, bold,
59
- * italics, lists, headings, headers, footnotes, and Table of Contents,
60
- * Quotes, and Code Blocks, . Removes repeated headers, links footnote anchors to the footnote,
61
- * and preserves number of the PDF page with invisible I element.
62
- *
63
- * This function uses [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless)
64
- * to work in more environments than PDF.js-based tools:
65
- * Cloudflare workers, serverless, node.js, and front-end only.
66
- * @param {string} pdfURLOrBuffer - URL to a PDF file or buffer from fs.readFile
67
- * @param {Object} [options]
68
- * @param {boolean} options.addPageNumbers default=false - Adds # to end of each page
69
- * @param {boolean} options.removePageHeaders default=true - Removes repeated headers found on each page
70
- * @returns {string|Object} HTML formatted text
71
- * @category Extract
72
- * @author [vtempest (2025)](https://github.com/vtempest),
73
- * [pdf-to-markdown (2017)](https://github.com/jzillmann/pdf-to-markdown/tree/master),
74
- * [pdf.js (2012-)](https://github.com/mozilla/pdf.js/releases),
75
- */
76
- export async function convertPDFToHTML(
77
- pdfURLOrBuffer: any,
78
- options: { addPageNumbers?: boolean; addCitation?: boolean } = {},
79
- ) {
80
- // try {
81
- var { addPageNumbers = false, addCitation = true } = options;
82
-
83
- // pass in databuffer or download all pdf data
84
- // and convert to array buffer
85
- var buffer =
86
- typeof pdfURLOrBuffer === "string"
87
- ? await grab(pdfURLOrBuffer, {
88
- responseType: "arraybuffer",
89
- timeout: 10,
90
- })
91
- : pdfURLOrBuffer;
92
-
93
- let pdfDocument;
94
- try {
95
- let { resolvePDFJS } = await import("https://cdn.jsdelivr.net/npm/pdfjs-serverless@1.1.0/+esm" as any);
96
-
97
- const { getDocument } = await resolvePDFJS();
98
- pdfDocument = await getDocument({
99
- data: new Uint8Array(buffer),
100
- useSystemFonts: true,
101
- verbosity: 0,
102
- }).promise;
103
- } catch (e: any) {
104
- return { error: e.message };
105
- }
106
-
107
- const pages = [...Array(pdfDocument.numPages).keys()].map(
108
- (index) => new Page({ index }),
109
- );
110
-
111
- let pageIndexNumMap = {};
112
- let firstPage;
113
- for (let j = 1; j <= pdfDocument.numPages; j++) {
114
- const page = await pdfDocument.getPage(j);
115
- const textContent = await page.getTextContent();
116
-
117
- if (Object.keys(pageIndexNumMap).length < 10) {
118
- pageIndexNumMap = findPageNumbers(
119
- pageIndexNumMap,
120
- page.pageNumber - 1,
121
- textContent.items,
122
- );
123
- } else {
124
- firstPage = findFirstPage(pageIndexNumMap);
125
- break;
126
- }
127
- }
128
-
129
- let pageNum = firstPage ? firstPage.pageNum : 0;
130
- for (let j = 1; j <= pdfDocument.numPages; j++) {
131
- const page = await pdfDocument.getPage(j);
132
-
133
- // Trigger the font retrieval for the page
134
- await page.getOperatorList();
135
-
136
- const scale = 1.0;
137
- const viewport = page.getViewport({ scale });
138
- let textContent = await page.getTextContent();
139
- if (firstPage && (page as any).pageIndex >= firstPage.pageIndex) {
140
- textContent = removePageNumber(textContent as any, pageNum) as any;
141
- pageNum++;
142
- }
143
- const textItems = (textContent.items as any[]).map((item: any) => {
144
- const tx = [1, 0, 0, 1, 0, 0];
145
- for (let i = 0; i < 6; i++) {
146
- tx[i] += item.transform[i] * viewport.transform[i % 2 ? 3 : 0];
147
- if (i % 2) {
148
- tx[i + 1] += item.transform[i] * viewport.transform[1];
149
- }
150
- }
151
-
152
- const fontHeight = Math.sqrt(tx[2] * tx[2] + tx[3] * tx[3]);
153
- const dividedHeight = item.height / fontHeight;
154
- return new TextItem({
155
- x: Math.round(item.transform[4]),
156
- y: Math.round(item.transform[5]),
157
- width: Math.round(item.width),
158
- height: Math.round(dividedHeight <= 1 ? item.height : dividedHeight),
159
- text: item.str,
160
- font: item.fontName,
161
- });
162
- });
163
- pages[page.pageNumber - 1].items = textItems;
164
- }
165
-
166
- var parseResult = new ParseResult({ pages });
167
-
168
- let lastTransformation: (typeof transformations)[number] | undefined,
169
- transformations = [
170
- new CalculateGlobalStats(),
171
- new CompactLines(),
172
- new RemoveRepetitiveElements(),
173
- new VerticalToHorizontal(),
174
- new DetectTOC(),
175
- new DetectHeaders(),
176
- new DetectListItems(),
177
-
178
- new GatherBlocks(),
179
- new DetectCodeQuoteBlocks(),
180
- new DetectListLevels(),
181
-
182
- new ToTextBlocks(),
183
- new ToHTML(),
184
- ];
185
-
186
- transformations?.forEach((transformation) => {
187
- if (lastTransformation) {
188
- parseResult = lastTransformation.completeTransform(parseResult);
189
- }
190
- parseResult = transformation.transform(parseResult);
191
- lastTransformation = transformation;
192
- });
193
-
194
- var html = parseResult.pages.reduce((acc, page, pageNumber) => {
195
- return (
196
- acc +
197
- `<p id="page-${pageNumber + 1}">${
198
- addPageNumbers ? ` [${pageNumber + 1}] ` : ""
199
- }${page.items.join('</p><p id="page-' + pageNumber + '">')}</p>`
200
- );
201
- }, "");
202
-
203
- if (addCitation) {
204
- // Get metadata
205
- // avoid using date as it is unreliable sand generally file mod date
206
- var metadata = await pdfDocument.getMetadata();
207
- var { Author: author, Title: title } = metadata.info as any;
208
- // date =
209
- // date.slice(2, 6) + "-" + date.slice(6, 8) + "-" + date.slice(8, 10);
210
- // date = date ? new Date(date)?.toISOString().split("T")[0] : null;
211
-
212
- //look for date in first page
213
- // date = chrono
214
- // .parseDate(content.slice(0, 400))
215
- // ?.toISOString()
216
- // .split("T")[0];
217
- // // || date;
218
-
219
- title = html.slice(0, 400).match(/<h[0-9]>(.*?)<\/h[0-9]>/)?.[1] || title;
220
- }
221
-
222
- return { author, title, html, format: "pdf" };
223
- }
224
-
225
-
1
+ /**
2
+ * @fileoverview High-fidelity PDF-to-HTML conversion pipeline.
3
+ * Extracts structural elements (headers, lists, code blocks) and handles page-level metadata.
4
+ */
5
+ import {
6
+ findPageNumbers,
7
+ findFirstPage,
8
+ removePageNumber,
9
+ } from "./utils/page-number-functions";
10
+ import TextItem from "./models/text-item";
11
+ import Page from "./models/page";
12
+
13
+ /**
14
+ * Fetch wrapper for grabbing binary content
15
+ */
16
+ async function grab(url: string, options: { responseType?: string; timeout?: number } = {}) {
17
+ const timeout = options.timeout ? options.timeout * 1000 : 10000;
18
+ const controller = new AbortController();
19
+ const timeoutId = setTimeout(() => controller.abort(), timeout);
20
+
21
+ try {
22
+ const response = await fetch(url, {
23
+ signal: controller.signal,
24
+ });
25
+ clearTimeout(timeoutId);
26
+
27
+ if (!response.ok) {
28
+ throw new Error(`HTTP ${response.status}`);
29
+ }
30
+
31
+ if (options.responseType === "arraybuffer") {
32
+ return await response.arrayBuffer();
33
+ }
34
+ return await response.text();
35
+ } catch (error) {
36
+ clearTimeout(timeoutId);
37
+ throw error;
38
+ }
39
+ }
40
+
41
+ import CalculateGlobalStats from "./transforms/calculate-global-stats";
42
+ import CompactLines from "./transforms/line-item/compact-lines";
43
+ import RemoveRepetitiveElements from "./transforms/line-item/remove-repetitive-elements";
44
+ import VerticalToHorizontal from "./transforms/line-item/vertical-to-horizontal";
45
+ import DetectTOC from "./transforms/line-item/detect-toc";
46
+ import DetectListItems from "./transforms/line-item/detect-list-items";
47
+ import DetectHeaders from "./transforms/line-item/detect-headers";
48
+
49
+ import GatherBlocks from "./transforms/block/gather-blocks";
50
+ import DetectCodeQuoteBlocks from "./transforms/block/detect-code-quote-blocks";
51
+ import DetectListLevels from "./transforms/block/detect-list-levels";
52
+ import ToTextBlocks from "./transforms/to-text-blocks";
53
+ import ToHTML from "./transforms/to-html";
54
+ import ParseResult from "./models/parse-result";
55
+
56
+ /**
57
+ * Extracts formatted text from PDF with parsing of linebreaks ,
58
+ * page headers, footnotes, and section headings. Supports fonts, links, bold,
59
+ * italics, lists, headings, headers, footnotes, and Table of Contents,
60
+ * Quotes, and Code Blocks, . Removes repeated headers, links footnote anchors to the footnote,
61
+ * and preserves number of the PDF page with invisible I element.
62
+ *
63
+ * This function uses [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless)
64
+ * to work in more environments than PDF.js-based tools:
65
+ * Cloudflare workers, serverless, node.js, and front-end only.
66
+ * @param {string} pdfURLOrBuffer - URL to a PDF file or buffer from fs.readFile
67
+ * @param {Object} [options]
68
+ * @param {boolean} options.addPageNumbers default=false - Adds # to end of each page
69
+ * @param {boolean} options.removePageHeaders default=true - Removes repeated headers found on each page
70
+ * @returns {string|Object} HTML formatted text
71
+ * @category Extract
72
+ * @author [vtempest (2025)](https://github.com/vtempest),
73
+ * [pdf-to-markdown (2017)](https://github.com/jzillmann/pdf-to-markdown/tree/master),
74
+ * [pdf.js (2012-)](https://github.com/mozilla/pdf.js/releases),
75
+ */
76
+ export async function convertPDFToHTML(
77
+ pdfURLOrBuffer: any,
78
+ options: { addPageNumbers?: boolean; addCitation?: boolean } = {},
79
+ ) {
80
+ // try {
81
+ var { addPageNumbers = false, addCitation = true } = options;
82
+
83
+ // pass in databuffer or download all pdf data
84
+ // and convert to array buffer
85
+ var buffer =
86
+ typeof pdfURLOrBuffer === "string"
87
+ ? await grab(pdfURLOrBuffer, {
88
+ responseType: "arraybuffer",
89
+ timeout: 10,
90
+ })
91
+ : pdfURLOrBuffer;
92
+
93
+ let pdfDocument;
94
+ try {
95
+ let { resolvePDFJS } = await import("https://cdn.jsdelivr.net/npm/pdfjs-serverless@1.1.0/+esm" as any);
96
+
97
+ const { getDocument } = await resolvePDFJS();
98
+ pdfDocument = await getDocument({
99
+ data: new Uint8Array(buffer),
100
+ useSystemFonts: true,
101
+ verbosity: 0,
102
+ }).promise;
103
+ } catch (e: any) {
104
+ return { error: e.message };
105
+ }
106
+
107
+ const pages = [...Array(pdfDocument.numPages).keys()].map(
108
+ (index) => new Page({ index }),
109
+ );
110
+
111
+ let pageIndexNumMap = {};
112
+ let firstPage;
113
+ for (let j = 1; j <= pdfDocument.numPages; j++) {
114
+ const page = await pdfDocument.getPage(j);
115
+ const textContent = await page.getTextContent();
116
+
117
+ if (Object.keys(pageIndexNumMap).length < 10) {
118
+ pageIndexNumMap = findPageNumbers(
119
+ pageIndexNumMap,
120
+ page.pageNumber - 1,
121
+ textContent.items,
122
+ );
123
+ } else {
124
+ firstPage = findFirstPage(pageIndexNumMap);
125
+ break;
126
+ }
127
+ }
128
+
129
+ let pageNum = firstPage ? firstPage.pageNum : 0;
130
+ for (let j = 1; j <= pdfDocument.numPages; j++) {
131
+ const page = await pdfDocument.getPage(j);
132
+
133
+ // Trigger the font retrieval for the page
134
+ await page.getOperatorList();
135
+
136
+ const scale = 1.0;
137
+ const viewport = page.getViewport({ scale });
138
+ let textContent = await page.getTextContent();
139
+ if (firstPage && (page as any).pageIndex >= firstPage.pageIndex) {
140
+ textContent = removePageNumber(textContent as any, pageNum) as any;
141
+ pageNum++;
142
+ }
143
+ const textItems = (textContent.items as any[]).map((item: any) => {
144
+ const tx = [1, 0, 0, 1, 0, 0];
145
+ for (let i = 0; i < 6; i++) {
146
+ tx[i] += item.transform[i] * viewport.transform[i % 2 ? 3 : 0];
147
+ if (i % 2) {
148
+ tx[i + 1] += item.transform[i] * viewport.transform[1];
149
+ }
150
+ }
151
+
152
+ const fontHeight = Math.sqrt(tx[2] * tx[2] + tx[3] * tx[3]);
153
+ const dividedHeight = item.height / fontHeight;
154
+ return new TextItem({
155
+ x: Math.round(item.transform[4]),
156
+ y: Math.round(item.transform[5]),
157
+ width: Math.round(item.width),
158
+ height: Math.round(dividedHeight <= 1 ? item.height : dividedHeight),
159
+ text: item.str,
160
+ font: item.fontName,
161
+ });
162
+ });
163
+ pages[page.pageNumber - 1].items = textItems;
164
+ }
165
+
166
+ var parseResult = new ParseResult({ pages });
167
+
168
+ let lastTransformation: (typeof transformations)[number] | undefined,
169
+ transformations = [
170
+ new CalculateGlobalStats(),
171
+ new CompactLines(),
172
+ new RemoveRepetitiveElements(),
173
+ new VerticalToHorizontal(),
174
+ new DetectTOC(),
175
+ new DetectHeaders(),
176
+ new DetectListItems(),
177
+
178
+ new GatherBlocks(),
179
+ new DetectCodeQuoteBlocks(),
180
+ new DetectListLevels(),
181
+
182
+ new ToTextBlocks(),
183
+ new ToHTML(),
184
+ ];
185
+
186
+ transformations?.forEach((transformation) => {
187
+ if (lastTransformation) {
188
+ parseResult = lastTransformation.completeTransform(parseResult);
189
+ }
190
+ parseResult = transformation.transform(parseResult);
191
+ lastTransformation = transformation;
192
+ });
193
+
194
+ var html = parseResult.pages.reduce((acc, page, pageNumber) => {
195
+ return (
196
+ acc +
197
+ `<p id="page-${pageNumber + 1}">${
198
+ addPageNumbers ? ` [${pageNumber + 1}] ` : ""
199
+ }${page.items.join('</p><p id="page-' + pageNumber + '">')}</p>`
200
+ );
201
+ }, "");
202
+
203
+ if (addCitation) {
204
+ // Get metadata
205
+ // avoid using date as it is unreliable sand generally file mod date
206
+ var metadata = await pdfDocument.getMetadata();
207
+ var { Author: author, Title: title } = metadata.info as any;
208
+ // date =
209
+ // date.slice(2, 6) + "-" + date.slice(6, 8) + "-" + date.slice(8, 10);
210
+ // date = date ? new Date(date)?.toISOString().split("T")[0] : null;
211
+
212
+ //look for date in first page
213
+ // date = chrono
214
+ // .parseDate(content.slice(0, 400))
215
+ // ?.toISOString()
216
+ // .split("T")[0];
217
+ // // || date;
218
+
219
+ title = html.slice(0, 400).match(/<h[0-9]>(.*?)<\/h[0-9]>/)?.[1] || title;
220
+ }
221
+
222
+ return { author, title, html, format: "pdf" };
223
+ }
224
+
225
+
@@ -1,29 +1,29 @@
1
- /**
2
- * @description Abstract base for pipeline stages that produce `LineItemBlock`-typed
3
- * page items (GatherBlocks, DetectCodeQuoteBlocks, DetectListLevels). Provides
4
- * `completeTransform` which strips `REMOVED_ANNOTATION` items and clears all
5
- * remaining annotations before passing the result to the next stage.
6
- */
7
- import Transformation from './transformation'
8
- import LineItemBlock from '../../models/line-item-block'
9
- import ParseResult from '../../models/parse-result'
10
- import { REMOVED_ANNOTATION } from '../../models/annotation'
11
-
12
- // Abstract class for transformations producing LineItemBlock(s) to be shown in the LineItemBlockPageView
13
- export default class ToLineItemBlockTransformation extends Transformation {
14
- constructor(name: string) {
15
- super(name, LineItemBlock.name)
16
- if (this.constructor === ToLineItemBlockTransformation) {
17
- throw new TypeError('Can not construct abstract class.')
18
- }
19
- }
20
-
21
- completeTransform(parseResult: ParseResult): ParseResult {
22
- parseResult.messages = []
23
- parseResult.pages.forEach(page => {
24
- page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
25
- page.items.forEach(item => (item.annotation = null))
26
- })
27
- return parseResult
28
- }
29
- }
1
+ /**
2
+ * @description Abstract base for pipeline stages that produce `LineItemBlock`-typed
3
+ * page items (GatherBlocks, DetectCodeQuoteBlocks, DetectListLevels). Provides
4
+ * `completeTransform` which strips `REMOVED_ANNOTATION` items and clears all
5
+ * remaining annotations before passing the result to the next stage.
6
+ */
7
+ import Transformation from './transformation'
8
+ import LineItemBlock from '../../models/line-item-block'
9
+ import ParseResult from '../../models/parse-result'
10
+ import { REMOVED_ANNOTATION } from '../../models/annotation'
11
+
12
+ // Abstract class for transformations producing LineItemBlock(s) to be shown in the LineItemBlockPageView
13
+ export default class ToLineItemBlockTransformation extends Transformation {
14
+ constructor(name: string) {
15
+ super(name, LineItemBlock.name)
16
+ if (this.constructor === ToLineItemBlockTransformation) {
17
+ throw new TypeError('Can not construct abstract class.')
18
+ }
19
+ }
20
+
21
+ completeTransform(parseResult: ParseResult): ParseResult {
22
+ parseResult.messages = []
23
+ parseResult.pages.forEach(page => {
24
+ page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
25
+ page.items.forEach(item => (item.annotation = null))
26
+ })
27
+ return parseResult
28
+ }
29
+ }
@@ -1,29 +1,29 @@
1
- /**
2
- * @description Abstract base for pipeline stages that produce `LineItem`-typed
3
- * page items (CompactLines, RemoveRepetitiveElements, VerticalToHorizontal,
4
- * DetectTOC, DetectHeaders, DetectListItems). Provides `completeTransform` which
5
- * strips `REMOVED_ANNOTATION` items and clears annotations before the next stage.
6
- */
7
- import Transformation from './transformation'
8
- import LineItem from '../../models/line-item'
9
- import ParseResult from '../../models/parse-result'
10
- import { REMOVED_ANNOTATION } from '../../models/annotation'
11
-
12
- // Abstract class for transformations producing LineItem(s) to be shown in the LineItemPageView
13
- export default class ToLineItemTransformation extends Transformation {
14
- constructor(name: string) {
15
- super(name, LineItem.name)
16
- if (this.constructor === ToLineItemTransformation) {
17
- throw new TypeError('Can not construct abstract class.')
18
- }
19
- }
20
-
21
- completeTransform(parseResult: ParseResult): ParseResult {
22
- parseResult.messages = []
23
- parseResult.pages.forEach(page => {
24
- page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
25
- page.items.forEach(item => (item.annotation = null))
26
- })
27
- return parseResult
28
- }
29
- }
1
+ /**
2
+ * @description Abstract base for pipeline stages that produce `LineItem`-typed
3
+ * page items (CompactLines, RemoveRepetitiveElements, VerticalToHorizontal,
4
+ * DetectTOC, DetectHeaders, DetectListItems). Provides `completeTransform` which
5
+ * strips `REMOVED_ANNOTATION` items and clears annotations before the next stage.
6
+ */
7
+ import Transformation from './transformation'
8
+ import LineItem from '../../models/line-item'
9
+ import ParseResult from '../../models/parse-result'
10
+ import { REMOVED_ANNOTATION } from '../../models/annotation'
11
+
12
+ // Abstract class for transformations producing LineItem(s) to be shown in the LineItemPageView
13
+ export default class ToLineItemTransformation extends Transformation {
14
+ constructor(name: string) {
15
+ super(name, LineItem.name)
16
+ if (this.constructor === ToLineItemTransformation) {
17
+ throw new TypeError('Can not construct abstract class.')
18
+ }
19
+ }
20
+
21
+ completeTransform(parseResult: ParseResult): ParseResult {
22
+ parseResult.messages = []
23
+ parseResult.pages.forEach(page => {
24
+ page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
25
+ page.items.forEach(item => (item.annotation = null))
26
+ })
27
+ return parseResult
28
+ }
29
+ }
@@ -1,28 +1,28 @@
1
- /**
2
- * @description Abstract base for pipeline stages that operate on raw `TextItem`
3
- * page items (the earliest stages, currently only `CalculateGlobalStats`). Provides
4
- * the standard `completeTransform` cleanup that removes `REMOVED_ANNOTATION` items
5
- * and clears annotations before passing control to the next transformation.
6
- */
7
- import Transformation from './transformation'
8
- import TextItem from '../../models/text-item'
9
- import ParseResult from '../../models/parse-result'
10
- import { REMOVED_ANNOTATION } from '../../models/annotation'
11
-
12
- export default class ToTextItemTransformation extends Transformation {
13
- constructor(name: string) {
14
- super(name, TextItem.name)
15
- if (this.constructor === ToTextItemTransformation) {
16
- throw new TypeError('Can not construct abstract class.')
17
- }
18
- }
19
-
20
- completeTransform(parseResult: ParseResult): ParseResult {
21
- parseResult.messages = []
22
- parseResult.pages.forEach(page => {
23
- page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
24
- page.items.forEach(item => (item.annotation = null))
25
- })
26
- return parseResult
27
- }
28
- }
1
+ /**
2
+ * @description Abstract base for pipeline stages that operate on raw `TextItem`
3
+ * page items (the earliest stages, currently only `CalculateGlobalStats`). Provides
4
+ * the standard `completeTransform` cleanup that removes `REMOVED_ANNOTATION` items
5
+ * and clears annotations before passing control to the next transformation.
6
+ */
7
+ import Transformation from './transformation'
8
+ import TextItem from '../../models/text-item'
9
+ import ParseResult from '../../models/parse-result'
10
+ import { REMOVED_ANNOTATION } from '../../models/annotation'
11
+
12
+ export default class ToTextItemTransformation extends Transformation {
13
+ constructor(name: string) {
14
+ super(name, TextItem.name)
15
+ if (this.constructor === ToTextItemTransformation) {
16
+ throw new TypeError('Can not construct abstract class.')
17
+ }
18
+ }
19
+
20
+ completeTransform(parseResult: ParseResult): ParseResult {
21
+ parseResult.messages = []
22
+ parseResult.pages.forEach(page => {
23
+ page.items = page.items.filter(item => !item.annotation || item.annotation !== REMOVED_ANNOTATION)
24
+ page.items.forEach(item => (item.annotation = null))
25
+ })
26
+ return parseResult
27
+ }
28
+ }