extract-pdf 0.1.20 → 0.1.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -74
- package/package.json +1 -1
- package/src/models/annotation.ts +41 -41
- package/src/models/block-type.ts +203 -203
- package/src/models/line-converter.ts +224 -224
- package/src/models/metadata.ts +29 -29
- package/src/models/page.ts +16 -16
- package/src/models/parse-result.ts +32 -32
- package/src/models/parsed-elements.ts +29 -29
- package/src/models/stashing-stream.ts +86 -86
- package/src/models/text-item-line-grouper.ts +41 -41
- package/src/models/word.ts +31 -31
- package/src/pdf-to-html.ts +225 -225
- package/src/transforms/base/to-line-item-block-transform.ts +29 -29
- package/src/transforms/base/to-line-item-transform.ts +29 -29
- package/src/transforms/base/to-text-item-transform.ts +28 -28
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
- package/src/transforms/block/detect-list-levels.ts +64 -64
- package/src/transforms/block/gather-blocks.ts +113 -113
- package/src/transforms/calculate-global-stats.ts +132 -132
- package/src/transforms/line-item/compact-lines.ts +92 -92
- package/src/transforms/line-item/detect-headers.ts +173 -173
- package/src/transforms/line-item/detect-list-items.ts +68 -68
- package/src/transforms/line-item/detect-toc.ts +459 -459
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
- package/src/transforms/to-html.ts +46 -46
- package/src/utils/is-url-pdf.ts +33 -33
- package/src/utils/page-item-functions.ts +35 -35
- package/src/utils/string-functions.ts +124 -124
package/README.md
CHANGED
|
@@ -1,74 +1,74 @@
|
|
|
1
|
-
# extract-pdf
|
|
2
|
-
|
|
3
|
-
Converts a PDF (URL or `ArrayBuffer`) into clean HTML with structural tagging — headings, lists, footnotes, code blocks, bold/italic, and Table of Contents entries. Works in Node.js, Cloudflare Workers, and browser environments via [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless).
|
|
4
|
-
|
|
5
|
-
## Install
|
|
6
|
-
|
|
7
|
-
```sh
|
|
8
|
-
bun add extract-pdf
|
|
9
|
-
```
|
|
10
|
-
|
|
11
|
-
## Usage
|
|
12
|
-
|
|
13
|
-
```ts
|
|
14
|
-
import { convertPDFToHTML } from "extract-pdf";
|
|
15
|
-
|
|
16
|
-
const { html, title, author } = await convertPDFToHTML(
|
|
17
|
-
"https://example.com/paper.pdf",
|
|
18
|
-
);
|
|
19
|
-
// or pass an ArrayBuffer from fs.readFile / fetch
|
|
20
|
-
const { html } = await convertPDFToHTML(buffer, { addPageNumbers: true });
|
|
21
|
-
```
|
|
22
|
-
|
|
23
|
-
### Options
|
|
24
|
-
|
|
25
|
-
| Option | Default | Description |
|
|
26
|
-
| ---------------- | ------- | ------------------------------------------------------------------------------------------ |
|
|
27
|
-
| `addPageNumbers` | `false` | Inserts `[n]` markers at each page boundary |
|
|
28
|
-
| `addCitation` | `true` | Reads PDF metadata and first-page heading to populate `title`/`author` in the return value |
|
|
29
|
-
|
|
30
|
-
### Return value
|
|
31
|
-
|
|
32
|
-
```ts
|
|
33
|
-
{ html: string, title?: string, author?: string, format: "pdf" }
|
|
34
|
-
```
|
|
35
|
-
|
|
36
|
-
## Pipeline
|
|
37
|
-
|
|
38
|
-
The conversion runs a sequential chain of transformations on a `ParseResult` (pages → items):
|
|
39
|
-
|
|
40
|
-
```
|
|
41
|
-
Raw pdfjs text spans
|
|
42
|
-
→ CalculateGlobalStats — font heights, distances, format map
|
|
43
|
-
→ CompactLines — merge spans on the same y-line into LineItems
|
|
44
|
-
→ RemoveRepetitiveElements — strip recurring page headers/footers
|
|
45
|
-
→ VerticalToHorizontal — rotate vertical character runs
|
|
46
|
-
→ DetectTOC — identify Table of Contents pages, link headings
|
|
47
|
-
→ DetectHeaders — classify items as H1–H6 by font height
|
|
48
|
-
→ DetectListItems — detect bullet/numbered list items
|
|
49
|
-
→ GatherBlocks — group adjacent same-type lines into blocks
|
|
50
|
-
→ DetectCodeQuoteBlocks — mark indented blocks as CODE
|
|
51
|
-
→ DetectListLevels — add indentation for nested list levels
|
|
52
|
-
→ ToTextBlocks — flatten blocks to { category, text } pairs
|
|
53
|
-
→ ToHTML — render pairs as <p>, <h1>–<h6>, <ul>, <code>
|
|
54
|
-
```
|
|
55
|
-
|
|
56
|
-
## Folder structure
|
|
57
|
-
|
|
58
|
-
```
|
|
59
|
-
src/
|
|
60
|
-
pdf-to-html.ts — main entry point (convertPDFToHTML)
|
|
61
|
-
models/ — data classes: Page, ParseResult, TextItem,
|
|
62
|
-
│ LineItem, LineItemBlock, Word, BlockType, …
|
|
63
|
-
transforms/
|
|
64
|
-
│ base/ — abstract Transformation, ToLineItem*, ToLineItemBlock*
|
|
65
|
-
│ line-item/ — per-line-item transformations
|
|
66
|
-
│ block/ — per-block transformations
|
|
67
|
-
│ calculate-global-stats.ts
|
|
68
|
-
│ to-text-blocks.ts
|
|
69
|
-
│ to-html.ts
|
|
70
|
-
utils/
|
|
71
|
-
string-functions.ts
|
|
72
|
-
page-item-functions.ts
|
|
73
|
-
page-number-functions.ts
|
|
74
|
-
```
|
|
1
|
+
# extract-pdf
|
|
2
|
+
|
|
3
|
+
Converts a PDF (URL or `ArrayBuffer`) into clean HTML with structural tagging — headings, lists, footnotes, code blocks, bold/italic, and Table of Contents entries. Works in Node.js, Cloudflare Workers, and browser environments via [pdfjs-serverless](https://github.com/johannschopplich/pdfjs-serverless).
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
bun add extract-pdf
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Usage
|
|
12
|
+
|
|
13
|
+
```ts
|
|
14
|
+
import { convertPDFToHTML } from "extract-pdf";
|
|
15
|
+
|
|
16
|
+
const { html, title, author } = await convertPDFToHTML(
|
|
17
|
+
"https://example.com/paper.pdf",
|
|
18
|
+
);
|
|
19
|
+
// or pass an ArrayBuffer from fs.readFile / fetch
|
|
20
|
+
const { html } = await convertPDFToHTML(buffer, { addPageNumbers: true });
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
### Options
|
|
24
|
+
|
|
25
|
+
| Option | Default | Description |
|
|
26
|
+
| ---------------- | ------- | ------------------------------------------------------------------------------------------ |
|
|
27
|
+
| `addPageNumbers` | `false` | Inserts `[n]` markers at each page boundary |
|
|
28
|
+
| `addCitation` | `true` | Reads PDF metadata and first-page heading to populate `title`/`author` in the return value |
|
|
29
|
+
|
|
30
|
+
### Return value
|
|
31
|
+
|
|
32
|
+
```ts
|
|
33
|
+
{ html: string, title?: string, author?: string, format: "pdf" }
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## Pipeline
|
|
37
|
+
|
|
38
|
+
The conversion runs a sequential chain of transformations on a `ParseResult` (pages → items):
|
|
39
|
+
|
|
40
|
+
```
|
|
41
|
+
Raw pdfjs text spans
|
|
42
|
+
→ CalculateGlobalStats — font heights, distances, format map
|
|
43
|
+
→ CompactLines — merge spans on the same y-line into LineItems
|
|
44
|
+
→ RemoveRepetitiveElements — strip recurring page headers/footers
|
|
45
|
+
→ VerticalToHorizontal — rotate vertical character runs
|
|
46
|
+
→ DetectTOC — identify Table of Contents pages, link headings
|
|
47
|
+
→ DetectHeaders — classify items as H1–H6 by font height
|
|
48
|
+
→ DetectListItems — detect bullet/numbered list items
|
|
49
|
+
→ GatherBlocks — group adjacent same-type lines into blocks
|
|
50
|
+
→ DetectCodeQuoteBlocks — mark indented blocks as CODE
|
|
51
|
+
→ DetectListLevels — add indentation for nested list levels
|
|
52
|
+
→ ToTextBlocks — flatten blocks to { category, text } pairs
|
|
53
|
+
→ ToHTML — render pairs as <p>, <h1>–<h6>, <ul>, <code>
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Folder structure
|
|
57
|
+
|
|
58
|
+
```
|
|
59
|
+
src/
|
|
60
|
+
pdf-to-html.ts — main entry point (convertPDFToHTML)
|
|
61
|
+
models/ — data classes: Page, ParseResult, TextItem,
|
|
62
|
+
│ LineItem, LineItemBlock, Word, BlockType, …
|
|
63
|
+
transforms/
|
|
64
|
+
│ base/ — abstract Transformation, ToLineItem*, ToLineItemBlock*
|
|
65
|
+
│ line-item/ — per-line-item transformations
|
|
66
|
+
│ block/ — per-block transformations
|
|
67
|
+
│ calculate-global-stats.ts
|
|
68
|
+
│ to-text-blocks.ts
|
|
69
|
+
│ to-html.ts
|
|
70
|
+
utils/
|
|
71
|
+
string-functions.ts
|
|
72
|
+
page-item-functions.ts
|
|
73
|
+
page-number-functions.ts
|
|
74
|
+
```
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "extract-pdf",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.22",
|
|
4
4
|
"description": "Convert a PDF (URL or ArrayBuffer) into clean HTML with structural tagging — headings, lists, footnotes, code blocks, bold/italic. Works in Node.js, Cloudflare Workers, and browser environments.",
|
|
5
5
|
"author": "vtempest <grokthiscontact@gmail.com>",
|
|
6
6
|
"license": "rights.institute/PROSPER",
|
package/src/models/annotation.ts
CHANGED
|
@@ -1,41 +1,41 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Diff-annotation markers used to tag document items during pipeline
|
|
3
|
-
* transformations. Each `Annotation` carries a human-readable `category` and a
|
|
4
|
-
* display `color`. Pre-built singletons (`ADDED_ANNOTATION`, `REMOVED_ANNOTATION`,
|
|
5
|
-
* `UNCHANGED_ANNOTATION`, `DETECTED_ANNOTATION`, `MODIFIED_ANNOTATION`) are
|
|
6
|
-
* attached to page items so debug views can highlight what each stage changed.
|
|
7
|
-
*/
|
|
8
|
-
export default class Annotation {
|
|
9
|
-
category: string;
|
|
10
|
-
color: string;
|
|
11
|
-
|
|
12
|
-
constructor(options: { category: string; color: string }) {
|
|
13
|
-
this.category = options.category;
|
|
14
|
-
this.color = options.color;
|
|
15
|
-
}
|
|
16
|
-
}
|
|
17
|
-
|
|
18
|
-
export const ADDED_ANNOTATION = new Annotation({
|
|
19
|
-
category: "Added",
|
|
20
|
-
color: "green",
|
|
21
|
-
});
|
|
22
|
-
|
|
23
|
-
export const REMOVED_ANNOTATION = new Annotation({
|
|
24
|
-
category: "Removed",
|
|
25
|
-
color: "red",
|
|
26
|
-
});
|
|
27
|
-
|
|
28
|
-
export const UNCHANGED_ANNOTATION = new Annotation({
|
|
29
|
-
category: "Unchanged",
|
|
30
|
-
color: "brown",
|
|
31
|
-
});
|
|
32
|
-
|
|
33
|
-
export const DETECTED_ANNOTATION = new Annotation({
|
|
34
|
-
category: "Detected",
|
|
35
|
-
color: "green",
|
|
36
|
-
});
|
|
37
|
-
|
|
38
|
-
export const MODIFIED_ANNOTATION = new Annotation({
|
|
39
|
-
category: "Modified",
|
|
40
|
-
color: "green",
|
|
41
|
-
});
|
|
1
|
+
/**
|
|
2
|
+
* @description Diff-annotation markers used to tag document items during pipeline
|
|
3
|
+
* transformations. Each `Annotation` carries a human-readable `category` and a
|
|
4
|
+
* display `color`. Pre-built singletons (`ADDED_ANNOTATION`, `REMOVED_ANNOTATION`,
|
|
5
|
+
* `UNCHANGED_ANNOTATION`, `DETECTED_ANNOTATION`, `MODIFIED_ANNOTATION`) are
|
|
6
|
+
* attached to page items so debug views can highlight what each stage changed.
|
|
7
|
+
*/
|
|
8
|
+
export default class Annotation {
|
|
9
|
+
category: string;
|
|
10
|
+
color: string;
|
|
11
|
+
|
|
12
|
+
constructor(options: { category: string; color: string }) {
|
|
13
|
+
this.category = options.category;
|
|
14
|
+
this.color = options.color;
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export const ADDED_ANNOTATION = new Annotation({
|
|
19
|
+
category: "Added",
|
|
20
|
+
color: "green",
|
|
21
|
+
});
|
|
22
|
+
|
|
23
|
+
export const REMOVED_ANNOTATION = new Annotation({
|
|
24
|
+
category: "Removed",
|
|
25
|
+
color: "red",
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
export const UNCHANGED_ANNOTATION = new Annotation({
|
|
29
|
+
category: "Unchanged",
|
|
30
|
+
color: "brown",
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
export const DETECTED_ANNOTATION = new Annotation({
|
|
34
|
+
category: "Detected",
|
|
35
|
+
color: "green",
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
export const MODIFIED_ANNOTATION = new Annotation({
|
|
39
|
+
category: "Modified",
|
|
40
|
+
color: "green",
|
|
41
|
+
});
|
package/src/models/block-type.ts
CHANGED
|
@@ -1,203 +1,203 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Enum-like registry of semantic document block types (H1–H6, TOC,
|
|
3
|
-
* FOOTNOTES, CODE, LIST, PARAGRAPH). Each entry carries a `toText()` serializer
|
|
4
|
-
* that converts a `LineItemBlock` to HTML. Helper methods `isHeadline`,
|
|
5
|
-
* `blockToText`, and `headlineByLevel` are mixed in on the registry object.
|
|
6
|
-
* The inner `linesToText` function handles inline bold/italic/link/footnote
|
|
7
|
-
* formatting by tracking open `WordFormat` spans across words and lines.
|
|
8
|
-
*/
|
|
9
|
-
|
|
10
|
-
const BlockTypes: Record<string, any> = {
|
|
11
|
-
H1: {
|
|
12
|
-
name: "H1",
|
|
13
|
-
headline: true,
|
|
14
|
-
headlineLevel: 1,
|
|
15
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
16
|
-
return "<h1>" + linesToText(block.items, true) + "</h1>";
|
|
17
|
-
},
|
|
18
|
-
},
|
|
19
|
-
H2: {
|
|
20
|
-
name: "H2",
|
|
21
|
-
headline: true,
|
|
22
|
-
headlineLevel: 2,
|
|
23
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
24
|
-
return "<h2>" + linesToText(block.items, true) + "</h2>";
|
|
25
|
-
},
|
|
26
|
-
},
|
|
27
|
-
H3: {
|
|
28
|
-
name: "H3",
|
|
29
|
-
headline: true,
|
|
30
|
-
headlineLevel: 3,
|
|
31
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
32
|
-
return "<h3>" + linesToText(block.items, true) + "</h3>";
|
|
33
|
-
},
|
|
34
|
-
},
|
|
35
|
-
H4: {
|
|
36
|
-
name: "H4",
|
|
37
|
-
headline: true,
|
|
38
|
-
headlineLevel: 4,
|
|
39
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
40
|
-
return "<h4>" + linesToText(block.items, true) + "</h4>";
|
|
41
|
-
},
|
|
42
|
-
},
|
|
43
|
-
H5: {
|
|
44
|
-
name: "H5",
|
|
45
|
-
headline: true,
|
|
46
|
-
headlineLevel: 5,
|
|
47
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
48
|
-
return "<h5>" + linesToText(block.items, true) + "</h5>";
|
|
49
|
-
},
|
|
50
|
-
},
|
|
51
|
-
H6: {
|
|
52
|
-
name: "H6",
|
|
53
|
-
headline: true,
|
|
54
|
-
headlineLevel: 6,
|
|
55
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
56
|
-
return "<h6>" + linesToText(block.items, true) + "</h6>";
|
|
57
|
-
},
|
|
58
|
-
},
|
|
59
|
-
TOC: {
|
|
60
|
-
name: "TOC",
|
|
61
|
-
mergeToBlock: true,
|
|
62
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
63
|
-
return linesToText(block.items, true);
|
|
64
|
-
},
|
|
65
|
-
},
|
|
66
|
-
FOOTNOTES: {
|
|
67
|
-
name: "FOOTNOTES",
|
|
68
|
-
mergeToBlock: true,
|
|
69
|
-
mergeFollowingNonTypedItems: true,
|
|
70
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
71
|
-
return "<p>" + linesToText(block.items, false) + "</p>";
|
|
72
|
-
},
|
|
73
|
-
},
|
|
74
|
-
CODE: {
|
|
75
|
-
name: "CODE",
|
|
76
|
-
mergeToBlock: true,
|
|
77
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
78
|
-
return "<code>" + linesToText(block.items, true) + "</code>";
|
|
79
|
-
},
|
|
80
|
-
},
|
|
81
|
-
LIST: {
|
|
82
|
-
name: "LIST",
|
|
83
|
-
mergeToBlock: false,
|
|
84
|
-
mergeFollowingNonTypedItemsWithSmallDistance: true,
|
|
85
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
86
|
-
return "<ul>\n" + linesToText(block.items, false) + "</ul>";
|
|
87
|
-
},
|
|
88
|
-
},
|
|
89
|
-
PARAGRAPH: {
|
|
90
|
-
name: "PARAGRAPH",
|
|
91
|
-
toText(block /*: LineItemBlock */) /*: string */ {
|
|
92
|
-
return "<p>" + linesToText(block.items, false) + "</p>";
|
|
93
|
-
},
|
|
94
|
-
},
|
|
95
|
-
};
|
|
96
|
-
|
|
97
|
-
function firstFormat(lineItem: { words: Array<{ format: any }> }): any {
|
|
98
|
-
if (lineItem.words.length === 0) {
|
|
99
|
-
return null;
|
|
100
|
-
}
|
|
101
|
-
return lineItem.words[0].format;
|
|
102
|
-
}
|
|
103
|
-
|
|
104
|
-
function isPunctationCharacter(string: string): boolean {
|
|
105
|
-
if (string.length !== 1) {
|
|
106
|
-
return false;
|
|
107
|
-
}
|
|
108
|
-
return string[0] === "." || string[0] === "!" || string[0] === "?";
|
|
109
|
-
}
|
|
110
|
-
|
|
111
|
-
function linesToText(lineItems: any[], disableInlineFormats: boolean): string {
|
|
112
|
-
var text = "";
|
|
113
|
-
var openFormat;
|
|
114
|
-
|
|
115
|
-
const closeFormat = () => {
|
|
116
|
-
text += openFormat.endSymbol;
|
|
117
|
-
openFormat = null;
|
|
118
|
-
};
|
|
119
|
-
|
|
120
|
-
lineItems.forEach((line, lineIndex) => {
|
|
121
|
-
if (!line) return;
|
|
122
|
-
|
|
123
|
-
line.words.forEach((word, i) => {
|
|
124
|
-
const wordType = word.type;
|
|
125
|
-
const wordFormat = word.format;
|
|
126
|
-
if (openFormat && (!wordFormat || wordFormat !== openFormat)) {
|
|
127
|
-
closeFormat();
|
|
128
|
-
}
|
|
129
|
-
|
|
130
|
-
if (
|
|
131
|
-
i > 0 &&
|
|
132
|
-
!(wordType && wordType.attachWithoutWhitespace) &&
|
|
133
|
-
!isPunctationCharacter(word.string)
|
|
134
|
-
) {
|
|
135
|
-
text += " ";
|
|
136
|
-
}
|
|
137
|
-
|
|
138
|
-
if (wordFormat && !openFormat && !disableInlineFormats) {
|
|
139
|
-
openFormat = wordFormat;
|
|
140
|
-
text += openFormat.startSymbol;
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
if (wordType && (!disableInlineFormats || wordType.plainTextFormat)) {
|
|
144
|
-
text += wordType.toText(word.string);
|
|
145
|
-
} else {
|
|
146
|
-
text += word.string;
|
|
147
|
-
}
|
|
148
|
-
});
|
|
149
|
-
if (
|
|
150
|
-
openFormat &&
|
|
151
|
-
(lineIndex === lineItems.length - 1 ||
|
|
152
|
-
firstFormat(lineItems[lineIndex + 1]) !== openFormat)
|
|
153
|
-
) {
|
|
154
|
-
closeFormat();
|
|
155
|
-
}
|
|
156
|
-
text += "\n";
|
|
157
|
-
});
|
|
158
|
-
return text;
|
|
159
|
-
}
|
|
160
|
-
|
|
161
|
-
// Make the object immutable
|
|
162
|
-
// Object.freeze(BlockTypes);
|
|
163
|
-
// Object.values(BlockTypes).forEach(value => Object.freeze(value));
|
|
164
|
-
|
|
165
|
-
BlockTypes.enumValueOf = function enumValueOf(name: string) {
|
|
166
|
-
return BlockTypes[name];
|
|
167
|
-
};
|
|
168
|
-
|
|
169
|
-
BlockTypes.isHeadline = function isHeadline(
|
|
170
|
-
type /*: typeof BlockTypes[keyof typeof BlockTypes] */,
|
|
171
|
-
) /*: boolean */ {
|
|
172
|
-
return type && type.name.length === 2 && type.name[0] === "H";
|
|
173
|
-
};
|
|
174
|
-
|
|
175
|
-
BlockTypes.blockToText = function blockToText(
|
|
176
|
-
block /*: LineItemBlock */,
|
|
177
|
-
) /*: string */ {
|
|
178
|
-
if (!block.type) {
|
|
179
|
-
return linesToText(block.items, false);
|
|
180
|
-
}
|
|
181
|
-
return block.type.toText(block);
|
|
182
|
-
};
|
|
183
|
-
|
|
184
|
-
BlockTypes.headlineByLevel = function headlineByLevel(level) {
|
|
185
|
-
if (level === 1) return BlockTypes.H1;
|
|
186
|
-
if (level === 2) return BlockTypes.H2;
|
|
187
|
-
if (level === 3) return BlockTypes.H3;
|
|
188
|
-
if (level === 4) return BlockTypes.H4;
|
|
189
|
-
if (level === 5) return BlockTypes.H5;
|
|
190
|
-
|
|
191
|
-
// if level is >= 6, just use BlockType H6
|
|
192
|
-
if (level > 6) {
|
|
193
|
-
// eslint-disable-next-line no-console
|
|
194
|
-
console.warn(
|
|
195
|
-
"Unsupported headline level: " +
|
|
196
|
-
level +
|
|
197
|
-
" (supported are 1-6), defaulting to level 6",
|
|
198
|
-
);
|
|
199
|
-
}
|
|
200
|
-
return BlockTypes.H6;
|
|
201
|
-
};
|
|
202
|
-
|
|
203
|
-
export default BlockTypes;
|
|
1
|
+
/**
|
|
2
|
+
* @description Enum-like registry of semantic document block types (H1–H6, TOC,
|
|
3
|
+
* FOOTNOTES, CODE, LIST, PARAGRAPH). Each entry carries a `toText()` serializer
|
|
4
|
+
* that converts a `LineItemBlock` to HTML. Helper methods `isHeadline`,
|
|
5
|
+
* `blockToText`, and `headlineByLevel` are mixed in on the registry object.
|
|
6
|
+
* The inner `linesToText` function handles inline bold/italic/link/footnote
|
|
7
|
+
* formatting by tracking open `WordFormat` spans across words and lines.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
const BlockTypes: Record<string, any> = {
|
|
11
|
+
H1: {
|
|
12
|
+
name: "H1",
|
|
13
|
+
headline: true,
|
|
14
|
+
headlineLevel: 1,
|
|
15
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
16
|
+
return "<h1>" + linesToText(block.items, true) + "</h1>";
|
|
17
|
+
},
|
|
18
|
+
},
|
|
19
|
+
H2: {
|
|
20
|
+
name: "H2",
|
|
21
|
+
headline: true,
|
|
22
|
+
headlineLevel: 2,
|
|
23
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
24
|
+
return "<h2>" + linesToText(block.items, true) + "</h2>";
|
|
25
|
+
},
|
|
26
|
+
},
|
|
27
|
+
H3: {
|
|
28
|
+
name: "H3",
|
|
29
|
+
headline: true,
|
|
30
|
+
headlineLevel: 3,
|
|
31
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
32
|
+
return "<h3>" + linesToText(block.items, true) + "</h3>";
|
|
33
|
+
},
|
|
34
|
+
},
|
|
35
|
+
H4: {
|
|
36
|
+
name: "H4",
|
|
37
|
+
headline: true,
|
|
38
|
+
headlineLevel: 4,
|
|
39
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
40
|
+
return "<h4>" + linesToText(block.items, true) + "</h4>";
|
|
41
|
+
},
|
|
42
|
+
},
|
|
43
|
+
H5: {
|
|
44
|
+
name: "H5",
|
|
45
|
+
headline: true,
|
|
46
|
+
headlineLevel: 5,
|
|
47
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
48
|
+
return "<h5>" + linesToText(block.items, true) + "</h5>";
|
|
49
|
+
},
|
|
50
|
+
},
|
|
51
|
+
H6: {
|
|
52
|
+
name: "H6",
|
|
53
|
+
headline: true,
|
|
54
|
+
headlineLevel: 6,
|
|
55
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
56
|
+
return "<h6>" + linesToText(block.items, true) + "</h6>";
|
|
57
|
+
},
|
|
58
|
+
},
|
|
59
|
+
TOC: {
|
|
60
|
+
name: "TOC",
|
|
61
|
+
mergeToBlock: true,
|
|
62
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
63
|
+
return linesToText(block.items, true);
|
|
64
|
+
},
|
|
65
|
+
},
|
|
66
|
+
FOOTNOTES: {
|
|
67
|
+
name: "FOOTNOTES",
|
|
68
|
+
mergeToBlock: true,
|
|
69
|
+
mergeFollowingNonTypedItems: true,
|
|
70
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
71
|
+
return "<p>" + linesToText(block.items, false) + "</p>";
|
|
72
|
+
},
|
|
73
|
+
},
|
|
74
|
+
CODE: {
|
|
75
|
+
name: "CODE",
|
|
76
|
+
mergeToBlock: true,
|
|
77
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
78
|
+
return "<code>" + linesToText(block.items, true) + "</code>";
|
|
79
|
+
},
|
|
80
|
+
},
|
|
81
|
+
LIST: {
|
|
82
|
+
name: "LIST",
|
|
83
|
+
mergeToBlock: false,
|
|
84
|
+
mergeFollowingNonTypedItemsWithSmallDistance: true,
|
|
85
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
86
|
+
return "<ul>\n" + linesToText(block.items, false) + "</ul>";
|
|
87
|
+
},
|
|
88
|
+
},
|
|
89
|
+
PARAGRAPH: {
|
|
90
|
+
name: "PARAGRAPH",
|
|
91
|
+
toText(block /*: LineItemBlock */) /*: string */ {
|
|
92
|
+
return "<p>" + linesToText(block.items, false) + "</p>";
|
|
93
|
+
},
|
|
94
|
+
},
|
|
95
|
+
};
|
|
96
|
+
|
|
97
|
+
function firstFormat(lineItem: { words: Array<{ format: any }> }): any {
|
|
98
|
+
if (lineItem.words.length === 0) {
|
|
99
|
+
return null;
|
|
100
|
+
}
|
|
101
|
+
return lineItem.words[0].format;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
function isPunctationCharacter(string: string): boolean {
|
|
105
|
+
if (string.length !== 1) {
|
|
106
|
+
return false;
|
|
107
|
+
}
|
|
108
|
+
return string[0] === "." || string[0] === "!" || string[0] === "?";
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
function linesToText(lineItems: any[], disableInlineFormats: boolean): string {
|
|
112
|
+
var text = "";
|
|
113
|
+
var openFormat;
|
|
114
|
+
|
|
115
|
+
const closeFormat = () => {
|
|
116
|
+
text += openFormat.endSymbol;
|
|
117
|
+
openFormat = null;
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
lineItems.forEach((line, lineIndex) => {
|
|
121
|
+
if (!line) return;
|
|
122
|
+
|
|
123
|
+
line.words.forEach((word, i) => {
|
|
124
|
+
const wordType = word.type;
|
|
125
|
+
const wordFormat = word.format;
|
|
126
|
+
if (openFormat && (!wordFormat || wordFormat !== openFormat)) {
|
|
127
|
+
closeFormat();
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
if (
|
|
131
|
+
i > 0 &&
|
|
132
|
+
!(wordType && wordType.attachWithoutWhitespace) &&
|
|
133
|
+
!isPunctationCharacter(word.string)
|
|
134
|
+
) {
|
|
135
|
+
text += " ";
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
if (wordFormat && !openFormat && !disableInlineFormats) {
|
|
139
|
+
openFormat = wordFormat;
|
|
140
|
+
text += openFormat.startSymbol;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
if (wordType && (!disableInlineFormats || wordType.plainTextFormat)) {
|
|
144
|
+
text += wordType.toText(word.string);
|
|
145
|
+
} else {
|
|
146
|
+
text += word.string;
|
|
147
|
+
}
|
|
148
|
+
});
|
|
149
|
+
if (
|
|
150
|
+
openFormat &&
|
|
151
|
+
(lineIndex === lineItems.length - 1 ||
|
|
152
|
+
firstFormat(lineItems[lineIndex + 1]) !== openFormat)
|
|
153
|
+
) {
|
|
154
|
+
closeFormat();
|
|
155
|
+
}
|
|
156
|
+
text += "\n";
|
|
157
|
+
});
|
|
158
|
+
return text;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
// Make the object immutable
|
|
162
|
+
// Object.freeze(BlockTypes);
|
|
163
|
+
// Object.values(BlockTypes).forEach(value => Object.freeze(value));
|
|
164
|
+
|
|
165
|
+
BlockTypes.enumValueOf = function enumValueOf(name: string) {
|
|
166
|
+
return BlockTypes[name];
|
|
167
|
+
};
|
|
168
|
+
|
|
169
|
+
BlockTypes.isHeadline = function isHeadline(
|
|
170
|
+
type /*: typeof BlockTypes[keyof typeof BlockTypes] */,
|
|
171
|
+
) /*: boolean */ {
|
|
172
|
+
return type && type.name.length === 2 && type.name[0] === "H";
|
|
173
|
+
};
|
|
174
|
+
|
|
175
|
+
BlockTypes.blockToText = function blockToText(
|
|
176
|
+
block /*: LineItemBlock */,
|
|
177
|
+
) /*: string */ {
|
|
178
|
+
if (!block.type) {
|
|
179
|
+
return linesToText(block.items, false);
|
|
180
|
+
}
|
|
181
|
+
return block.type.toText(block);
|
|
182
|
+
};
|
|
183
|
+
|
|
184
|
+
BlockTypes.headlineByLevel = function headlineByLevel(level) {
|
|
185
|
+
if (level === 1) return BlockTypes.H1;
|
|
186
|
+
if (level === 2) return BlockTypes.H2;
|
|
187
|
+
if (level === 3) return BlockTypes.H3;
|
|
188
|
+
if (level === 4) return BlockTypes.H4;
|
|
189
|
+
if (level === 5) return BlockTypes.H5;
|
|
190
|
+
|
|
191
|
+
// if level is >= 6, just use BlockType H6
|
|
192
|
+
if (level > 6) {
|
|
193
|
+
// eslint-disable-next-line no-console
|
|
194
|
+
console.warn(
|
|
195
|
+
"Unsupported headline level: " +
|
|
196
|
+
level +
|
|
197
|
+
" (supported are 1-6), defaulting to level 6",
|
|
198
|
+
);
|
|
199
|
+
}
|
|
200
|
+
return BlockTypes.H6;
|
|
201
|
+
};
|
|
202
|
+
|
|
203
|
+
export default BlockTypes;
|