@bendyline/squisq-formats 2.1.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/NOTICE.md +20 -0
- package/README.md +1 -1
- package/dist/{chunk-NNHKUXKA.js → chunk-26ISNJ7Y.js} +85 -65
- package/dist/{chunk-NKAJPJ4G.js → chunk-2JJ5RFDZ.js} +0 -1
- package/dist/{chunk-WQSHGBLN.js → chunk-3NKXBZSR.js} +193 -42
- package/dist/{chunk-KURGXM4I.js → chunk-4V3KCHAP.js} +3 -4
- package/dist/{chunk-MLX2BOJC.js → chunk-6RQOV3B3.js} +1 -2
- package/dist/{chunk-EW54IRRS.js → chunk-6S6GU3ZG.js} +5 -6
- package/dist/{chunk-FE6OJV6O.js → chunk-7AWFHP5U.js} +1 -1
- package/dist/{chunk-RFAPOKHJ.js → chunk-AD2WT564.js} +59 -9
- package/dist/{chunk-O3GVVND4.js → chunk-AONELFLA.js} +0 -1
- package/dist/{chunk-SC67HYQJ.js → chunk-EJTNGKEA.js} +5 -8
- package/dist/chunk-GX7RAUME.js +121 -0
- package/dist/{chunk-SSUPBUF5.js → chunk-IIQYS2YH.js} +0 -1
- package/dist/{chunk-RLU7UFYU.js → chunk-IPN56VLW.js} +83 -58
- package/dist/{chunk-DTDF6QDP.js → chunk-JE6LSIHE.js} +81 -20
- package/dist/{chunk-U4MRIFKL.js → chunk-JU2RHXUB.js} +0 -1
- package/dist/{chunk-4VUWTSGM.js → chunk-K6XRMVPW.js} +64 -31
- package/dist/{chunk-ODL3SSPT.js → chunk-KXOZMWBS.js} +0 -1
- package/dist/chunk-OGS5VCGJ.js +446 -0
- package/dist/{chunk-GVS2XXV6.js → chunk-PJXJI2LY.js} +449 -57
- package/dist/{chunk-74GO3FVS.js → chunk-PU7REGWV.js} +5 -8
- package/dist/{chunk-PN52A5AA.js → chunk-SBUW7NHR.js} +0 -1
- package/dist/{chunk-QFLDYKCR.js → chunk-TAAENIRB.js} +5 -8
- package/dist/{chunk-7ARKUCQT.js → chunk-X2DEAXNK.js} +54 -2
- package/dist/container/index.js +1 -2
- package/dist/csv/index.d.ts +27 -2
- package/dist/csv/index.js +1 -2
- package/dist/docx/index.d.ts +5 -1
- package/dist/docx/index.js +9 -11
- package/dist/epub/index.d.ts +2 -0
- package/dist/epub/index.js +5 -6
- package/dist/{export-D2NkylDT.d.ts → export-D9msROJS.d.ts} +18 -6
- package/dist/extract-MN7LA3NL.js +13 -0
- package/dist/html/index.d.ts +11 -4
- package/dist/html/index.js +3 -4
- package/dist/images-ESPQKVTW.js +6 -0
- package/dist/{import-K8mfc0fz.d.ts → import-C3htUTss.d.ts} +5 -1
- package/dist/{import-DTkDxHmZ.d.ts → import-C8whCC7_.d.ts} +6 -0
- package/dist/index.d.ts +7 -7
- package/dist/index.js +28 -26
- package/dist/infer/index.d.ts +3 -3
- package/dist/infer/index.js +7 -9
- package/dist/{layouts-BHrgZ5FS.d.ts → layouts-CTdPlB-u.d.ts} +1 -1
- package/dist/layouts-DRWZGSPD.js +10 -0
- package/dist/{mapTheme-IR27S6IV.js → mapTheme-4TWH25FT.js} +1 -2
- package/dist/ooxml/index.d.ts +3 -3
- package/dist/ooxml/index.js +14 -13
- package/dist/pdf/index.d.ts +18 -0
- package/dist/pdf/index.js +2 -3
- package/dist/pptx/index.d.ts +4 -4
- package/dist/pptx/index.js +11 -13
- package/dist/{reader-B9L8Ucbj.d.ts → reader-B_m1aKZC.d.ts} +30 -1
- package/dist/registry/index.d.ts +21 -5
- package/dist/registry/index.js +9 -6
- package/dist/{themeReader-DJKErl_j.d.ts → themeReader-DCtwC83Q.d.ts} +1 -1
- package/dist/xlsx/index.d.ts +3 -3
- package/dist/xlsx/index.js +6 -7
- package/package.json +6 -3
- package/dist/chunk-4VUWTSGM.js.map +0 -1
- package/dist/chunk-6M7Z25LA.js +0 -46
- package/dist/chunk-6M7Z25LA.js.map +0 -1
- package/dist/chunk-74GO3FVS.js.map +0 -1
- package/dist/chunk-7ARKUCQT.js.map +0 -1
- package/dist/chunk-DTDF6QDP.js.map +0 -1
- package/dist/chunk-EW54IRRS.js.map +0 -1
- package/dist/chunk-FE6OJV6O.js.map +0 -1
- package/dist/chunk-GVS2XXV6.js.map +0 -1
- package/dist/chunk-KURGXM4I.js.map +0 -1
- package/dist/chunk-MLX2BOJC.js.map +0 -1
- package/dist/chunk-NKAJPJ4G.js.map +0 -1
- package/dist/chunk-NNHKUXKA.js.map +0 -1
- package/dist/chunk-O3GVVND4.js.map +0 -1
- package/dist/chunk-ODL3SSPT.js.map +0 -1
- package/dist/chunk-PN52A5AA.js.map +0 -1
- package/dist/chunk-QFLDYKCR.js.map +0 -1
- package/dist/chunk-RFAPOKHJ.js.map +0 -1
- package/dist/chunk-RLU7UFYU.js.map +0 -1
- package/dist/chunk-SC67HYQJ.js.map +0 -1
- package/dist/chunk-SSUPBUF5.js.map +0 -1
- package/dist/chunk-U4MRIFKL.js.map +0 -1
- package/dist/chunk-UGYF5AZE.js +0 -275
- package/dist/chunk-UGYF5AZE.js.map +0 -1
- package/dist/chunk-WQSHGBLN.js.map +0 -1
- package/dist/chunk-YRT7GQ5Y.js +0 -28
- package/dist/chunk-YRT7GQ5Y.js.map +0 -1
- package/dist/container/index.js.map +0 -1
- package/dist/csv/index.js.map +0 -1
- package/dist/docx/index.js.map +0 -1
- package/dist/epub/index.js.map +0 -1
- package/dist/extract-OJ7ZQV6P.js +0 -15
- package/dist/extract-OJ7ZQV6P.js.map +0 -1
- package/dist/html/index.js.map +0 -1
- package/dist/images-7FBWPKE3.js +0 -7
- package/dist/images-7FBWPKE3.js.map +0 -1
- package/dist/index.js.map +0 -1
- package/dist/infer/index.js.map +0 -1
- package/dist/layouts-5VDIRPIJ.js +0 -12
- package/dist/layouts-5VDIRPIJ.js.map +0 -1
- package/dist/mapTheme-IR27S6IV.js.map +0 -1
- package/dist/ooxml/index.js.map +0 -1
- package/dist/pdf/index.js.map +0 -1
- package/dist/pptx/index.js.map +0 -1
- package/dist/registry/index.js.map +0 -1
- package/dist/xlsx/index.js.map +0 -1
- package/src/__tests__/container.test.ts +0 -230
- package/src/__tests__/convert.test.ts +0 -495
- package/src/__tests__/csvImport.test.ts +0 -84
- package/src/__tests__/docxExport.test.ts +0 -491
- package/src/__tests__/docxImport.test.ts +0 -531
- package/src/__tests__/epub.test.ts +0 -649
- package/src/__tests__/exportThemeReconciliation.test.ts +0 -87
- package/src/__tests__/formatRegistry.test.ts +0 -174
- package/src/__tests__/html.test.ts +0 -439
- package/src/__tests__/htmlImport.test.ts +0 -57
- package/src/__tests__/inferTheme.test.ts +0 -135
- package/src/__tests__/lossyWarnings.test.ts +0 -146
- package/src/__tests__/ooxml.test.ts +0 -271
- package/src/__tests__/ooxmlCancellation.test.ts +0 -113
- package/src/__tests__/ooxmlThemeReader.test.ts +0 -92
- package/src/__tests__/pdfExport.test.ts +0 -322
- package/src/__tests__/pdfImport.test.ts +0 -384
- package/src/__tests__/plainHtml.test.ts +0 -417
- package/src/__tests__/plainHtmlBundle.test.ts +0 -253
- package/src/__tests__/pptxExport.test.ts +0 -138
- package/src/__tests__/pptxImport.test.ts +0 -145
- package/src/__tests__/pptxInferFixtures.ts +0 -314
- package/src/__tests__/pptxLayoutInfer.test.ts +0 -395
- package/src/__tests__/roundTrip.test.ts +0 -201
- package/src/__tests__/roundTripAssets.test.ts +0 -50
- package/src/__tests__/roundTripMatrix.fixtures.ts +0 -86
- package/src/__tests__/roundTripMatrix.helpers.ts +0 -154
- package/src/__tests__/roundTripMatrix.test.ts +0 -142
- package/src/__tests__/sharedContainer.test.ts +0 -41
- package/src/__tests__/sharedImages.test.ts +0 -61
- package/src/__tests__/xlsxExport.test.ts +0 -164
- package/src/__tests__/xlsxImport.test.ts +0 -80
- package/src/__tests__/zipSafety.test.ts +0 -317
- package/src/container/index.ts +0 -94
- package/src/csv/index.ts +0 -188
- package/src/docx/export.ts +0 -1375
- package/src/docx/import.ts +0 -1250
- package/src/docx/index.ts +0 -26
- package/src/docx/styles.ts +0 -145
- package/src/epub/export.ts +0 -968
- package/src/epub/index.ts +0 -20
- package/src/html/docsHtmlBundle.ts +0 -373
- package/src/html/htmlTemplate.ts +0 -385
- package/src/html/imageUtils.ts +0 -61
- package/src/html/import.ts +0 -297
- package/src/html/index.ts +0 -212
- package/src/html/plainHtml.ts +0 -790
- package/src/html/plainHtmlBundle.ts +0 -421
- package/src/index.ts +0 -109
- package/src/infer/extract.ts +0 -127
- package/src/infer/index.ts +0 -199
- package/src/infer/mapTheme.ts +0 -176
- package/src/infer/types.ts +0 -27
- package/src/ooxml/index.ts +0 -111
- package/src/ooxml/namespaces.ts +0 -217
- package/src/ooxml/readUtils.ts +0 -44
- package/src/ooxml/reader.ts +0 -318
- package/src/ooxml/themeReader.ts +0 -197
- package/src/ooxml/types.ts +0 -103
- package/src/ooxml/writer.ts +0 -339
- package/src/ooxml/xmlUtils.ts +0 -123
- package/src/pdf/export.ts +0 -1084
- package/src/pdf/import.ts +0 -1164
- package/src/pdf/index.ts +0 -29
- package/src/pdf/styles.ts +0 -180
- package/src/pptx/export.ts +0 -1184
- package/src/pptx/import.ts +0 -455
- package/src/pptx/index.ts +0 -52
- package/src/pptx/layouts.ts +0 -1222
- package/src/pptx/styles.ts +0 -96
- package/src/pptx/templates.ts +0 -187
- package/src/registry/convert.ts +0 -433
- package/src/registry/defaultFormats.ts +0 -413
- package/src/registry/errors.ts +0 -46
- package/src/registry/index.ts +0 -43
- package/src/registry/registry.ts +0 -48
- package/src/registry/types.ts +0 -170
- package/src/shared/boundedZipArchive.ts +0 -383
- package/src/shared/container.ts +0 -28
- package/src/shared/fidelity.ts +0 -130
- package/src/shared/images.ts +0 -44
- package/src/shared/inlineRuns.ts +0 -99
- package/src/shared/text.ts +0 -41
- package/src/shared/zipEntryCount.ts +0 -151
- package/src/shared/zipLimits.ts +0 -296
- package/src/shared/zipSafety.ts +0 -19
- package/src/xlsx/export.ts +0 -253
- package/src/xlsx/import.ts +0 -160
- package/src/xlsx/index.ts +0 -35
|
@@ -1,384 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Tests for PDF import: pdfToMarkdownDoc, pdfToDoc.
|
|
3
|
-
*
|
|
4
|
-
* Builds PDFs with pdf-lib (adding known text), then imports them
|
|
5
|
-
* via pdfToMarkdownDoc and inspects the resulting MarkdownDocument.
|
|
6
|
-
*
|
|
7
|
-
* NOTE: pdfjs-dist text extraction quality depends on the environment.
|
|
8
|
-
* Standard fonts embedded by pdf-lib should produce readable text items.
|
|
9
|
-
*/
|
|
10
|
-
|
|
11
|
-
import { describe, it, expect } from 'vitest';
|
|
12
|
-
import { PDFDocument, StandardFonts, rgb } from 'pdf-lib';
|
|
13
|
-
|
|
14
|
-
import { pdfToMarkdownDoc, pdfToDoc, insertImageBlocks } from '../pdf/import';
|
|
15
|
-
import type { ExtractedImage } from '../pdf/import';
|
|
16
|
-
import type {
|
|
17
|
-
MarkdownHeading,
|
|
18
|
-
MarkdownBlockNode,
|
|
19
|
-
MarkdownCodeBlock,
|
|
20
|
-
MarkdownParagraph,
|
|
21
|
-
} from '@bendyline/squisq/markdown';
|
|
22
|
-
|
|
23
|
-
/** Structural shape shared by all markdown node types, for recursive walkers. */
|
|
24
|
-
type TreeNode = { type: string; value?: string; children?: TreeNode[] };
|
|
25
|
-
|
|
26
|
-
// ============================================
|
|
27
|
-
// Helpers
|
|
28
|
-
// ============================================
|
|
29
|
-
|
|
30
|
-
/**
|
|
31
|
-
* Create a simple PDF buffer with text drawn at specified sizes/positions.
|
|
32
|
-
*/
|
|
33
|
-
async function buildSimplePdf(
|
|
34
|
-
lines: Array<{ text: string; x: number; y: number; fontSize: number; fontName?: string }>,
|
|
35
|
-
): Promise<Uint8Array> {
|
|
36
|
-
const doc = await PDFDocument.create();
|
|
37
|
-
const page = doc.addPage([612, 792]); // Letter
|
|
38
|
-
|
|
39
|
-
const regular = await doc.embedFont(StandardFonts.Helvetica);
|
|
40
|
-
const bold = await doc.embedFont(StandardFonts.HelveticaBold);
|
|
41
|
-
const italic = await doc.embedFont(StandardFonts.HelveticaOblique);
|
|
42
|
-
const mono = await doc.embedFont(StandardFonts.Courier);
|
|
43
|
-
|
|
44
|
-
for (const line of lines) {
|
|
45
|
-
let font = regular;
|
|
46
|
-
const fName = line.fontName?.toLowerCase() ?? '';
|
|
47
|
-
if (fName.includes('bold')) font = bold;
|
|
48
|
-
else if (fName.includes('italic') || fName.includes('oblique')) font = italic;
|
|
49
|
-
else if (fName.includes('courier') || fName.includes('mono')) font = mono;
|
|
50
|
-
|
|
51
|
-
page.drawText(line.text, {
|
|
52
|
-
x: line.x,
|
|
53
|
-
y: line.y,
|
|
54
|
-
size: line.fontSize,
|
|
55
|
-
font,
|
|
56
|
-
color: rgb(0, 0, 0),
|
|
57
|
-
});
|
|
58
|
-
}
|
|
59
|
-
|
|
60
|
-
return doc.save();
|
|
61
|
-
}
|
|
62
|
-
|
|
63
|
-
function flatText(node: MarkdownBlockNode): string {
|
|
64
|
-
const parts: string[] = [];
|
|
65
|
-
function walk(n: TreeNode) {
|
|
66
|
-
if (n.type === 'text' && n.value) parts.push(n.value);
|
|
67
|
-
if (n.value && n.type === 'code') parts.push(n.value);
|
|
68
|
-
if (n.value && n.type === 'inlineCode') parts.push(n.value);
|
|
69
|
-
if (n.children) for (const c of n.children) walk(c);
|
|
70
|
-
}
|
|
71
|
-
walk(node);
|
|
72
|
-
return parts.join('');
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
// ============================================
|
|
76
|
-
// Basic Import
|
|
77
|
-
// ============================================
|
|
78
|
-
|
|
79
|
-
describe('pdfToMarkdownDoc', () => {
|
|
80
|
-
it('returns an empty document for a blank PDF', async () => {
|
|
81
|
-
const doc = await PDFDocument.create();
|
|
82
|
-
doc.addPage([612, 792]);
|
|
83
|
-
const buffer = await doc.save();
|
|
84
|
-
|
|
85
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
86
|
-
expect(md.type).toBe('document');
|
|
87
|
-
expect(md.children.length).toBe(0);
|
|
88
|
-
});
|
|
89
|
-
|
|
90
|
-
it('imports a single paragraph', async () => {
|
|
91
|
-
const buffer = await buildSimplePdf([{ text: 'Hello World', x: 72, y: 700, fontSize: 11 }]);
|
|
92
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
93
|
-
expect(md.children.length).toBeGreaterThanOrEqual(1);
|
|
94
|
-
|
|
95
|
-
// The first block should contain "Hello World"
|
|
96
|
-
const allText = md.children.map(flatText).join(' ');
|
|
97
|
-
expect(allText).toContain('Hello');
|
|
98
|
-
expect(allText).toContain('World');
|
|
99
|
-
});
|
|
100
|
-
|
|
101
|
-
it('detects a heading by font size', async () => {
|
|
102
|
-
const buffer = await buildSimplePdf([
|
|
103
|
-
{ text: 'Big Title', x: 72, y: 700, fontSize: 24 },
|
|
104
|
-
{ text: 'Body text here.', x: 72, y: 660, fontSize: 11 },
|
|
105
|
-
]);
|
|
106
|
-
// Provide explicit bodyFontSize — with only 2 lines auto-detection
|
|
107
|
-
// picks the first seen size (24) which masks the heading.
|
|
108
|
-
const md = await pdfToMarkdownDoc(buffer, { bodyFontSize: 11 });
|
|
109
|
-
|
|
110
|
-
// Find heading block
|
|
111
|
-
const headings = md.children.filter((c) => c.type === 'heading');
|
|
112
|
-
expect(headings.length).toBeGreaterThanOrEqual(1);
|
|
113
|
-
if (headings.length > 0) {
|
|
114
|
-
const h = headings[0] as MarkdownHeading;
|
|
115
|
-
expect(h.depth).toBeLessThanOrEqual(2);
|
|
116
|
-
expect(flatText(h)).toContain('Title');
|
|
117
|
-
}
|
|
118
|
-
});
|
|
119
|
-
|
|
120
|
-
it('renders bold text as plain text (standard font limitation)', async () => {
|
|
121
|
-
// pdfjs-dist returns opaque font IDs ("g_d0_f2") for standard PDF fonts,
|
|
122
|
-
// not the actual font name. Bold/italic detection only works with
|
|
123
|
-
// embedded fonts whose names contain "Bold"/"Italic" — a known heuristic
|
|
124
|
-
// limitation. Here we verify the text is still extracted.
|
|
125
|
-
const buffer = await buildSimplePdf([
|
|
126
|
-
{ text: 'Bold text', x: 72, y: 700, fontSize: 11, fontName: 'Helvetica-Bold' },
|
|
127
|
-
]);
|
|
128
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
129
|
-
expect(md.children.length).toBeGreaterThanOrEqual(1);
|
|
130
|
-
const allText = md.children.map(flatText).join(' ');
|
|
131
|
-
expect(allText).toContain('Bold text');
|
|
132
|
-
});
|
|
133
|
-
|
|
134
|
-
it('renders italic text as plain text (standard font limitation)', async () => {
|
|
135
|
-
// Same as bold — standard PDF font names are not exposed by pdfjs-dist.
|
|
136
|
-
const buffer = await buildSimplePdf([
|
|
137
|
-
{ text: 'Italic text', x: 72, y: 700, fontSize: 11, fontName: 'Helvetica-Oblique' },
|
|
138
|
-
]);
|
|
139
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
140
|
-
expect(md.children.length).toBeGreaterThanOrEqual(1);
|
|
141
|
-
const allText = md.children.map(flatText).join(' ');
|
|
142
|
-
expect(allText).toContain('Italic text');
|
|
143
|
-
});
|
|
144
|
-
|
|
145
|
-
it('detects monospace text as inline code', async () => {
|
|
146
|
-
const buffer = await buildSimplePdf([
|
|
147
|
-
{ text: 'const x = 1;', x: 72, y: 700, fontSize: 10, fontName: 'Courier' },
|
|
148
|
-
]);
|
|
149
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
150
|
-
expect(md.children.length).toBeGreaterThanOrEqual(1);
|
|
151
|
-
|
|
152
|
-
// Monospace lines should become code blocks or inline code
|
|
153
|
-
function hasCode(node: TreeNode): boolean {
|
|
154
|
-
if (node.type === 'code' || node.type === 'inlineCode') return true;
|
|
155
|
-
if (node.children) return node.children.some(hasCode);
|
|
156
|
-
return false;
|
|
157
|
-
}
|
|
158
|
-
expect(md.children.some(hasCode)).toBe(true);
|
|
159
|
-
});
|
|
160
|
-
|
|
161
|
-
it('detects consecutive monospace lines as a code block', async () => {
|
|
162
|
-
const buffer = await buildSimplePdf([
|
|
163
|
-
{ text: 'function hello() {', x: 72, y: 700, fontSize: 10, fontName: 'Courier' },
|
|
164
|
-
{ text: ' return 42;', x: 72, y: 686, fontSize: 10, fontName: 'Courier' },
|
|
165
|
-
{ text: '}', x: 72, y: 672, fontSize: 10, fontName: 'Courier' },
|
|
166
|
-
]);
|
|
167
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
168
|
-
|
|
169
|
-
const codeBlocks = md.children.filter((c) => c.type === 'code');
|
|
170
|
-
expect(codeBlocks.length).toBeGreaterThanOrEqual(1);
|
|
171
|
-
if (codeBlocks.length > 0) {
|
|
172
|
-
const cb = codeBlocks[0] as MarkdownCodeBlock;
|
|
173
|
-
expect(cb.value).toContain('function');
|
|
174
|
-
expect(cb.value).toContain('return');
|
|
175
|
-
}
|
|
176
|
-
});
|
|
177
|
-
|
|
178
|
-
it('detects bullet list items', async () => {
|
|
179
|
-
const buffer = await buildSimplePdf([
|
|
180
|
-
{ text: '\u2022 First item', x: 72, y: 700, fontSize: 11 },
|
|
181
|
-
{ text: '\u2022 Second item', x: 72, y: 684, fontSize: 11 },
|
|
182
|
-
]);
|
|
183
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
184
|
-
|
|
185
|
-
const lists = md.children.filter((c) => c.type === 'list');
|
|
186
|
-
expect(lists.length).toBeGreaterThanOrEqual(1);
|
|
187
|
-
});
|
|
188
|
-
|
|
189
|
-
it('detects ordered list items', async () => {
|
|
190
|
-
const buffer = await buildSimplePdf([
|
|
191
|
-
{ text: '1. First step', x: 72, y: 700, fontSize: 11 },
|
|
192
|
-
{ text: '2. Second step', x: 72, y: 684, fontSize: 11 },
|
|
193
|
-
]);
|
|
194
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
195
|
-
|
|
196
|
-
const lists = md.children.filter((c) => c.type === 'list');
|
|
197
|
-
expect(lists.length).toBeGreaterThanOrEqual(1);
|
|
198
|
-
});
|
|
199
|
-
|
|
200
|
-
it('detects indented text as blockquote', async () => {
|
|
201
|
-
const buffer = await buildSimplePdf([
|
|
202
|
-
{ text: 'Normal paragraph.', x: 72, y: 720, fontSize: 11 },
|
|
203
|
-
{ text: 'This is a quote.', x: 110, y: 700, fontSize: 11 },
|
|
204
|
-
]);
|
|
205
|
-
const md = await pdfToMarkdownDoc(buffer, { detectBlockquotes: true });
|
|
206
|
-
|
|
207
|
-
const quotes = md.children.filter((c) => c.type === 'blockquote');
|
|
208
|
-
expect(quotes.length).toBeGreaterThanOrEqual(1);
|
|
209
|
-
});
|
|
210
|
-
|
|
211
|
-
it('detects URLs in text as links', async () => {
|
|
212
|
-
const buffer = await buildSimplePdf([
|
|
213
|
-
{ text: 'Visit https://example.com for more.', x: 72, y: 700, fontSize: 11 },
|
|
214
|
-
]);
|
|
215
|
-
const md = await pdfToMarkdownDoc(buffer, { detectLinks: true });
|
|
216
|
-
|
|
217
|
-
function hasLink(node: TreeNode): boolean {
|
|
218
|
-
if (node.type === 'link') return true;
|
|
219
|
-
if (node.children) return node.children.some(hasLink);
|
|
220
|
-
return false;
|
|
221
|
-
}
|
|
222
|
-
expect(md.children.some(hasLink)).toBe(true);
|
|
223
|
-
});
|
|
224
|
-
|
|
225
|
-
it('handles multi-page documents', async () => {
|
|
226
|
-
const doc = await PDFDocument.create();
|
|
227
|
-
const font = await doc.embedFont(StandardFonts.Helvetica);
|
|
228
|
-
|
|
229
|
-
for (let p = 0; p < 3; p++) {
|
|
230
|
-
const page = doc.addPage([612, 792]);
|
|
231
|
-
page.drawText(`Page ${p + 1} content`, {
|
|
232
|
-
x: 72,
|
|
233
|
-
y: 700,
|
|
234
|
-
size: 11,
|
|
235
|
-
font,
|
|
236
|
-
});
|
|
237
|
-
}
|
|
238
|
-
const buffer = await doc.save();
|
|
239
|
-
|
|
240
|
-
const md = await pdfToMarkdownDoc(buffer);
|
|
241
|
-
expect(md.children.length).toBeGreaterThanOrEqual(3);
|
|
242
|
-
});
|
|
243
|
-
|
|
244
|
-
it('respects bodyFontSize option', async () => {
|
|
245
|
-
// If we say body is 14pt, then a 16pt line should still be heading
|
|
246
|
-
// but borderline 13pt would not be
|
|
247
|
-
const buffer = await buildSimplePdf([
|
|
248
|
-
{ text: 'Medium heading', x: 72, y: 700, fontSize: 16 },
|
|
249
|
-
{ text: 'Body text at 14pt', x: 72, y: 670, fontSize: 14 },
|
|
250
|
-
]);
|
|
251
|
-
const md = await pdfToMarkdownDoc(buffer, { bodyFontSize: 14 });
|
|
252
|
-
|
|
253
|
-
const headings = md.children.filter((c) => c.type === 'heading');
|
|
254
|
-
expect(headings.length).toBeGreaterThanOrEqual(1);
|
|
255
|
-
});
|
|
256
|
-
|
|
257
|
-
it('accepts Uint8Array input', async () => {
|
|
258
|
-
const buffer = await buildSimplePdf([{ text: 'Uint8Array test', x: 72, y: 700, fontSize: 11 }]);
|
|
259
|
-
const uint8 = new Uint8Array(buffer);
|
|
260
|
-
const md = await pdfToMarkdownDoc(uint8);
|
|
261
|
-
expect(md.children.length).toBeGreaterThanOrEqual(1);
|
|
262
|
-
});
|
|
263
|
-
|
|
264
|
-
it('can disable table detection', async () => {
|
|
265
|
-
const buffer = await buildSimplePdf([
|
|
266
|
-
{ text: 'Col A', x: 72, y: 700, fontSize: 11 },
|
|
267
|
-
{ text: 'Col B', x: 250, y: 700, fontSize: 11 },
|
|
268
|
-
{ text: 'Val 1', x: 72, y: 684, fontSize: 11 },
|
|
269
|
-
{ text: 'Val 2', x: 250, y: 684, fontSize: 11 },
|
|
270
|
-
]);
|
|
271
|
-
const md = await pdfToMarkdownDoc(buffer, { detectTables: false });
|
|
272
|
-
const tables = md.children.filter((c) => c.type === 'table');
|
|
273
|
-
expect(tables.length).toBe(0);
|
|
274
|
-
});
|
|
275
|
-
});
|
|
276
|
-
|
|
277
|
-
// ============================================
|
|
278
|
-
// pdfToDoc convenience wrapper
|
|
279
|
-
// ============================================
|
|
280
|
-
|
|
281
|
-
// ============================================
|
|
282
|
-
// insertImageBlocks — page-level image placement (pure)
|
|
283
|
-
// ============================================
|
|
284
|
-
|
|
285
|
-
describe('insertImageBlocks', () => {
|
|
286
|
-
/** Build a synthetic paragraph block carrying a marker text value. */
|
|
287
|
-
function para(label: string): MarkdownParagraph {
|
|
288
|
-
return { type: 'paragraph', children: [{ type: 'text', value: label }] };
|
|
289
|
-
}
|
|
290
|
-
|
|
291
|
-
/** Build a synthetic extracted image on a given page (data/y are irrelevant here). */
|
|
292
|
-
function img(index: number, page: number): ExtractedImage {
|
|
293
|
-
return { path: `images/image${index}.png`, data: new ArrayBuffer(0), page, y: 0 };
|
|
294
|
-
}
|
|
295
|
-
|
|
296
|
-
/** Extract a flat list of block descriptors: paragraph label or image url. */
|
|
297
|
-
function describeBlocks(blocks: MarkdownBlockNode[]): string[] {
|
|
298
|
-
return blocks.map((b) => {
|
|
299
|
-
if (b.type === 'paragraph') {
|
|
300
|
-
const child = (b as MarkdownParagraph).children[0];
|
|
301
|
-
if (child && child.type === 'image') return `img:${child.url}`;
|
|
302
|
-
if (child && child.type === 'text') return `p:${child.value}`;
|
|
303
|
-
}
|
|
304
|
-
return b.type;
|
|
305
|
-
});
|
|
306
|
-
}
|
|
307
|
-
|
|
308
|
-
it('is a no-op when there are no images', () => {
|
|
309
|
-
const blocks = [para('A'), para('B')];
|
|
310
|
-
const pages = [0, 0];
|
|
311
|
-
const result = insertImageBlocks(blocks, pages, []);
|
|
312
|
-
expect(result).toBe(blocks);
|
|
313
|
-
});
|
|
314
|
-
|
|
315
|
-
it('places an image after the last block of its page', () => {
|
|
316
|
-
// Page 0: A, B Page 1: C
|
|
317
|
-
const blocks = [para('A'), para('B'), para('C')];
|
|
318
|
-
const pages = [0, 0, 1];
|
|
319
|
-
const result = insertImageBlocks(blocks, pages, [img(1, 0)]);
|
|
320
|
-
expect(describeBlocks(result)).toEqual(['p:A', 'p:B', 'img:images/image1.png', 'p:C']);
|
|
321
|
-
});
|
|
322
|
-
|
|
323
|
-
it('places images on the correct page in a multi-page document', () => {
|
|
324
|
-
// Page 0: A Page 1: B Page 2: C
|
|
325
|
-
const blocks = [para('A'), para('B'), para('C')];
|
|
326
|
-
const pages = [0, 1, 2];
|
|
327
|
-
const result = insertImageBlocks(blocks, pages, [img(1, 2), img(2, 0)]);
|
|
328
|
-
// image2 → after A (page 0), image1 → after C (page 2)
|
|
329
|
-
expect(describeBlocks(result)).toEqual([
|
|
330
|
-
'p:A',
|
|
331
|
-
'img:images/image2.png',
|
|
332
|
-
'p:B',
|
|
333
|
-
'p:C',
|
|
334
|
-
'img:images/image1.png',
|
|
335
|
-
]);
|
|
336
|
-
});
|
|
337
|
-
|
|
338
|
-
it('preserves image order within the same page', () => {
|
|
339
|
-
const blocks = [para('A'), para('B')];
|
|
340
|
-
const pages = [0, 0];
|
|
341
|
-
const result = insertImageBlocks(blocks, pages, [img(1, 0), img(2, 0)]);
|
|
342
|
-
expect(describeBlocks(result)).toEqual([
|
|
343
|
-
'p:A',
|
|
344
|
-
'p:B',
|
|
345
|
-
'img:images/image1.png',
|
|
346
|
-
'img:images/image2.png',
|
|
347
|
-
]);
|
|
348
|
-
});
|
|
349
|
-
|
|
350
|
-
it('falls back to the nearest preceding page for an image-only page', () => {
|
|
351
|
-
// Page 0: A Page 1: (no blocks) image on page 1 → after A
|
|
352
|
-
const blocks = [para('A'), para('B')];
|
|
353
|
-
const pages = [0, 2];
|
|
354
|
-
// Image on page 1 has no blocks; nearest preceding page with blocks is 0 → after A.
|
|
355
|
-
const result = insertImageBlocks(blocks, pages, [img(1, 1)]);
|
|
356
|
-
expect(describeBlocks(result)).toEqual(['p:A', 'img:images/image1.png', 'p:B']);
|
|
357
|
-
});
|
|
358
|
-
|
|
359
|
-
it('appends at the end when no preceding page has blocks', () => {
|
|
360
|
-
// Only page 5 has blocks; image on page 2 has no preceding page → append at end.
|
|
361
|
-
const blocks = [para('A'), para('B')];
|
|
362
|
-
const pages = [5, 5];
|
|
363
|
-
const result = insertImageBlocks(blocks, pages, [img(1, 2)]);
|
|
364
|
-
expect(describeBlocks(result)).toEqual(['p:A', 'p:B', 'img:images/image1.png']);
|
|
365
|
-
});
|
|
366
|
-
|
|
367
|
-
it('appends all images when there are no text blocks', () => {
|
|
368
|
-
const result = insertImageBlocks([], [], [img(1, 0), img(2, 3)]);
|
|
369
|
-
expect(describeBlocks(result)).toEqual(['img:images/image1.png', 'img:images/image2.png']);
|
|
370
|
-
});
|
|
371
|
-
});
|
|
372
|
-
|
|
373
|
-
describe('pdfToDoc', () => {
|
|
374
|
-
it('converts PDF to a Doc object', async () => {
|
|
375
|
-
const buffer = await buildSimplePdf([
|
|
376
|
-
{ text: 'Test Title', x: 72, y: 700, fontSize: 24 },
|
|
377
|
-
{ text: 'Some body text.', x: 72, y: 660, fontSize: 11 },
|
|
378
|
-
]);
|
|
379
|
-
const doc = await pdfToDoc(buffer);
|
|
380
|
-
expect(doc).toBeDefined();
|
|
381
|
-
expect(doc.blocks).toBeDefined();
|
|
382
|
-
expect(doc.blocks.length).toBeGreaterThan(0);
|
|
383
|
-
});
|
|
384
|
-
});
|