@bendyline/squisq-formats 2.0.1 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (195) hide show
  1. package/LICENSE +21 -0
  2. package/NOTICE.md +20 -0
  3. package/README.md +1 -1
  4. package/dist/{chunk-CRAVSMPZ.js → chunk-26ISNJ7Y.js} +356 -114
  5. package/dist/{chunk-NKAJPJ4G.js → chunk-2JJ5RFDZ.js} +0 -1
  6. package/dist/{chunk-HTW2M27H.js → chunk-3NKXBZSR.js} +193 -42
  7. package/dist/{chunk-U32AG3G3.js → chunk-4V3KCHAP.js} +3 -4
  8. package/dist/{chunk-MLX2BOJC.js → chunk-6RQOV3B3.js} +1 -2
  9. package/dist/{chunk-QRVN6A6E.js → chunk-6S6GU3ZG.js} +5 -6
  10. package/dist/{chunk-FE6OJV6O.js → chunk-7AWFHP5U.js} +1 -1
  11. package/dist/{chunk-RFAPOKHJ.js → chunk-AD2WT564.js} +59 -9
  12. package/dist/{chunk-O3GVVND4.js → chunk-AONELFLA.js} +0 -1
  13. package/dist/{chunk-XKUMNGBW.js → chunk-EJTNGKEA.js} +5 -8
  14. package/dist/chunk-GX7RAUME.js +121 -0
  15. package/dist/{chunk-SSUPBUF5.js → chunk-IIQYS2YH.js} +0 -1
  16. package/dist/{chunk-ABVI556T.js → chunk-IPN56VLW.js} +83 -58
  17. package/dist/{chunk-LXYLOOST.js → chunk-JE6LSIHE.js} +83 -22
  18. package/dist/{chunk-U4MRIFKL.js → chunk-JU2RHXUB.js} +0 -1
  19. package/dist/{chunk-4VUWTSGM.js → chunk-K6XRMVPW.js} +64 -31
  20. package/dist/{chunk-ODL3SSPT.js → chunk-KXOZMWBS.js} +0 -1
  21. package/dist/chunk-OGS5VCGJ.js +446 -0
  22. package/dist/{chunk-GVS2XXV6.js → chunk-PJXJI2LY.js} +449 -57
  23. package/dist/{chunk-2KPARF2P.js → chunk-PU7REGWV.js} +5 -8
  24. package/dist/{chunk-PN52A5AA.js → chunk-SBUW7NHR.js} +0 -1
  25. package/dist/{chunk-VSYHZECT.js → chunk-TAAENIRB.js} +5 -8
  26. package/dist/{chunk-WC7WULGV.js → chunk-X2DEAXNK.js} +62 -2
  27. package/dist/container/index.js +1 -2
  28. package/dist/csv/index.d.ts +27 -2
  29. package/dist/csv/index.js +1 -2
  30. package/dist/docx/index.d.ts +5 -1
  31. package/dist/docx/index.js +9 -10
  32. package/dist/epub/index.d.ts +2 -0
  33. package/dist/epub/index.js +5 -6
  34. package/dist/{export-D2NkylDT.d.ts → export-D9msROJS.d.ts} +18 -6
  35. package/dist/extract-MN7LA3NL.js +13 -0
  36. package/dist/html/index.d.ts +11 -4
  37. package/dist/html/index.js +3 -4
  38. package/dist/images-ESPQKVTW.js +6 -0
  39. package/dist/{import-K8mfc0fz.d.ts → import-C3htUTss.d.ts} +5 -1
  40. package/dist/{import-DTkDxHmZ.d.ts → import-C8whCC7_.d.ts} +6 -0
  41. package/dist/index.d.ts +7 -7
  42. package/dist/index.js +28 -26
  43. package/dist/infer/index.d.ts +3 -3
  44. package/dist/infer/index.js +7 -9
  45. package/dist/{layouts-BHrgZ5FS.d.ts → layouts-CTdPlB-u.d.ts} +1 -1
  46. package/dist/layouts-DRWZGSPD.js +10 -0
  47. package/dist/{mapTheme-IR27S6IV.js → mapTheme-4TWH25FT.js} +1 -2
  48. package/dist/ooxml/index.d.ts +3 -3
  49. package/dist/ooxml/index.js +14 -13
  50. package/dist/pdf/index.d.ts +18 -0
  51. package/dist/pdf/index.js +2 -3
  52. package/dist/pptx/index.d.ts +4 -4
  53. package/dist/pptx/index.js +11 -13
  54. package/dist/{reader-B9L8Ucbj.d.ts → reader-B_m1aKZC.d.ts} +30 -1
  55. package/dist/registry/index.d.ts +21 -5
  56. package/dist/registry/index.js +9 -6
  57. package/dist/{themeReader-DJKErl_j.d.ts → themeReader-DCtwC83Q.d.ts} +1 -1
  58. package/dist/xlsx/index.d.ts +3 -3
  59. package/dist/xlsx/index.js +6 -7
  60. package/package.json +6 -3
  61. package/dist/chunk-2KPARF2P.js.map +0 -1
  62. package/dist/chunk-4VUWTSGM.js.map +0 -1
  63. package/dist/chunk-6M7Z25LA.js +0 -46
  64. package/dist/chunk-6M7Z25LA.js.map +0 -1
  65. package/dist/chunk-ABVI556T.js.map +0 -1
  66. package/dist/chunk-CRAVSMPZ.js.map +0 -1
  67. package/dist/chunk-FE6OJV6O.js.map +0 -1
  68. package/dist/chunk-GVS2XXV6.js.map +0 -1
  69. package/dist/chunk-HTW2M27H.js.map +0 -1
  70. package/dist/chunk-LXYLOOST.js.map +0 -1
  71. package/dist/chunk-MLX2BOJC.js.map +0 -1
  72. package/dist/chunk-NKAJPJ4G.js.map +0 -1
  73. package/dist/chunk-O3GVVND4.js.map +0 -1
  74. package/dist/chunk-ODL3SSPT.js.map +0 -1
  75. package/dist/chunk-PN52A5AA.js.map +0 -1
  76. package/dist/chunk-QRVN6A6E.js.map +0 -1
  77. package/dist/chunk-RFAPOKHJ.js.map +0 -1
  78. package/dist/chunk-SSUPBUF5.js.map +0 -1
  79. package/dist/chunk-U32AG3G3.js.map +0 -1
  80. package/dist/chunk-U4MRIFKL.js.map +0 -1
  81. package/dist/chunk-VJJM2SSH.js +0 -275
  82. package/dist/chunk-VJJM2SSH.js.map +0 -1
  83. package/dist/chunk-VSYHZECT.js.map +0 -1
  84. package/dist/chunk-WC7WULGV.js.map +0 -1
  85. package/dist/chunk-XKUMNGBW.js.map +0 -1
  86. package/dist/chunk-YRT7GQ5Y.js +0 -28
  87. package/dist/chunk-YRT7GQ5Y.js.map +0 -1
  88. package/dist/container/index.js.map +0 -1
  89. package/dist/csv/index.js.map +0 -1
  90. package/dist/docx/index.js.map +0 -1
  91. package/dist/epub/index.js.map +0 -1
  92. package/dist/extract-H6RXJMHP.js +0 -15
  93. package/dist/extract-H6RXJMHP.js.map +0 -1
  94. package/dist/html/index.js.map +0 -1
  95. package/dist/images-7FBWPKE3.js +0 -7
  96. package/dist/images-7FBWPKE3.js.map +0 -1
  97. package/dist/index.js.map +0 -1
  98. package/dist/infer/index.js.map +0 -1
  99. package/dist/layouts-QVPK3ZCU.js +0 -12
  100. package/dist/layouts-QVPK3ZCU.js.map +0 -1
  101. package/dist/mapTheme-IR27S6IV.js.map +0 -1
  102. package/dist/ooxml/index.js.map +0 -1
  103. package/dist/pdf/index.js.map +0 -1
  104. package/dist/pptx/index.js.map +0 -1
  105. package/dist/registry/index.js.map +0 -1
  106. package/dist/xlsx/index.js.map +0 -1
  107. package/src/__tests__/container.test.ts +0 -230
  108. package/src/__tests__/convert.test.ts +0 -495
  109. package/src/__tests__/csvImport.test.ts +0 -84
  110. package/src/__tests__/docxExport.test.ts +0 -457
  111. package/src/__tests__/docxImport.test.ts +0 -410
  112. package/src/__tests__/epub.test.ts +0 -649
  113. package/src/__tests__/exportThemeReconciliation.test.ts +0 -87
  114. package/src/__tests__/formatRegistry.test.ts +0 -174
  115. package/src/__tests__/html.test.ts +0 -435
  116. package/src/__tests__/htmlImport.test.ts +0 -57
  117. package/src/__tests__/inferTheme.test.ts +0 -135
  118. package/src/__tests__/lossyWarnings.test.ts +0 -146
  119. package/src/__tests__/ooxml.test.ts +0 -271
  120. package/src/__tests__/ooxmlCancellation.test.ts +0 -113
  121. package/src/__tests__/ooxmlThemeReader.test.ts +0 -92
  122. package/src/__tests__/pdfExport.test.ts +0 -322
  123. package/src/__tests__/pdfImport.test.ts +0 -384
  124. package/src/__tests__/plainHtml.test.ts +0 -417
  125. package/src/__tests__/plainHtmlBundle.test.ts +0 -253
  126. package/src/__tests__/pptxExport.test.ts +0 -138
  127. package/src/__tests__/pptxImport.test.ts +0 -145
  128. package/src/__tests__/pptxInferFixtures.ts +0 -314
  129. package/src/__tests__/pptxLayoutInfer.test.ts +0 -395
  130. package/src/__tests__/roundTrip.test.ts +0 -201
  131. package/src/__tests__/roundTripAssets.test.ts +0 -50
  132. package/src/__tests__/roundTripMatrix.fixtures.ts +0 -86
  133. package/src/__tests__/roundTripMatrix.helpers.ts +0 -154
  134. package/src/__tests__/roundTripMatrix.test.ts +0 -142
  135. package/src/__tests__/sharedContainer.test.ts +0 -41
  136. package/src/__tests__/sharedImages.test.ts +0 -61
  137. package/src/__tests__/xlsxExport.test.ts +0 -164
  138. package/src/__tests__/xlsxImport.test.ts +0 -80
  139. package/src/__tests__/zipSafety.test.ts +0 -317
  140. package/src/container/index.ts +0 -94
  141. package/src/csv/index.ts +0 -188
  142. package/src/docx/export.ts +0 -1267
  143. package/src/docx/import.ts +0 -995
  144. package/src/docx/index.ts +0 -26
  145. package/src/docx/styles.ts +0 -145
  146. package/src/epub/export.ts +0 -968
  147. package/src/epub/index.ts +0 -20
  148. package/src/html/docsHtmlBundle.ts +0 -373
  149. package/src/html/htmlTemplate.ts +0 -385
  150. package/src/html/imageUtils.ts +0 -61
  151. package/src/html/import.ts +0 -297
  152. package/src/html/index.ts +0 -212
  153. package/src/html/plainHtml.ts +0 -790
  154. package/src/html/plainHtmlBundle.ts +0 -421
  155. package/src/index.ts +0 -109
  156. package/src/infer/extract.ts +0 -127
  157. package/src/infer/index.ts +0 -199
  158. package/src/infer/mapTheme.ts +0 -176
  159. package/src/infer/types.ts +0 -27
  160. package/src/ooxml/index.ts +0 -111
  161. package/src/ooxml/namespaces.ts +0 -196
  162. package/src/ooxml/readUtils.ts +0 -44
  163. package/src/ooxml/reader.ts +0 -318
  164. package/src/ooxml/themeReader.ts +0 -197
  165. package/src/ooxml/types.ts +0 -103
  166. package/src/ooxml/writer.ts +0 -339
  167. package/src/ooxml/xmlUtils.ts +0 -123
  168. package/src/pdf/export.ts +0 -1084
  169. package/src/pdf/import.ts +0 -1164
  170. package/src/pdf/index.ts +0 -29
  171. package/src/pdf/styles.ts +0 -180
  172. package/src/pptx/export.ts +0 -1184
  173. package/src/pptx/import.ts +0 -455
  174. package/src/pptx/index.ts +0 -52
  175. package/src/pptx/layouts.ts +0 -1222
  176. package/src/pptx/styles.ts +0 -96
  177. package/src/pptx/templates.ts +0 -187
  178. package/src/registry/convert.ts +0 -433
  179. package/src/registry/defaultFormats.ts +0 -413
  180. package/src/registry/errors.ts +0 -46
  181. package/src/registry/index.ts +0 -43
  182. package/src/registry/registry.ts +0 -48
  183. package/src/registry/types.ts +0 -170
  184. package/src/shared/boundedZipArchive.ts +0 -383
  185. package/src/shared/container.ts +0 -28
  186. package/src/shared/fidelity.ts +0 -130
  187. package/src/shared/images.ts +0 -44
  188. package/src/shared/inlineRuns.ts +0 -99
  189. package/src/shared/text.ts +0 -41
  190. package/src/shared/zipEntryCount.ts +0 -151
  191. package/src/shared/zipLimits.ts +0 -296
  192. package/src/shared/zipSafety.ts +0 -19
  193. package/src/xlsx/export.ts +0 -253
  194. package/src/xlsx/import.ts +0 -160
  195. package/src/xlsx/index.ts +0 -35
@@ -1,421 +0,0 @@
1
- /**
2
- * Recursive plain-HTML bundle export.
3
- *
4
- * `markdownDocsToPlainHtmlBundle` starts from a single entry markdown
5
- * file, walks its relative `[…](other.md)` links, recursively pulls in
6
- * any sibling/child documents (scope-limited to the entry doc's
7
- * directory tree), renders every visited file via
8
- * `markdownDocToPlainHtml`, rewrites cross-doc references from `.md`
9
- * to `.html`, and ships everything as a single ZIP.
10
- *
11
- * The function is provider-agnostic — callers pass `readDocument` and
12
- * `readBinary` callbacks that resolve relative paths against whatever
13
- * storage they use (`FileSystemContentContainer`, `MemoryContent-
14
- * Container`, in-memory map for tests, …). Failure to read any
15
- * discovered file aborts the whole export with a thrown error.
16
- */
17
-
18
- import JSZip from 'jszip';
19
- import { parseMarkdown, inferDocumentTitle } from '@bendyline/squisq/markdown';
20
- import type { MarkdownDocument, HtmlNode } from '@bendyline/squisq/markdown';
21
- import type { Theme, ThemeRegistry } from '@bendyline/squisq/schemas';
22
- import { markdownDocToPlainHtml } from './plainHtml.js';
23
-
24
- // ── Public Types ───────────────────────────────────────────────────
25
-
26
- export interface PlainHtmlBundleOptions {
27
- /** Entry document path relative to the container root (e.g. `'home.md'`). */
28
- entryPath: string;
29
- /** Reads a UTF-8 markdown file from the container. Returns null when absent. */
30
- readDocument: (path: string) => Promise<string | null>;
31
- /** Reads a binary asset (image) from the container. Returns null when absent. */
32
- readBinary: (path: string) => Promise<ArrayBuffer | null>;
33
- /** Optional document title for the entry. Others derive from filename. */
34
- title?: string;
35
- /** Optional theme applied uniformly to every page. Overrides {@link themeId}. */
36
- theme?: Theme;
37
- /**
38
- * Optional theme id (e.g. `'warm-earth'`, `'gezellig'`) applied to every
39
- * page. Each document resolves its own inline definition before consulting
40
- * {@link themeRegistry}. When both `theme` and `themeId` are supplied,
41
- * `theme` wins.
42
- */
43
- themeId?: string;
44
- /** Explicit caller-owned registry for non-document custom themes. */
45
- themeRegistry?: ThemeRegistry;
46
- /** Maximum recursion depth (default: unlimited; cycles always handled). */
47
- maxDepth?: number;
48
- /**
49
- * Emit the entry doc as `index.html` (preserving its parent directory)
50
- * instead of `<basename>.html`. Cross-doc links pointing at the entry
51
- * also rewrite to `index.html`, so a sibling `resume.md → home.md`
52
- * link doesn't 404 after the rename. Convenient for static-site
53
- * deploys where the landing page must be named `index.html`.
54
- * Default: false.
55
- */
56
- entryAsIndex?: boolean;
57
- }
58
-
59
- // ── Public API ─────────────────────────────────────────────────────
60
-
61
- /**
62
- * Render an entry markdown document and every reachable sibling/child
63
- * `.md` document it links to, bundled as a single ZIP with plain-HTML
64
- * pages, per-document asset folders, and cross-doc `<a href>`
65
- * references rewritten from `.md` to `.html`.
66
- */
67
- export async function markdownDocsToPlainHtmlBundle(
68
- options: PlainHtmlBundleOptions,
69
- ): Promise<Blob> {
70
- const {
71
- entryPath,
72
- readDocument,
73
- readBinary,
74
- title,
75
- theme,
76
- themeId,
77
- themeRegistry,
78
- maxDepth = Infinity,
79
- entryAsIndex = false,
80
- } = options;
81
- const entry = normalizePath(entryPath);
82
- if (!entry) {
83
- throw new Error('markdownDocsToPlainHtmlBundle: entryPath is required');
84
- }
85
- const scopeRoot = posixDirname(entry); // '' for root-level files
86
- // When `entryAsIndex`, the entry doc writes to `<entryDir>/index.html`
87
- // and every cross-doc link pointing at it rewrites to that path too.
88
- // Other docs keep the `<basename>.html` convention.
89
- const entryHtmlPath = entryAsIndex
90
- ? scopeRoot
91
- ? `${scopeRoot}/index.html`
92
- : 'index.html'
93
- : entry.slice(0, -3) + '.html';
94
- const htmlPathFor = (mdPath: string): string =>
95
- mdPath === entry ? entryHtmlPath : mdPath.slice(0, -3) + '.html';
96
-
97
- const zip = new JSZip();
98
- const visited = new Set<string>();
99
- const queue: Array<{ path: string; depth: number }> = [{ path: entry, depth: 0 }];
100
-
101
- while (queue.length > 0) {
102
- const { path, depth } = queue.shift()!;
103
- if (visited.has(path)) continue;
104
- visited.add(path);
105
-
106
- const source = await readDocument(path);
107
- if (source === null) {
108
- throw new Error(`markdownDocsToPlainHtmlBundle: failed to read "${path}"`);
109
- }
110
- const mdDoc = parseMarkdown(source);
111
-
112
- // Discover this doc's relative .md links — both for enqueueing new
113
- // targets and for building the per-doc link rewrite map. We resolve
114
- // each link to a canonical container-relative path so the same doc
115
- // referenced two different ways (./resume.md vs resume.md) is
116
- // visited once and rewritten consistently.
117
- const docDir = posixDirname(path);
118
- const linkMap = new Map<string, string>();
119
- for (const raw of collectLinkRefs(mdDoc)) {
120
- const parsed = parseLinkRef(raw);
121
- if (!parsed) continue;
122
- const resolved = resolveRelative(docDir, parsed.path);
123
- if (resolved === null) continue; // escaped via ..
124
- if (!isInScope(resolved, scopeRoot)) continue;
125
- if (!resolved.toLowerCase().endsWith('.md')) continue;
126
-
127
- // Compute the .html replacement relative to the current doc so
128
- // the rewritten href stays local (e.g. `subdir/notes.html`, not
129
- // an absolute container path). `htmlPathFor` also handles the
130
- // entry-as-index case so a sibling linking to the entry lands
131
- // on `index.html` after rename.
132
- const htmlTarget = htmlPathFor(resolved);
133
- const relHref = relativeFrom(docDir, htmlTarget) + parsed.fragment;
134
- linkMap.set(raw, relHref);
135
-
136
- if (depth + 1 <= maxDepth && !visited.has(resolved)) {
137
- queue.push({ path: resolved, depth: depth + 1 });
138
- }
139
- }
140
-
141
- // Walk images, fetch each one, write at the original relative path
142
- // resolved against the container root (so `resume_files/hero.png`
143
- // sits next to `resume.html` in the zip). Failures here are *not*
144
- // fatal — images can be missing without breaking the document
145
- // structure — but we surface them as console warnings.
146
- const images = await readImagesForDoc(mdDoc, docDir, readBinary);
147
- for (const [, { data, zipPath }] of images) {
148
- const safe = sanitizeZipPath(zipPath);
149
- if (!safe) continue;
150
- zip.file(safe, data);
151
- }
152
-
153
- // Render the document. The image rewrite map keys are the URLs
154
- // exactly as authored (so `markdownDocToPlainHtml` can match them
155
- // verbatim), and the values are paths relative to the rendered
156
- // doc's own location — so `<img src>` works after unzip without
157
- // any post-processing.
158
- const imageRewriteMap = new Map<string, string>();
159
- for (const [authored, { zipPath }] of images) {
160
- imageRewriteMap.set(authored, relativeFrom(docDir, zipPath));
161
- }
162
-
163
- const docTitle = depth === 0 ? title : undefined;
164
- const html = markdownDocToPlainHtml(mdDoc, {
165
- title: docTitle ?? titleForFilename(path, mdDoc),
166
- images: imageRewriteMap,
167
- links: linkMap,
168
- theme,
169
- themeId,
170
- themeRegistry,
171
- });
172
-
173
- const htmlPath = htmlPathFor(path);
174
- zip.file(htmlPath, html);
175
- }
176
-
177
- return zip.generateAsync({
178
- type: 'blob',
179
- compression: 'DEFLATE',
180
- compressionOptions: { level: 6 },
181
- });
182
- }
183
-
184
- // ── Link discovery ─────────────────────────────────────────────────
185
-
186
- /**
187
- * Collect every `<a>`-style link URL referenced in a document. Markdown
188
- * `link` nodes plus any raw HTML `<a href>` tags. Returns the raw URLs
189
- * as authored, so callers can use them as both the linkMap *key* and
190
- * the basis for resolution.
191
- */
192
- export function collectLinkRefs(doc: MarkdownDocument): Set<string> {
193
- const refs = new Set<string>();
194
-
195
- function visitHtml(nodes: HtmlNode[]): void {
196
- for (const n of nodes) {
197
- if (n.type !== 'htmlElement') continue;
198
- if (n.tagName.toLowerCase() === 'a') {
199
- const href = n.attributes.href;
200
- if (typeof href === 'string' && href) refs.add(href);
201
- }
202
- visitHtml(n.children);
203
- }
204
- }
205
-
206
- function visit(node: unknown): void {
207
- if (!node || typeof node !== 'object') return;
208
- const n = node as Record<string, unknown>;
209
- if (n.type === 'link' && typeof n.url === 'string' && n.url) {
210
- refs.add(n.url);
211
- }
212
- if ((n.type === 'htmlBlock' || n.type === 'htmlInline') && Array.isArray(n.htmlChildren)) {
213
- visitHtml(n.htmlChildren as HtmlNode[]);
214
- }
215
- if (Array.isArray(n.children)) {
216
- for (const child of n.children) visit(child);
217
- }
218
- }
219
-
220
- for (const child of doc.children) visit(child);
221
- return refs;
222
- }
223
-
224
- function collectImageRefs(doc: MarkdownDocument): Set<string> {
225
- const refs = new Set<string>();
226
- function visitHtml(nodes: HtmlNode[]): void {
227
- for (const n of nodes) {
228
- if (n.type !== 'htmlElement') continue;
229
- const tag = n.tagName.toLowerCase();
230
- // <img>/<video>/<audio>/<source> all reference media via `src`;
231
- // we feed them into the same `images` map (effectively a generic
232
- // media map — see header comment) so the export pipeline rewrites
233
- // and bundles each one the same way.
234
- if (tag === 'img' || tag === 'video' || tag === 'audio' || tag === 'source') {
235
- const src = n.attributes.src;
236
- if (typeof src === 'string' && src) refs.add(src);
237
- }
238
- if (tag === 'video' || tag === 'audio') {
239
- const poster = n.attributes.poster;
240
- if (typeof poster === 'string' && poster) refs.add(poster);
241
- }
242
- visitHtml(n.children);
243
- }
244
- }
245
- function visit(node: unknown): void {
246
- if (!node || typeof node !== 'object') return;
247
- const n = node as Record<string, unknown>;
248
- if (n.type === 'image' && typeof n.url === 'string' && n.url) refs.add(n.url);
249
- if ((n.type === 'htmlBlock' || n.type === 'htmlInline') && Array.isArray(n.htmlChildren)) {
250
- visitHtml(n.htmlChildren as HtmlNode[]);
251
- }
252
- if (Array.isArray(n.children)) for (const c of n.children) visit(c);
253
- }
254
- for (const child of doc.children) visit(child);
255
- return refs;
256
- }
257
-
258
- interface ParsedLinkRef {
259
- /** Pathname portion before any `#` or `?`. */
260
- path: string;
261
- /** `#fragment` suffix (with leading `#`) or empty string. */
262
- fragment: string;
263
- }
264
-
265
- /**
266
- * Split an authored URL into path + fragment, returning null for
267
- * external / non-document references (http(s), mailto, data, blob,
268
- * absolute paths, fragment-only). The fragment is preserved so a
269
- * link like `resume.md#experience` rewrites cleanly to
270
- * `resume.html#experience`.
271
- */
272
- function parseLinkRef(url: string): ParsedLinkRef | null {
273
- if (!url) return null;
274
- if (url.startsWith('#')) return null; // intra-doc anchor
275
- if (
276
- /^[a-z][a-z0-9+.-]*:/.test(url) || // http:, https:, mailto:, ftp:, …
277
- url.startsWith('//') ||
278
- url.startsWith('/')
279
- ) {
280
- return null;
281
- }
282
- const hashIdx = url.indexOf('#');
283
- const queryIdx = url.indexOf('?');
284
- const cut =
285
- hashIdx >= 0 && queryIdx >= 0 ? Math.min(hashIdx, queryIdx) : hashIdx >= 0 ? hashIdx : queryIdx;
286
- const path = cut >= 0 ? url.slice(0, cut) : url;
287
- const fragment = hashIdx >= 0 ? url.slice(hashIdx) : '';
288
- return { path, fragment };
289
- }
290
-
291
- // ── Image gathering ────────────────────────────────────────────────
292
-
293
- /**
294
- * Resolve every image referenced by a document to container-relative
295
- * paths and fetch their bytes. The returned map keys are the authored
296
- * URLs (so the renderer's `images` map can substitute them); the values
297
- * are tuples of `[bytes, zipPath]` where `zipPath` is the path used
298
- * inside the zip.
299
- *
300
- * Images that can't be read are silently dropped — the rendered HTML
301
- * will keep its `<img src>` pointing at the authored URL, which a
302
- * reader's browser will 404. That matches the user-facing "abort on
303
- * missing linked doc" rule which applies to `.md` links only.
304
- */
305
- async function readImagesForDoc(
306
- mdDoc: MarkdownDocument,
307
- docDir: string,
308
- readBinary: (path: string) => Promise<ArrayBuffer | null>,
309
- ): Promise<Map<string, { data: ArrayBuffer; zipPath: string }>> {
310
- const out = new Map<string, { data: ArrayBuffer; zipPath: string }>();
311
- for (const authored of collectImageRefs(mdDoc)) {
312
- if (/^[a-z][a-z0-9+.-]*:/.test(authored) || authored.startsWith('//')) continue;
313
- const cleanAuthored = authored.replace(/[#?].*$/, '');
314
- const resolved = resolveRelative(docDir, cleanAuthored);
315
- if (resolved === null) continue;
316
- const data = await readBinary(resolved);
317
- if (!data) continue;
318
- out.set(authored, { data, zipPath: resolved });
319
- }
320
- return out;
321
- }
322
-
323
- // ── Path helpers (POSIX-style, no Node `path`) ─────────────────────
324
-
325
- /**
326
- * Strip a path to its parent directory (POSIX-style). Returns `''`
327
- * for root-level files so subsequent joins read like a fresh path.
328
- */
329
- export function posixDirname(p: string): string {
330
- const idx = p.lastIndexOf('/');
331
- return idx < 0 ? '' : p.slice(0, idx);
332
- }
333
-
334
- /**
335
- * Normalize a POSIX path: collapse `./`, resolve `..`, drop trailing
336
- * slashes, never produce a leading `/`. Returns null when the path
337
- * escapes the start ("..").
338
- */
339
- export function normalizePath(p: string): string | null {
340
- const parts = p.split('/');
341
- const out: string[] = [];
342
- for (const segment of parts) {
343
- if (segment === '' || segment === '.') continue;
344
- if (segment === '..') {
345
- if (out.length === 0) return null;
346
- out.pop();
347
- continue;
348
- }
349
- out.push(segment);
350
- }
351
- return out.join('/');
352
- }
353
-
354
- /**
355
- * Resolve `rel` against `baseDir`. Returns null when the result
356
- * escapes the container root via `..`.
357
- */
358
- export function resolveRelative(baseDir: string, rel: string): string | null {
359
- if (rel.startsWith('/')) return null; // absolute paths are out of scope
360
- const joined = baseDir ? `${baseDir}/${rel}` : rel;
361
- return normalizePath(joined);
362
- }
363
-
364
- /**
365
- * True when `target` is inside `root` (or equal to it). Empty root
366
- * means "entire container is in scope".
367
- */
368
- export function isInScope(target: string, root: string): boolean {
369
- if (!root) return true;
370
- return target === root || target.startsWith(root + '/');
371
- }
372
-
373
- /**
374
- * Compute a relative path from `fromDir` to `toPath` (both POSIX-
375
- * normalized, container-relative). Used to emit hrefs that work after
376
- * unzipping anywhere — `home.md → subfolder/notes.md` becomes
377
- * `subfolder/notes.html`; `subfolder/intro.md → resume.md` becomes
378
- * `../resume.html`.
379
- */
380
- export function relativeFrom(fromDir: string, toPath: string): string {
381
- const fromParts = fromDir ? fromDir.split('/') : [];
382
- const toParts = toPath.split('/');
383
- let common = 0;
384
- while (
385
- common < fromParts.length &&
386
- common < toParts.length - 1 &&
387
- fromParts[common] === toParts[common]
388
- ) {
389
- common++;
390
- }
391
- const up = fromParts.length - common;
392
- const down = toParts.slice(common);
393
- const prefix = Array(up).fill('..').join('/');
394
- if (!prefix) return down.join('/');
395
- return `${prefix}/${down.join('/')}`;
396
- }
397
-
398
- /**
399
- * Sanitize a path for use as a zip entry. Strips leading slashes,
400
- * normalizes backslashes (Windows-authored paths), and rejects any
401
- * path containing `..` segments after normalization (defensive — the
402
- * resolver above already filters those out for links).
403
- */
404
- function sanitizeZipPath(path: string): string | null {
405
- const normalized = path.replace(/\\/g, '/').replace(/^\/+/, '');
406
- if (!normalized) return null;
407
- if (normalized.split('/').some((seg) => seg === '..')) return null;
408
- return normalized;
409
- }
410
-
411
- /**
412
- * Derive a default page title for a non-entry doc: prefer frontmatter
413
- * `title`, fall back to the shallowest heading text, then the filename
414
- * without extension.
415
- */
416
- function titleForFilename(path: string, mdDoc: MarkdownDocument): string {
417
- const inferred = inferDocumentTitle(mdDoc);
418
- if (inferred) return inferred;
419
- const base = path.split('/').pop() ?? path;
420
- return base.replace(/\.md$/i, '');
421
- }
package/src/index.ts DELETED
@@ -1,109 +0,0 @@
1
- /**
2
- * @bendyline/squisq-formats
3
- *
4
- * Format converters for squisq documents. Converts between squisq's
5
- * MarkdownDocument / Doc and various file formats via Office Open XML.
6
- *
7
- * Supported formats:
8
- * - **DOCX** — Microsoft Word (import + export) ✅
9
- * - **PDF** — Portable Document Format (import + export) ✅
10
- * - **PPTX** — Microsoft PowerPoint (import + export ✅; import extracts slide-level embedded images)
11
- * - **XLSX** — Microsoft Excel (import ✅, export ✅ tables-only)
12
- * - **CSV** — Comma-separated values (import ✅, export ✅)
13
- * - **HTML** — import + export (single-file, ZIP, plain, and player-embedding)
14
- * - **EPUB** — export ✅
15
- *
16
- * All converters run in the browser — no server or native binaries required.
17
- * The shared `ooxml/` subpath export provides reusable OOXML infrastructure, and the
18
- * `registry` subpath exposes a format registry + `convert()` pipeline over all of the above.
19
- *
20
- * @example
21
- * ```ts
22
- * // Import from root
23
- * import { markdownDocToDocx, docxToMarkdownDoc } from '@bendyline/squisq-formats';
24
- *
25
- * // Or import from subpath
26
- * import { markdownDocToDocx } from '@bendyline/squisq-formats/docx';
27
- * import { createPackage } from '@bendyline/squisq-formats/ooxml';
28
- * ```
29
- */
30
-
31
- // DOCX (fully implemented)
32
- export { markdownDocToDocx, docToDocx, docxToMarkdownDoc, docxToDoc } from './docx/index.js';
33
- export type { DocxExportOptions, DocxImportOptions } from './docx/index.js';
34
-
35
- // PPTX (import + export)
36
- export { markdownDocToPptx, docToPptx, pptxToMarkdownDoc, pptxToDoc } from './pptx/index.js';
37
- export type { PptxExportOptions, PptxImportOptions } from './pptx/index.js';
38
-
39
- // XLSX (import + export; export is tables-only → one worksheet per markdown table)
40
- export { markdownDocToXlsx, docToXlsx, xlsxToMarkdownDoc, xlsxToDoc } from './xlsx/index.js';
41
- export type { XlsxExportOptions, XlsxImportOptions } from './xlsx/index.js';
42
-
43
- // CSV (import + export)
44
- export { csvToMarkdownDoc, csvToDoc, markdownDocToCsv, parseCsv } from './csv/index.js';
45
- export type { CsvImportOptions, CsvExportOptions } from './csv/index.js';
46
-
47
- // PDF (fully implemented)
48
- export {
49
- markdownDocToPdf,
50
- docToPdf,
51
- pdfToMarkdownDoc,
52
- pdfToDoc,
53
- configurePdfWorker,
54
- } from './pdf/index.js';
55
- export type { PdfExportOptions, PdfImportOptions } from './pdf/index.js';
56
-
57
- // HTML (fully implemented)
58
- export { docToHtml, docToHtmlZip, collectImagePaths } from './html/index.js';
59
- export type { HtmlExportOptions, HtmlZipExportOptions } from './html/index.js';
60
- export { htmlToMarkdown, htmlToMarkdownDoc, htmlToMarkdownDocSync } from './html/index.js';
61
- export type { HtmlImportOptions } from './html/index.js';
62
-
63
- // EPUB (export)
64
- export { markdownDocToEpub, docToEpub } from './epub/index.js';
65
- export type { EpubExportOptions } from './epub/index.js';
66
-
67
- // Theme inference from file imports (DOCX/PPTX/XLSX theme1.xml → Squisq Theme)
68
- export { inferThemeFromFile, compileExtractedTheme } from './infer/index.js';
69
- export type {
70
- InferThemeOptions,
71
- InferredFileTheme,
72
- ExtractedFileTheme,
73
- InferSourceFormat,
74
- } from './infer/index.js';
75
-
76
- // Format registry + programmatic convert()
77
- export {
78
- convert,
79
- prepareConversion,
80
- createRegistry,
81
- defaultRegistry,
82
- defaultFormats,
83
- ConversionError,
84
- BUILTIN_FORMAT_IDS,
85
- } from './registry/index.js';
86
-
87
- // Shared bounded-decompression errors/options used by DBK and OOXML imports.
88
- export { ZipSafetyError } from './shared/zipSafety.js';
89
- export type {
90
- ZipSafetyLimits,
91
- ZipSafetyErrorCode,
92
- ZipSafetyErrorOptions,
93
- } from './shared/zipSafety.js';
94
- export type {
95
- FormatId,
96
- ConversionResult,
97
- NormalizedInput,
98
- ConvertOptions,
99
- FormatDefinition,
100
- FormatRegistry,
101
- ConvertSource,
102
- ConversionErrorCode,
103
- ConversionErrorOptions,
104
- BuiltinFormatOptions,
105
- MarkdownFormatOptions,
106
- DbkFormatOptions,
107
- PreparedConversion,
108
- PreparedExportOptions,
109
- } from './registry/index.js';
@@ -1,127 +0,0 @@
1
- /**
2
- * Per-format theme extraction: locate the theme part of an already-opened
3
- * OOXML package, parse it, and resolve the document's background/text color
4
- * mapping. Returns null when the file has no theme part at all.
5
- */
6
-
7
- import type { OoxmlPackage } from '../ooxml/types.js';
8
- import { getPartRelationships, getPartXml } from '../ooxml/reader.js';
9
- import { CONTENT_TYPE_PPTX_THEME, NS_PML, NS_WML, REL_SLIDE_MASTER } from '../ooxml/namespaces.js';
10
- import { findRelByType, resolveTarget } from '../ooxml/readUtils.js';
11
- import type { OoxmlTheme } from '../ooxml/themeReader.js';
12
- import { parseThemeXml, readThemePart } from '../ooxml/themeReader.js';
13
- import type { ExtractedFileTheme, SchemeSlot } from './types.js';
14
-
15
- const DEFAULT_COLOR_MAP: { bg1: SchemeSlot; tx1: SchemeSlot } = { bg1: 'lt1', tx1: 'dk1' };
16
-
17
- const SCHEME_SLOTS: readonly SchemeSlot[] = ['dk1', 'lt1', 'dk2', 'lt2'];
18
-
19
- function asSchemeSlot(value: string | null | undefined): SchemeSlot | undefined {
20
- return SCHEME_SLOTS.includes(value as SchemeSlot) ? (value as SchemeSlot) : undefined;
21
- }
22
-
23
- /**
24
- * Fallback theme lookup for packages whose theme relationship is missing or
25
- * unconventional: scan `[Content_Types].xml` overrides for the theme content
26
- * type (shared by all three formats) and parse the first match.
27
- */
28
- async function readThemeByContentType(pkg: OoxmlPackage): Promise<OoxmlTheme | null> {
29
- for (const [partPath, contentType] of pkg.contentTypes.overrides) {
30
- if (contentType !== CONTENT_TYPE_PPTX_THEME) continue;
31
- const doc = await getPartXml(pkg, partPath);
32
- if (doc) return parseThemeXml(doc);
33
- }
34
- return null;
35
- }
36
-
37
- function toExtracted(
38
- sourceFormat: ExtractedFileTheme['sourceFormat'],
39
- theme: OoxmlTheme,
40
- colorMap: { bg1: SchemeSlot; tx1: SchemeSlot },
41
- extraWarnings: string[] = [],
42
- ): ExtractedFileTheme {
43
- return {
44
- sourceFormat,
45
- ...(theme.name ? { themeName: theme.name } : {}),
46
- ...(theme.colors ? { colors: theme.colors } : {}),
47
- colorMap,
48
- ...(theme.fonts ? { fonts: theme.fonts } : {}),
49
- warnings: [...theme.warnings, ...extraWarnings],
50
- };
51
- }
52
-
53
- /**
54
- * DOCX: theme hangs off `word/document.xml`; an optional
55
- * `w:clrSchemeMapping` in `word/settings.xml` remaps bg1/t1 slots
56
- * (values `light1`/`dark1`/`light2`/`dark2`).
57
- */
58
- export async function extractDocxTheme(pkg: OoxmlPackage): Promise<ExtractedFileTheme | null> {
59
- const theme =
60
- (await readThemePart(pkg, 'word/document.xml')) ?? (await readThemeByContentType(pkg));
61
- if (!theme) return null;
62
-
63
- let colorMap = DEFAULT_COLOR_MAP;
64
- const settings = await getPartXml(pkg, 'word/settings.xml');
65
- const mappingEls = settings?.getElementsByTagNameNS(NS_WML, 'clrSchemeMapping');
66
- const mapping = mappingEls && mappingEls.length > 0 ? mappingEls[0]! : null;
67
- if (mapping) {
68
- const wordSlotToScheme: Record<string, SchemeSlot> = {
69
- light1: 'lt1',
70
- dark1: 'dk1',
71
- light2: 'lt2',
72
- dark2: 'dk2',
73
- };
74
- const bg1 =
75
- wordSlotToScheme[
76
- mapping.getAttributeNS(NS_WML, 'bg1') ?? mapping.getAttribute('w:bg1') ?? ''
77
- ];
78
- const tx1 =
79
- wordSlotToScheme[mapping.getAttributeNS(NS_WML, 't1') ?? mapping.getAttribute('w:t1') ?? ''];
80
- if (bg1 && tx1) colorMap = { bg1, tx1 };
81
- }
82
-
83
- return toExtracted('docx', theme, colorMap);
84
- }
85
-
86
- /**
87
- * PPTX: theme hangs off the first slide master, whose `<p:clrMap>` records
88
- * which scheme slots back the deck's background/text (dark decks set
89
- * `bg1="dk1" tx1="lt1"`).
90
- */
91
- export async function extractPptxTheme(pkg: OoxmlPackage): Promise<ExtractedFileTheme | null> {
92
- const presRels = await getPartRelationships(pkg, 'ppt/presentation.xml');
93
- const masterRel = findRelByType(presRels, REL_SLIDE_MASTER);
94
- const masterPath = masterRel ? resolveTarget('ppt', masterRel.target) : undefined;
95
-
96
- const warnings: string[] = [];
97
- let theme: OoxmlTheme | null = null;
98
- let colorMap = DEFAULT_COLOR_MAP;
99
-
100
- if (masterPath) {
101
- theme = await readThemePart(pkg, masterPath);
102
- const masterDoc = await getPartXml(pkg, masterPath);
103
- const clrMaps = masterDoc?.getElementsByTagNameNS(NS_PML, 'clrMap');
104
- const clrMap = clrMaps && clrMaps.length > 0 ? clrMaps[0]! : null;
105
- if (clrMap) {
106
- const bg1 = asSchemeSlot(clrMap.getAttribute('bg1'));
107
- const tx1 = asSchemeSlot(clrMap.getAttribute('tx1'));
108
- if (bg1 && tx1) {
109
- colorMap = { bg1, tx1 };
110
- } else if (clrMap.getAttribute('bg1') || clrMap.getAttribute('tx1')) {
111
- warnings.push('theme: unsupported clrMap slot mapping; using default bg1/tx1');
112
- }
113
- }
114
- }
115
- if (!theme) theme = await readThemeByContentType(pkg);
116
- if (!theme) return null;
117
-
118
- return toExtracted('pptx', theme, colorMap, warnings);
119
- }
120
-
121
- /** XLSX: theme hangs off `xl/workbook.xml`; no color remapping exists. */
122
- export async function extractXlsxTheme(pkg: OoxmlPackage): Promise<ExtractedFileTheme | null> {
123
- const theme =
124
- (await readThemePart(pkg, 'xl/workbook.xml')) ?? (await readThemeByContentType(pkg));
125
- if (!theme) return null;
126
- return toExtracted('xlsx', theme, DEFAULT_COLOR_MAP);
127
- }